Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
841437fa06 | ||
|
|
9c9a0024d4 | ||
|
|
a6c2d4459c | ||
|
|
da45fa3fb0 |
+16
-15
@@ -2,32 +2,33 @@
|
|||||||
|
|
||||||
## 项目
|
## 项目
|
||||||
|
|
||||||
| 日期 | slug | 领域 | 关键词 | 状态 |
|
| 日期 | slug | 领域 | 关键词 | 关联 OpenSpec | 状态 |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|---|---|
|
||||||
| 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented |
|
| 2026-07-09 | interview-demo-quality-audit | Agent eval/demo/Prompt audit | interview demo preflight, prompt_audit, gatekeeper rules, diagnosis baseline, 12 fixtures | openspec/changes/archive/2026-07-09-interview-demo-quality-audit | archived |
|
||||||
|
| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived |
|
||||||
|
| 2026-07-08 | diagnosis-eval-demo-gatekeeper-closure | Agent eval/demo/Gatekeeper | diagnosis eval matrix, stable demo scenarios, Gatekeeper rule set version, audit metadata | openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure | archived |
|
||||||
|
| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived |
|
||||||
| 2026-07-07 | executor-evidence-output-contract | Chat质量门禁/证据归因 | Executor structured output, evidence bindings, Verifier structured claims, LOW_CONFID, hallucination | openspec/changes/archive/2026-07-07-executor-evidence-output-contract | archived |
|
| 2026-07-07 | executor-evidence-output-contract | Chat质量门禁/证据归因 | Executor structured output, evidence bindings, Verifier structured claims, LOW_CONFID, hallucination | openspec/changes/archive/2026-07-07-executor-evidence-output-contract | archived |
|
||||||
| 2026-07-07 | executor-v2-output-contract | Chat质量门禁/证据归因 | executor_evidence_v2, user_facing_answer removal, diagnosis_summary removal, structured renderer | openspec/changes/archive/2026-07-07-executor-v2-output-contract | archived |
|
| 2026-07-07 | executor-v2-output-contract | Chat质量门禁/证据归因 | executor_evidence_v2, user_facing_answer removal, diagnosis_summary removal, structured renderer | openspec/changes/archive/2026-07-07-executor-v2-output-contract | archived |
|
||||||
| 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived |
|
| 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived |
|
||||||
| 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived |
|
| 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived |
|
||||||
| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived |
|
|
||||||
| 2026-07-08 | diagnosis-eval-demo-gatekeeper-closure | Agent eval/demo/Gatekeeper | diagnosis eval matrix, stable demo scenarios, Gatekeeper rule set version, audit metadata | openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure | archived |
|
|
||||||
| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived |
|
|
||||||
| 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived |
|
| 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived |
|
||||||
| 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived |
|
| 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived |
|
||||||
|
| 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented |
|
||||||
| 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived |
|
| 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived |
|
||||||
| 2026-07-05 | diagnosis-eval-baseline-diff | Agent 评测/回归 Diff | baseline diff, regression detection, evidence coverage, cost signal, markdown report | openspec/changes/archive/2026-07-05-diagnosis-eval-baseline-diff | archived |
|
| 2026-07-05 | diagnosis-eval-baseline-diff | Agent 评测/回归 Diff | baseline diff, regression detection, evidence coverage, cost signal, markdown report | openspec/changes/archive/2026-07-05-diagnosis-eval-baseline-diff | archived |
|
||||||
| 2026-07-04 | expand-diagnosis-eval-fixtures | Agent 评测/回归 Baseline | fixture coverage, baseline report, redis timeout, slow response, jvm memory risk | openspec/changes/archive/2026-07-05-expand-diagnosis-eval-fixtures | archived |
|
| 2026-07-04 | expand-diagnosis-eval-fixtures | Agent 评测/回归 Baseline | fixture coverage, baseline report, redis timeout, slow response, jvm memory risk | openspec/changes/archive/2026-07-05-expand-diagnosis-eval-fixtures | archived |
|
||||||
| 2026-07-04 | diagnosis-eval-harness | Agent 评测/回归 Harness | fixed cases, trace validation, evidence coverage, verdict distribution, markdown report | openspec/changes/archive/2026-07-04-diagnosis-eval-harness | archived |
|
| 2026-07-04 | diagnosis-eval-harness | Agent 评测/回归 Harness | fixed cases, trace validation, evidence coverage, verdict distribution, markdown report | openspec/changes/archive/2026-07-04-diagnosis-eval-harness | archived |
|
||||||
| 2026-07-04 | evidence-trace-hardening | 证据链/降级契约/离线验证 | ToolInvocationRecorder, ToolTraceSummaryService, lookup_knowledge, query_logs, query_metrics, LOW_CONFID, REJECT | openspec/changes/archive/2026-07-04-evidence-trace-hardening | archived |
|
| 2026-07-04 | evidence-trace-hardening | 证据链/降级契约/离线验证 | ToolInvocationRecorder, ToolTraceSummaryService, lookup_knowledge, query_logs, query_metrics, LOW_CONFID, REJECT | openspec/changes/archive/2026-07-04-evidence-trace-hardening | archived |
|
||||||
| 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived |
|
|
||||||
| 2026-07-04 | aiops-traceable-diagnosis-entry | AIOps/trace/alert diagnosis | ai_ops, SSE, alert input, sessionId, diagnosis_session, trace API | openspec/changes/archive/2026-07-04-aiops-traceable-diagnosis-entry | archived |
|
| 2026-07-04 | aiops-traceable-diagnosis-entry | AIOps/trace/alert diagnosis | ai_ops, SSE, alert input, sessionId, diagnosis_session, trace API | openspec/changes/archive/2026-07-04-aiops-traceable-diagnosis-entry | archived |
|
||||||
| 2026-07-04 | aiops-alert-scope-control | AIOps/scope/prompt control | payload mode, auto-discovery mode, queryPrometheusAlerts, HighCPUUsage | openspec/changes/archive/2026-07-04-aiops-alert-scope-control | archived |
|
| 2026-07-04 | aiops-alert-scope-control | AIOps/scope/prompt control | payload mode, auto-discovery mode, queryPrometheusAlerts, HighCPUUsage | openspec/changes/archive/2026-07-04-aiops-alert-scope-control | archived |
|
||||||
| 2026-05-29 | chatmodel-abstraction | 解耦/多模型路由 | ChatModel, EmbeddingModel, DeepSeek, BGE-M3, SiliconFlow, Spring AI | archived |
|
| 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived |
|
||||||
| 2026-06-23 | phase1-infrastructure | 基础设施/文档管理 | MySQL, Redis, Milvus, Flyway, JPA, 向量检索, 类别过滤 | archived |
|
|
||||||
| 2026-06-24 | lookup-knowledge-integration | 知识库检索 | L0精确匹配, L1语义检索, frontmatter, 混合检索 | archived |
|
|
||||||
| 2026-06-25 | doc-management-ui | 前端开发/文档管理 | 文档管理页面, CRUD, 状态监控, 纯静态页面, API集成 | archived |
|
|
||||||
| 2026-06-26 | session-storage | 会话存储/可观测 | diagnosis_session, agent_step, tool_invocation, token追踪, 多Agent路由 | openspec/changes/session-storage | archived |
|
|
||||||
| 2026-06-29 | confidence-feedback | 质量评估/反馈机制 | evidence_score, selfEvaluation, feedback, useful, not_useful, case_library, BAD_CASE, tool_invocation规则引擎, 反馈按钮, sessionId回传 | openspec/changes/confidence-feedback | archived |
|
|
||||||
| 2026-06-30 | session-dedup-knowledge-map | 去重/知识图谱 | RetrievedDocTracker, KnowledgeDomainService, knowledge_domain, covers, whenToRetrieve, Planner注入, ISS-001 | openspec/changes/archive/2026-06-30-session-dedup-knowledge-map | archived |
|
|
||||||
| 2026-07-01 | executor-action-memory-relevance | 检索质量/行动记忆 | relevanceLevel, completenessHint, Min-Max归一化, RetrievedDocTracker域级记录, Executor检索约束, ISS-002 | openspec/changes/archive/2026-07-01-executor-action-memory-relevance | archived |
|
|
||||||
| 2026-07-02 | chat-verifier-agent | Chat质量门禁/可追溯验证 | Verifier, groundedness_score, facts_checked, evidence_refs, tool_trace_summary, self_evaluation | openspec/changes/archive/2026-07-03-chat-verifier-agent | archived |
|
| 2026-07-02 | chat-verifier-agent | Chat质量门禁/可追溯验证 | Verifier, groundedness_score, facts_checked, evidence_refs, tool_trace_summary, self_evaluation | openspec/changes/archive/2026-07-03-chat-verifier-agent | archived |
|
||||||
|
| 2026-07-01 | executor-action-memory-relevance | 检索质量/行动记忆 | relevanceLevel, completenessHint, Min-Max归一化, RetrievedDocTracker域级记录, Executor检索约束, ISS-002 | openspec/changes/archive/2026-07-01-executor-action-memory-relevance | archived |
|
||||||
|
| 2026-06-30 | session-dedup-knowledge-map | 去重/知识图谱 | RetrievedDocTracker, KnowledgeDomainService, knowledge_domain, covers, whenToRetrieve, Planner注入, ISS-001 | openspec/changes/archive/2026-06-30-session-dedup-knowledge-map | archived |
|
||||||
|
| 2026-06-29 | confidence-feedback | 质量评估/反馈机制 | evidence_score, selfEvaluation, feedback, useful, not_useful, case_library, BAD_CASE, tool_invocation规则引擎, 反馈按钮, sessionId回传 | openspec/changes/confidence-feedback | archived |
|
||||||
|
| 2026-06-26 | session-storage | 会话存储/可观测 | diagnosis_session, agent_step, tool_invocation, token追踪, 多Agent路由 | openspec/changes/session-storage | archived |
|
||||||
|
| 2026-06-25 | doc-management-ui | 前端开发/文档管理 | 文档管理页面, CRUD, 状态监控, 纯静态页面, API集成 | archived |
|
||||||
|
| 2026-06-24 | lookup-knowledge-integration | 知识库检索 | L0精确匹配, L1语义检索, frontmatter, 混合检索 | archived |
|
||||||
|
| 2026-06-23 | phase1-infrastructure | 基础设施/文档管理 | MySQL, Redis, Milvus, Flyway, JPA, 向量检索, 类别过滤 | archived |
|
||||||
|
| 2026-05-29 | chatmodel-abstraction | 解耦/多模型路由 | ChatModel, EmbeddingModel, DeepSeek, BGE-M3, SiliconFlow, Spring AI | archived |
|
||||||
|
|||||||
@@ -48,7 +48,7 @@
|
|||||||
|
|
||||||
## 遗留问题
|
## 遗留问题
|
||||||
|
|
||||||
ISS-002:Executor 无约束重复调用 `lookup_knowledge`(单会话 20+ 次),knowledge map 和检索约束只注入了 Planner 未注入 Executor。详见 `mvp/issues/ISS-002-executor-unconstrained-lookup.md`。
|
ISS-002:Executor 无约束重复调用 `lookup_knowledge`(单会话 20+ 次),knowledge map 和检索约束只注入了 Planner 未注入 Executor。详见 `mvp/issues/archived/ISS-002-executor-unconstrained-lookup.md`。
|
||||||
|
|
||||||
## 已知限制
|
## 已知限制
|
||||||
|
|
||||||
|
|||||||
@@ -9,8 +9,8 @@
|
|||||||
## Context
|
## Context
|
||||||
|
|
||||||
- `devflow/index.md` was checked. Relevant history includes `session-storage`, `confidence-feedback`, `executor-action-memory-relevance`, and `chat-verifier-agent`.
|
- `devflow/index.md` was checked. Relevant history includes `session-storage`, `confidence-feedback`, `executor-action-memory-relevance`, and `chat-verifier-agent`.
|
||||||
- `mvp/notes/agent-engineering-decisions.md` already recommends the next phase as "可复现 MVP Demo", including `mvp-demo` profile, fixed diagnosis case, one-click request, and `GET /api/diagnosis/{sessionId}/trace`.
|
- `mvp/archive/2026-07-09-doc-cleanup/notes/agent-engineering-decisions.md` already recommends the next phase as "可复现 MVP Demo", including `mvp-demo` profile, fixed diagnosis case, one-click request, and `GET /api/diagnosis/{sessionId}/trace`.
|
||||||
- `mvp/issues/ISS-003-mvp-design-implementation-review.md` identifies test stability, session traceability, verifier evidence chain, upload path, and SupervisorAgent consistency as recent MVP concerns. Security cleanup is intentionally deferred by user decision.
|
- `mvp/issues/active/ISS-003-mvp-design-implementation-review.md` identifies test stability, session traceability, verifier evidence chain, upload path, and SupervisorAgent consistency as recent MVP concerns. Security cleanup is intentionally deferred by user decision.
|
||||||
|
|
||||||
## Question Pool
|
## Question Pool
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
## Draft Acceptance
|
## Draft Acceptance
|
||||||
|
|
||||||
- [x] Issue exists: `mvp/issues/executor-evidence-attribution-hallucination.md`.
|
- [x] Issue exists: `mvp/issues/active/executor-evidence-attribution-hallucination.md`.
|
||||||
- [x] OpenSpec change artifacts exist.
|
- [x] OpenSpec change artifacts exist.
|
||||||
- [x] devflow tracking files exist.
|
- [x] devflow tracking files exist.
|
||||||
- [x] OpenSpec validation passes.
|
- [x] OpenSpec validation passes.
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
- `devflow/projects/2026-07-07-executor-evidence-output-contract`: V1 evidence-attribution contract kept `user_facing_answer`.
|
- `devflow/projects/2026-07-07-executor-evidence-output-contract`: V1 evidence-attribution contract kept `user_facing_answer`.
|
||||||
- `devflow/projects/2026-07-02-chat-verifier-agent`: Verifier consumes explicit inputs and should not see intermediate reasoning.
|
- `devflow/projects/2026-07-02-chat-verifier-agent`: Verifier consumes explicit inputs and should not see intermediate reasoning.
|
||||||
- `devflow/projects/2026-07-04-evidence-trace-hardening`: evidence summaries and tool invocation references are the evidence foundation.
|
- `devflow/projects/2026-07-04-evidence-trace-hardening`: evidence summaries and tool invocation references are the evidence foundation.
|
||||||
- `mvp/issues/executor-structured-output-v2.md`: staged implementation design; stage one is Executor V2 output contract.
|
- `mvp/issues/design-notes/executor-structured-output-v2.md`: staged implementation design; stage one is Executor V2 output contract.
|
||||||
|
|
||||||
## Code Evidence
|
## Code Evidence
|
||||||
|
|
||||||
|
|||||||
@@ -41,4 +41,4 @@ Make Executor cite concrete tool evidence, make Gatekeeper validate that citatio
|
|||||||
## OpenSpec
|
## OpenSpec
|
||||||
|
|
||||||
- Change: `openspec/changes/verifier-evidence-reference-fidelity`
|
- Change: `openspec/changes/verifier-evidence-reference-fidelity`
|
||||||
- Source issue: `mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md`
|
- Source issue: `mvp/issues/archived/ISS-007-verifier-evidence-summary-fidelity.md`
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
# Acceptance
|
||||||
|
|
||||||
|
## Static Verification
|
||||||
|
|
||||||
|
- `openspec validate interview-demo-quality-audit --strict`
|
||||||
|
- Result: passed.
|
||||||
|
- Coverage: OpenSpec proposal/design/spec/tasks consistency.
|
||||||
|
- PowerShell parser/runtime readiness check:
|
||||||
|
- Command: `powershell -NoProfile -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 -BaseUrl http://127.0.0.1:1 -OutputDir target/demo-check-syntax`
|
||||||
|
- Result: expected failure with actionable readiness message.
|
||||||
|
- Coverage: script parses under Windows PowerShell and fails before issuing diagnosis requests when service is unreachable.
|
||||||
|
|
||||||
|
## Script Verification
|
||||||
|
|
||||||
|
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`
|
||||||
|
- Result: passed.
|
||||||
|
- Coverage: 12/12 fixed eval fixtures, Prompt audit evaluator checks, Gatekeeper rule metadata checks, regenerated baseline reports.
|
||||||
|
- `mvn -q "-Dtest=ChatServiceSequentialAgentTest" test`
|
||||||
|
- Result: passed.
|
||||||
|
- Coverage: Chat verifier evaluation persists `prompt_audit`.
|
||||||
|
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ChatServiceSequentialAgentTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`
|
||||||
|
- Result: passed.
|
||||||
|
- Coverage: broader eval, baseline diff, Chat sequential flow, Gatekeeper, and Verifier input hook regression set.
|
||||||
|
- `mvn -q -DskipTests compile`
|
||||||
|
- Result: passed.
|
||||||
|
- Coverage: main source compilation.
|
||||||
|
|
||||||
|
## E2E Verification
|
||||||
|
|
||||||
|
- Started service with:
|
||||||
|
- `mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo`
|
||||||
|
- Ran:
|
||||||
|
- `powershell -NoProfile -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 -BaseUrl http://localhost:9900 -SessionId mvp-demo-interview-quality-audit-001`
|
||||||
|
- Result: passed.
|
||||||
|
- Summary:
|
||||||
|
- `chatSuccess=true`
|
||||||
|
- `verdict=LOW_CONFID`
|
||||||
|
- `gatekeeperStatus=fail`
|
||||||
|
- `gatekeeperRuleSetVersion=gatekeeper-rules-v1`
|
||||||
|
- `promptAuditVersion=chat-prompts-v1`
|
||||||
|
- tools included `lookup_knowledge`, `query_logs`, `query_metrics`, and `get_available_log_topics`
|
||||||
|
- Note: live E2E remains a compatibility check, not the deterministic PASS oracle. The fixed fixture baseline is the regression source of truth.
|
||||||
|
|
||||||
|
## Not Verified
|
||||||
|
|
||||||
|
- Browser UI inspection was not required for this change because the scope is backend trace/eval/demo script documentation, not frontend behavior.
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
# Interview Demo Quality Audit Brief
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The MVP already demonstrates traceable Agent diagnosis with Planner, Executor, Gatekeeper, Verifier, Composer, evidence tools, trace persistence, and deterministic eval fixtures. The remaining interview-readiness gap is not a new Agent architecture; it is making the demo easier to run and making prompt/rule changes easier to audit.
|
||||||
|
|
||||||
|
## Goal
|
||||||
|
|
||||||
|
Stabilize the interview demo path, expand fixture-backed evaluation, and persist prompt/Gatekeeper audit metadata so the project can explain and verify Agent behavior during interviews.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Add prompt audit metadata to Chat verifier evaluation.
|
||||||
|
- Extend deterministic eval cases and baseline reports.
|
||||||
|
- Add an interview demo preflight/check script.
|
||||||
|
- Update MVP demo and architecture documentation.
|
||||||
|
|
||||||
|
## Non-goals
|
||||||
|
|
||||||
|
- No public API or database schema changes.
|
||||||
|
- No new SubAgent split, MCP migration, process isolation, or AIOps LLM Verifier.
|
||||||
|
- No guarantee that every live LLM run returns PASS.
|
||||||
|
|
||||||
@@ -0,0 +1,111 @@
|
|||||||
|
# interview-demo-quality-audit Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: stabilize the interview demo, expand deterministic eval coverage, and add Prompt/Gatekeeper version audit.
|
||||||
|
- Slug: `interview-demo-quality-audit`.
|
||||||
|
- Devflow scale: `standard`.
|
||||||
|
- Interface impact: L2 internal contract change because `verifier_evaluation` gains `prompt_audit`; no public HTTP API or database schema change.
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- `devflow/index.md` used: related entries found for `diagnosis-eval-demo-gatekeeper-closure`, `executor-composer-final-answer`, `verifier-evidence-reference-fidelity`, `mvp-demo-interview-runbook`, and `diagnosis-eval-baseline-diff`.
|
||||||
|
- Relevant glossary:
|
||||||
|
- Evidence Tools produce incident facts and must be recorded in `tool_invocation`.
|
||||||
|
- Verifier should not use skills/runbooks as incident evidence.
|
||||||
|
- Diagnosis Playbook Skill is workflow guidance, not a fact source.
|
||||||
|
- Historical constraints that must enter OpenSpec:
|
||||||
|
- Diagnosis eval is deterministic and fixture-backed; no LLM-as-judge.
|
||||||
|
- Stable demo scenarios are documentation/payloads plus deterministic fixtures; live E2E is a compatibility check, not a guaranteed PASS oracle.
|
||||||
|
- Gatekeeper rule metadata is already metadata-only and should not become dynamic rule execution.
|
||||||
|
- Composer is the final expression layer and must not leak raw Executor JSON.
|
||||||
|
|
||||||
|
## Question Pool
|
||||||
|
|
||||||
|
| ID | Dimension | Mode | Question | Status |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Q1 | Terminology | evidence-driven | What should the new audit metadata be called? | Resolved |
|
||||||
|
| Q2 | Boundary | evidence-driven | Does this require public API or schema changes? | Resolved |
|
||||||
|
| Q3 | Acceptance | evidence-driven | Which current assets define deterministic acceptance? | Resolved |
|
||||||
|
| Q4 | Technical | evidence-driven | Where should prompt version metadata live with minimal implementation risk? | Resolved |
|
||||||
|
| Q5 | Scope | user-interview | Should live E2E be mandatory for all scenarios? | Confirmed by objective as conditional |
|
||||||
|
|
||||||
|
## Evidence-driven Conclusions
|
||||||
|
|
||||||
|
- Q1 conclusion: use `prompt_audit` for prompt version metadata and keep existing `gatekeeper_result.rule_set_version`.
|
||||||
|
- Q2 conclusion: keep this as an internal trace/self-evaluation contract change. Do not add endpoints, tables, or new Agent roles.
|
||||||
|
- Q3 conclusion: `DiagnosisTraceEvaluatorTest`, baseline reports, fixed fixtures, and demo scripts define current acceptance style.
|
||||||
|
- Q4 conclusion: add a small Chat prompt audit catalog near `ChatService` prompt loading and persist a compact snapshot with verifier evaluation.
|
||||||
|
- Q5 conclusion: run live E2E with `mvp-demo` profile if dependencies are available; otherwise record the blocker and rely on deterministic eval/unit evidence.
|
||||||
|
|
||||||
|
## Specify / Alignment
|
||||||
|
|
||||||
|
| Check | Status | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| proposal goals -> proposal | Aligned | Proposal covers demo preflight, eval expansion, prompt audit, and docs. |
|
||||||
|
| proposal scope/constraints -> design | Aligned | Design records no public API/schema changes, prompt audit shape, eval fields, and demo script behavior. |
|
||||||
|
| design decisions -> specs/tasks | Aligned | Specs cover persisted prompt audit, evaluator checks, baseline, and demo script outputs. |
|
||||||
|
| specs observable behavior -> tasks | Aligned | Each requirement has implementation and verification tasks. |
|
||||||
|
|
||||||
|
## Audit
|
||||||
|
|
||||||
|
Input -> processing -> output chain:
|
||||||
|
|
||||||
|
```text
|
||||||
|
prompt resource metadata
|
||||||
|
-> ChatService / PromptAudit snapshot
|
||||||
|
-> verifier_evaluation.prompt_audit
|
||||||
|
-> Trace API / eval fixtures
|
||||||
|
-> DiagnosisTraceEvaluator baseline
|
||||||
|
|
||||||
|
run-interview-demo-check.ps1
|
||||||
|
-> service readiness
|
||||||
|
-> chat / trace / feedback
|
||||||
|
-> mvp/demo/output summary
|
||||||
|
```
|
||||||
|
|
||||||
|
Architecture risk assessment:
|
||||||
|
|
||||||
|
1. The audit shape is intentionally compact and internal; storing full prompt text would create noisy traces and possible sensitive-content risk.
|
||||||
|
2. Eval should assert versions by explicit metadata, not by prompt content hashes that churn during local prompt edits.
|
||||||
|
3. Live demo checks may still be LOW_CONFID because LLM output is not deterministic; deterministic fixtures remain the regression source of truth.
|
||||||
|
4. No devflow/OpenSpec conflict found.
|
||||||
|
|
||||||
|
## Commit Gate
|
||||||
|
|
||||||
|
- `openspec validate interview-demo-quality-audit --strict`: passed.
|
||||||
|
- File completeness:
|
||||||
|
- `proposal.md`: present.
|
||||||
|
- `design.md`: present.
|
||||||
|
- `specs/`: present for `chat-verifier-agent`, `diagnosis-eval-harness`, and `mvp-demo-trace-acceptance`.
|
||||||
|
- `tasks.md`: present.
|
||||||
|
- Consistency:
|
||||||
|
- Proposal goals map to design sections.
|
||||||
|
- Design decisions map to spec requirements and executable tasks.
|
||||||
|
- Task acceptance checks are verifiable.
|
||||||
|
- `.committed` marker created.
|
||||||
|
|
||||||
|
## Current Checkpoint
|
||||||
|
|
||||||
|
- Commit completed.
|
||||||
|
- Apply is authorized by the original objective: "完成后归档提交".
|
||||||
|
|
||||||
|
## Pre-apply Research
|
||||||
|
|
||||||
|
- Capability source: sm-flow built-in apply protocol. `openspec-apply-change` was not invoked directly in this session.
|
||||||
|
- Repository semantic search/LSP note: the requested `codebase-retrieval` and LSP tools were not available in the exposed toolset, so impact analysis used `rg`, direct file reads, OpenSpec/devflow artifacts, and targeted tests.
|
||||||
|
- Reference implementation and reuse:
|
||||||
|
- `ChatService.persistVerifierEvaluation(...)` is the single persistence point for Chat verifier/composer audit data; prompt audit was added there to cover normal, fallback, and degraded Composer paths.
|
||||||
|
- `DiagnosisTraceEvaluator` and `DiagnosisEvalReportWriter` are the deterministic eval extension points; no LLM judge was introduced.
|
||||||
|
- `mvp/demo/scripts/run-payment-timeout-demo.ps1` provided the request/trace/feedback flow reused by the new interview preflight script.
|
||||||
|
- Interface impact remains L2 internal trace contract: `verifier_evaluation.prompt_audit` and eval report fields are added; no public endpoint, table, or request DTO changed.
|
||||||
|
|
||||||
|
## Apply Notes
|
||||||
|
|
||||||
|
- Added compact Chat prompt audit metadata: `chat-prompts-v1`, with planner/executor/verifier/composer prompt versions and resource paths.
|
||||||
|
- Extended diagnosis eval schema, result reporting, baseline fixtures, JSON report, and Markdown report for Prompt audit and Gatekeeper rule metadata.
|
||||||
|
- Added two fixture-backed audit cases:
|
||||||
|
- `prompt-gatekeeper-audit-closure`
|
||||||
|
- `audit-metadata-low-confid`
|
||||||
|
- Added `mvp/demo/scripts/run-interview-demo-check.ps1` to run service readiness, Chat, Trace, feedback, and summary output.
|
||||||
|
- Updated MVP demo/eval/architecture docs to explain `prompt_audit.version`, `gatekeeper_result.rule_set_version`, and deterministic fixture baseline.
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
# Evidence
|
||||||
|
|
||||||
|
## Context Files Read
|
||||||
|
|
||||||
|
- `devflow/index.md`
|
||||||
|
- `devflow/glossary/CONTEXT.md`
|
||||||
|
- `devflow/projects/2026-07-08-diagnosis-eval-demo-gatekeeper-closure/decisions.md`
|
||||||
|
- `devflow/projects/2026-07-08-executor-composer-final-answer/decisions.md`
|
||||||
|
- `mvp/architecture/current-mvp-architecture.md`
|
||||||
|
- `mvp/architecture/agent-orchestration.md`
|
||||||
|
- `mvp/architecture/executor-evidence-pipeline-refactor.md`
|
||||||
|
- `mvp/architecture/harness-quality-gates.md`
|
||||||
|
- `mvp/demo/README.md`
|
||||||
|
- `mvp/demo/ten-minute-interview-demo.md`
|
||||||
|
- `mvp/eval/README.md`
|
||||||
|
- `mvp/eval/cases/diagnosis-cases.json`
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ChatService.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ExecutorGatekeeperService.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java`
|
||||||
|
- `src/main/resources/gatekeeper/gatekeeper-rules.json`
|
||||||
|
|
||||||
|
## Tooling Note
|
||||||
|
|
||||||
|
The required `codebase-retrieval` and LSP tools were not exposed in this session. Impact analysis used `rg`, direct file reads, existing OpenSpec/devflow artifacts, and targeted tests instead.
|
||||||
|
|
||||||
|
## Implementation Evidence
|
||||||
|
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ChatService.java`
|
||||||
|
- Adds `prompt_audit` under `verifier_evaluation` through the shared `persistVerifierEvaluation(...)` path.
|
||||||
|
- Uses compact metadata only: audit version, prompt names, prompt versions, and resource paths.
|
||||||
|
- `src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java`
|
||||||
|
- Adds deterministic checks for `requirePromptAudit`, `expectedPromptAuditVersion`, `expectedPromptVersions`, and `requireGatekeeperRules`.
|
||||||
|
- `src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java`
|
||||||
|
- Adds Prompt Audit and Gatekeeper rule count columns to Markdown reports.
|
||||||
|
- `mvp/eval/cases/diagnosis-cases.json`
|
||||||
|
- Expands fixed baseline to 12 fixture-backed cases.
|
||||||
|
- `mvp/eval/fixtures/prompt-gatekeeper-audit-closure-pass.json`
|
||||||
|
- Positive PASS fixture proving Prompt audit and Gatekeeper rule metadata closure.
|
||||||
|
- `mvp/eval/fixtures/audit-metadata-low-confid.json`
|
||||||
|
- LOW_CONFID fixture proving safe answer behavior while audit metadata remains present.
|
||||||
|
- `mvp/demo/scripts/run-interview-demo-check.ps1`
|
||||||
|
- Adds service readiness, Chat, Trace, feedback, and summary output for interview preflight.
|
||||||
|
|
||||||
|
## Verification Evidence
|
||||||
|
|
||||||
|
- OpenSpec:
|
||||||
|
- `openspec validate interview-demo-quality-audit --strict`: passed before archive.
|
||||||
|
- `openspec validate --specs --strict`: 10 specs passed after merging deltas into main specs.
|
||||||
|
- Unit/eval:
|
||||||
|
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`: passed.
|
||||||
|
- `mvn -q "-Dtest=ChatServiceSequentialAgentTest" test`: passed.
|
||||||
|
- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ChatServiceSequentialAgentTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`: passed.
|
||||||
|
- Compile:
|
||||||
|
- `mvn -q -DskipTests compile`: passed.
|
||||||
|
- E2E:
|
||||||
|
- Started `mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo`.
|
||||||
|
- Ran `mvp/demo/scripts/run-interview-demo-check.ps1` against `http://localhost:9900`.
|
||||||
|
- Summary recorded `chatSuccess=true`, `verdict=LOW_CONFID`, `gatekeeperRuleSetVersion=gatekeeper-rules-v1`, and `promptAuditVersion=chat-prompts-v1`.
|
||||||
+22
-15
@@ -1,8 +1,8 @@
|
|||||||
# SuperBizAgent MVP 文档
|
# SuperBizAgent MVP 文档
|
||||||
|
|
||||||
**更新日期**:2026-07-05
|
**更新日期**:2026-07-09
|
||||||
|
|
||||||
本目录保存 MVP 阶段的架构、问题、演示、评测和数据表说明。当前架构入口已经整理到 `mvp/architecture/`,旧版架构材料已归档,避免继续把历史方案当成当前实现。
|
本目录保存 MVP 阶段的架构、问题、演示、评测和数据表说明。当前材料按“当前入口”和“历史归档”拆开,避免把早期设计稿当成当前实现。
|
||||||
|
|
||||||
## 当前入口
|
## 当前入口
|
||||||
|
|
||||||
@@ -12,6 +12,7 @@
|
|||||||
| [architecture/current-mvp-architecture.md](architecture/current-mvp-architecture.md) | 当前可运行系统架构 |
|
| [architecture/current-mvp-architecture.md](architecture/current-mvp-architecture.md) | 当前可运行系统架构 |
|
||||||
| [architecture/interview-one-pager.md](architecture/interview-one-pager.md) | 面试一页式架构讲解 |
|
| [architecture/interview-one-pager.md](architecture/interview-one-pager.md) | 面试一页式架构讲解 |
|
||||||
| [architecture/agent-orchestration.md](architecture/agent-orchestration.md) | Agent 编排架构 |
|
| [architecture/agent-orchestration.md](architecture/agent-orchestration.md) | Agent 编排架构 |
|
||||||
|
| [architecture/executor-evidence-pipeline-refactor.md](architecture/executor-evidence-pipeline-refactor.md) | Executor 证据链路改造记录 |
|
||||||
| [architecture/harness-quality-gates.md](architecture/harness-quality-gates.md) | Harness 与质量门禁 |
|
| [architecture/harness-quality-gates.md](architecture/harness-quality-gates.md) | Harness 与质量门禁 |
|
||||||
| [architecture/rag-architecture.md](architecture/rag-architecture.md) | RAG/知识检索新架构 |
|
| [architecture/rag-architecture.md](architecture/rag-architecture.md) | RAG/知识检索新架构 |
|
||||||
| [architecture/retrieval-observability.md](architecture/retrieval-observability.md) | 检索与可观测性架构 |
|
| [architecture/retrieval-observability.md](architecture/retrieval-observability.md) | 检索与可观测性架构 |
|
||||||
@@ -20,11 +21,12 @@
|
|||||||
| [architecture/knowledge-base-authoring.md](architecture/knowledge-base-authoring.md) | 知识库文档编写与维护 |
|
| [architecture/knowledge-base-authoring.md](architecture/knowledge-base-authoring.md) | 知识库文档编写与维护 |
|
||||||
| [architecture/data-model.md](architecture/data-model.md) | 数据模型总览 |
|
| [architecture/data-model.md](architecture/data-model.md) | 数据模型总览 |
|
||||||
| [architecture/evolution-roadmap.md](architecture/evolution-roadmap.md) | Agent 架构演进路线 |
|
| [architecture/evolution-roadmap.md](architecture/evolution-roadmap.md) | Agent 架构演进路线 |
|
||||||
| [issues/rag-refactor-plan.md](issues/rag-refactor-plan.md) | RAG 重构计划和阶段拆解 |
|
| [issues/README.md](issues/README.md) | MVP issue 索引 |
|
||||||
|
| [issues/active/rag-refactor-plan.md](issues/active/rag-refactor-plan.md) | RAG 重构计划和阶段拆解 |
|
||||||
|
| [tables/README.md](tables/README.md) | 当前 MySQL 表说明 |
|
||||||
| [demo/README.md](demo/README.md) | Demo 运行和面试演示材料 |
|
| [demo/README.md](demo/README.md) | Demo 运行和面试演示材料 |
|
||||||
| [demo/ten-minute-interview-demo.md](demo/ten-minute-interview-demo.md) | 10 分钟面试演示脚本 |
|
| [demo/ten-minute-interview-demo.md](demo/ten-minute-interview-demo.md) | 10 分钟面试演示脚本 |
|
||||||
| [eval/README.md](eval/README.md) | 诊断评测材料 |
|
| [eval/README.md](eval/README.md) | 诊断评测材料 |
|
||||||
| [issues/README.md](issues/README.md) | MVP issue 索引 |
|
|
||||||
|
|
||||||
## 当前系统一句话
|
## 当前系统一句话
|
||||||
|
|
||||||
@@ -39,6 +41,7 @@ mvp/
|
|||||||
current-mvp-architecture.md
|
current-mvp-architecture.md
|
||||||
interview-one-pager.md
|
interview-one-pager.md
|
||||||
agent-orchestration.md
|
agent-orchestration.md
|
||||||
|
executor-evidence-pipeline-refactor.md
|
||||||
harness-quality-gates.md
|
harness-quality-gates.md
|
||||||
rag-architecture.md
|
rag-architecture.md
|
||||||
retrieval-observability.md
|
retrieval-observability.md
|
||||||
@@ -47,12 +50,17 @@ mvp/
|
|||||||
knowledge-base-authoring.md
|
knowledge-base-authoring.md
|
||||||
data-model.md
|
data-model.md
|
||||||
evolution-roadmap.md
|
evolution-roadmap.md
|
||||||
archive/2026-07-05-legacy/
|
archive/
|
||||||
issues/
|
issues/
|
||||||
README.md
|
README.md
|
||||||
rag-refactor-plan.md
|
active/
|
||||||
ISS-*.md
|
archived/
|
||||||
rag-*.md
|
design-notes/
|
||||||
|
rag/
|
||||||
|
tables/
|
||||||
|
README.md
|
||||||
|
*表-*.md
|
||||||
|
archive/
|
||||||
demo/
|
demo/
|
||||||
README.md
|
README.md
|
||||||
ten-minute-interview-demo.md
|
ten-minute-interview-demo.md
|
||||||
@@ -65,9 +73,7 @@ mvp/
|
|||||||
cases/
|
cases/
|
||||||
fixtures/
|
fixtures/
|
||||||
reports/
|
reports/
|
||||||
notes/
|
archive/
|
||||||
plan/
|
|
||||||
tables/
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## 当前核心设计
|
## 当前核心设计
|
||||||
@@ -107,10 +113,11 @@ RAG
|
|||||||
-> tool_invocation
|
-> tool_invocation
|
||||||
```
|
```
|
||||||
|
|
||||||
## 旧文档说明
|
## 归档说明
|
||||||
|
|
||||||
旧版架构文档已移动到:
|
历史材料分两类:
|
||||||
|
|
||||||
- [architecture/archive/2026-07-05-legacy/](architecture/archive/2026-07-05-legacy/)
|
- 旧架构文档:[architecture/archive/2026-07-05-legacy/](architecture/archive/2026-07-05-legacy/)
|
||||||
|
- 本次文档清理归档:[archive/2026-07-09-doc-cleanup/](archive/2026-07-09-doc-cleanup/)
|
||||||
|
|
||||||
归档文档只用于追溯设计历史。当前实现和后续规划以 `architecture/current-mvp-architecture.md` 与 `architecture/rag-architecture.md` 为准。
|
归档文档只用于追溯设计历史。当前实现和后续规划以 `architecture/`、`issues/README.md`、`tables/README.md` 和 OpenSpec/devflow 的最新记录为准。
|
||||||
|
|||||||
@@ -154,4 +154,4 @@ ALTER TABLE diagnosis_session ADD COLUMN answer LONGTEXT COMMENT 'Agent 返回
|
|||||||
|
|
||||||
- **LLM 观点层**:在 `selfEvaluation` 的 `llm_opinion` 字段叠加 LLM 结构化观点(has_root_cause、has_solution 等),作为独立 factors,不改变现有规则逻辑
|
- **LLM 观点层**:在 `selfEvaluation` 的 `llm_opinion` 字段叠加 LLM 结构化观点(has_root_cause、has_solution 等),作为独立 factors,不改变现有规则逻辑
|
||||||
- **案例结构化字段**:useful 触发时自动提取 faultCategory / errorCode,替代暂时的 GENERAL
|
- **案例结构化字段**:useful 触发时自动提取 faultCategory / errorCode,替代暂时的 GENERAL
|
||||||
- **重复召回问题**:Executor Prompt 约束或工具层 session 维度去重(见 [ISS-001](../issues/ISS-001-duplicate-retrieval.md))
|
- **重复召回问题**:Executor Prompt 约束或工具层 session 维度去重(见 [ISS-001](../../../issues/archived/ISS-001-duplicate-retrieval.md))
|
||||||
|
|||||||
@@ -82,6 +82,23 @@ Prompt 层当前承担的门禁:
|
|||||||
- Chat Composer 不允许补事实,尤其不能把 `$.no_evidence` 表达为“已排除/确认没有”。
|
- Chat Composer 不允许补事实,尤其不能把 `$.no_evidence` 表达为“已排除/确认没有”。
|
||||||
- AIOps payload 模式必须聚焦输入告警。
|
- AIOps payload 模式必须聚焦输入告警。
|
||||||
|
|
||||||
|
Chat 链路还会在 `verifier_evaluation.prompt_audit` 中持久化紧凑 Prompt 审计快照:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"version": "chat-prompts-v1",
|
||||||
|
"prompts": [
|
||||||
|
{
|
||||||
|
"name": "chat_executor",
|
||||||
|
"version": "chat-executor-v2",
|
||||||
|
"resource": "prompts/chat-executor-prompt.md"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
该快照只保存版本和资源路径,不保存完整 Prompt 文本。它用于面试演示、trace 回放和离线 baseline 解释“本次诊断使用了哪套 Prompt 契约”。
|
||||||
|
|
||||||
## 4. Trace Hooks
|
## 4. Trace Hooks
|
||||||
|
|
||||||
`AgentLoggingHook` 是当前 Agent step 可观测性的核心。
|
`AgentLoggingHook` 是当前 Agent step 可观测性的核心。
|
||||||
@@ -196,7 +213,7 @@ Verifier 不再逐字核验 excerpt 真伪;这由 Gatekeeper 完成。Verifier
|
|||||||
diagnosis_session.self_evaluation.verifier_evaluation
|
diagnosis_session.self_evaluation.verifier_evaluation
|
||||||
```
|
```
|
||||||
|
|
||||||
其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary` 和 `composer_output`,用于 Trace 回放。
|
其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary`、`prompt_audit` 和 `composer_output`,用于 Trace 回放。
|
||||||
|
|
||||||
## 7. AIOps 规则门禁
|
## 7. AIOps 规则门禁
|
||||||
|
|
||||||
@@ -233,7 +250,7 @@ diagnosis_session.self_evaluation.aiops_rule_evaluation
|
|||||||
- 同一工具调用次数上限。
|
- 同一工具调用次数上限。
|
||||||
- 工具超时的统一熔断。
|
- 工具超时的统一熔断。
|
||||||
- Gatekeeper 规则远程化或三层分离:索引层、元数据层、规则实现层。
|
- Gatekeeper 规则远程化或三层分离:索引层、元数据层、规则实现层。
|
||||||
- Prompt 版本记录和回滚。
|
- Prompt 版本回滚和更细粒度变更审计。
|
||||||
- Verifier 对 AIOps 报告的 LLM 级事实校验。
|
- Verifier 对 AIOps 报告的 LLM 级事实校验。
|
||||||
|
|
||||||
这些应在评测集扩大后逐步加入,避免一次性把诊断流程卡得过死。
|
这些应在评测集扩大后逐步加入,避免一次性把诊断流程卡得过死。
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
**更新日期**:2026-07-06
|
**更新日期**:2026-07-06
|
||||||
**状态**:当前主架构 + 后续演进边界
|
**状态**:当前主架构 + 后续演进边界
|
||||||
**关联计划**:`mvp/issues/rag-refactor-plan.md`
|
**关联计划**:[`mvp/issues/active/rag-refactor-plan.md`](../issues/active/rag-refactor-plan.md)
|
||||||
|
|
||||||
## 1. 架构目标
|
## 1. 架构目标
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,17 @@
|
|||||||
|
# 2026-07-09 MVP 文档清理归档
|
||||||
|
|
||||||
|
本目录保存本次清理中从当前入口移出的历史设计材料。这些文档仍有追溯价值,但不再代表当前可运行实现。
|
||||||
|
|
||||||
|
## 归档内容
|
||||||
|
|
||||||
|
| 目录 | 内容 | 归档原因 |
|
||||||
|
|---|---|---|
|
||||||
|
| `discuss/` | 早期 Executor Prompt、L0、RAG 讨论稿 | 已被当前 architecture、OpenSpec change 和 issue 取代 |
|
||||||
|
| `plan/` | `session-storage-design.md` | 会话存储已实现,当前表以 Flyway 和 `mvp/tables/` 为准 |
|
||||||
|
| `notes/` | 早期工程决策和 Demo Trace 验收笔记 | 相关内容已沉淀到 architecture、demo、eval 和 devflow |
|
||||||
|
|
||||||
|
## 使用原则
|
||||||
|
|
||||||
|
- 当前架构以 `mvp/architecture/` 为准。
|
||||||
|
- 当前表结构以 `mvp/tables/`、Flyway migration 和实体类为准。
|
||||||
|
- 当前问题入口以 `mvp/issues/README.md` 为准。
|
||||||
+8
-3
@@ -8,7 +8,9 @@
|
|||||||
- `interview-walkthrough.md`:面试讲解话术。
|
- `interview-walkthrough.md`:面试讲解话术。
|
||||||
- `evidence-pipeline-scenarios.md`:PASS / LOW_CONFID / REJECT / no-evidence 场景矩阵。
|
- `evidence-pipeline-scenarios.md`:PASS / LOW_CONFID / REJECT / no-evidence 场景矩阵。
|
||||||
- `trace-inspection-checklist.md`:Trace 字段检查清单。
|
- `trace-inspection-checklist.md`:Trace 字段检查清单。
|
||||||
|
- `scripts/run-interview-demo-check.ps1`:面试预检脚本,包含服务可达性、Chat、Trace、反馈和 summary 输出。
|
||||||
- `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。
|
- `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。
|
||||||
|
- `interview-q-and-a.md`:面试追问回答,覆盖 Agent 工程取舍、审计和评测。
|
||||||
- `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。
|
- `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。
|
||||||
- `requests/narrow-highcpu-chat.json`:窄范围正向观察请求。
|
- `requests/narrow-highcpu-chat.json`:窄范围正向观察请求。
|
||||||
- `requests/hikari-no-evidence-chat.json`:no-evidence 负向观察请求。
|
- `requests/hikari-no-evidence-chat.json`:no-evidence 负向观察请求。
|
||||||
@@ -37,7 +39,7 @@ http://localhost:9900
|
|||||||
最快方式:
|
最快方式:
|
||||||
|
|
||||||
```powershell
|
```powershell
|
||||||
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1
|
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1
|
||||||
```
|
```
|
||||||
|
|
||||||
脚本会生成:
|
脚本会生成:
|
||||||
@@ -46,6 +48,7 @@ powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-de
|
|||||||
mvp/demo/output/chat-response.json
|
mvp/demo/output/chat-response.json
|
||||||
mvp/demo/output/trace-response.json
|
mvp/demo/output/trace-response.json
|
||||||
mvp/demo/output/feedback-response.json
|
mvp/demo/output/feedback-response.json
|
||||||
|
mvp/demo/output/interview-demo-summary.json
|
||||||
```
|
```
|
||||||
|
|
||||||
手动请求:
|
手动请求:
|
||||||
@@ -85,6 +88,8 @@ Invoke-RestMethod `
|
|||||||
- `data.steps` 包含 planner / executor / verifier 等步骤
|
- `data.steps` 包含 planner / executor / verifier 等步骤
|
||||||
- `data.toolInvocations` 包含 `lookup_knowledge`、`query_logs`、`query_metrics` 等证据工具
|
- `data.toolInvocations` 包含 `lookup_knowledge`、`query_logs`、`query_metrics` 等证据工具
|
||||||
- `data.session.selfEvaluation` 包含 verifier 或 rule evaluation
|
- `data.session.selfEvaluation` 包含 verifier 或 rule evaluation
|
||||||
|
- Chat V2 链路中,`data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` 记录 Chat Prompt 审计版本
|
||||||
|
- Chat V2 链路中,`data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` 记录 Gatekeeper 规则集版本
|
||||||
|
|
||||||
## 5. 提交反馈
|
## 5. 提交反馈
|
||||||
|
|
||||||
@@ -176,8 +181,8 @@ AIOps 主线:
|
|||||||
|
|
||||||
面试时不要把所有安全场景都压到 live LLM 现场表现上。建议使用:
|
面试时不要把所有安全场景都压到 live LLM 现场表现上。建议使用:
|
||||||
|
|
||||||
- `scripts/run-payment-timeout-demo.ps1` 跑主路径。
|
- `scripts/run-interview-demo-check.ps1` 跑主路径和预检 summary。
|
||||||
- `evidence-pipeline-scenarios.md` 讲解 PASS / LOW_CONFID / REJECT / no-evidence 矩阵。
|
- `evidence-pipeline-scenarios.md` 讲解 PASS / LOW_CONFID / REJECT / no-evidence 矩阵。
|
||||||
- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 10/10 通过。
|
- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 12/12 通过。
|
||||||
|
|
||||||
这样可以同时展示真实链路和确定性回归能力。
|
这样可以同时展示真实链路和确定性回归能力。
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
# 面试追问 Q&A
|
||||||
|
|
||||||
|
## 为什么不用普通 Chatbot?
|
||||||
|
|
||||||
|
这个项目的重点不是生成一段诊断文本,而是把诊断拆成可审计链路:Planner 拆解问题,Executor 调工具拿证据,Gatekeeper 用代码核验证据引用,Verifier 判断可推导性,Composer 生成最终表达。每次运行都能通过同一个 `sessionId` 回放。
|
||||||
|
|
||||||
|
## 为什么 RAG 要做成显式工具?
|
||||||
|
|
||||||
|
`lookup_knowledge` 保持显式工具调用,才能在 `tool_invocation` 里看到 Agent 查了什么、命中了什么、相关性等级是什么,以及最终答案是否真的使用了这些证据。隐式 Advisor 更方便,但不利于审计 Agent 决策。
|
||||||
|
|
||||||
|
## 怎么防止 Executor 幻觉?
|
||||||
|
|
||||||
|
Executor 不直接负责最终用户答案,而是输出 `executor_evidence_v2` 的微观事实和证据引用。Gatekeeper 会校验 `source_invocation_id`、`raw_path`、`evidence_excerpt` 是否真实存在;Verifier 再判断 claim 是否能由已验真的证据推出;Composer 只表达 Verifier 允许输出的内容。
|
||||||
|
|
||||||
|
## LOW_CONFID 是失败吗?
|
||||||
|
|
||||||
|
不是。`LOW_CONFID` 表示当前证据不足以支撑强结论,但系统仍然可以安全表达已确认事实和缺失信息。面试时可以把它作为“没有证据就不强答”的质量门禁,而不是模型能力失败。
|
||||||
|
|
||||||
|
## Prompt 改了怎么审计?
|
||||||
|
|
||||||
|
Chat verifier evaluation 里会记录 `prompt_audit.version`,并列出 planner、executor、verifier、composer 的 Prompt 版本和资源路径。它不保存完整 Prompt 文本,只保留用于回放和回归解释的紧凑元数据。
|
||||||
|
|
||||||
|
## Gatekeeper 改了怎么审计?
|
||||||
|
|
||||||
|
Gatekeeper 结果里记录 `gatekeeper_result.rule_set_version` 和已启用规则元数据摘要。规则执行仍是确定性 Java 代码,版本和规则元数据用于解释“这次引用验真用的是哪套规则”。
|
||||||
|
|
||||||
|
## 为什么现在不拆 SubAgent?
|
||||||
|
|
||||||
|
当前 MVP 的主要风险不是 Agent 数量不够,而是证据、验证和回归是否稳定。文档里的演进路线把 SubAgent 放在 P2:等故障类型、工具权限和评测集足够明确后再拆,避免只是移动复杂度。
|
||||||
|
|
||||||
|
## 为什么 baseline 比 live demo 更重要?
|
||||||
|
|
||||||
|
live demo 证明链路在当前环境能跑通,但 LLM 和外部依赖会波动。`mvp/eval` 的固定 fixture baseline 是确定性回归来源,用来判断 Prompt、工具、Gatekeeper、Verifier 或 Composer 的改动有没有让系统退化。
|
||||||
@@ -0,0 +1,161 @@
|
|||||||
|
param(
|
||||||
|
[string]$BaseUrl = "http://localhost:9900",
|
||||||
|
[string]$SessionId = "mvp-demo-interview-payment-timeout-001",
|
||||||
|
[string]$RequestFile = "$PSScriptRoot/../requests/payment-timeout-chat.json",
|
||||||
|
[string]$OutputDir = "$PSScriptRoot/../output"
|
||||||
|
)
|
||||||
|
|
||||||
|
$ErrorActionPreference = "Stop"
|
||||||
|
|
||||||
|
function Test-ServiceReachable {
|
||||||
|
param([string]$Url)
|
||||||
|
|
||||||
|
try {
|
||||||
|
$request = [System.Net.WebRequest]::Create($Url)
|
||||||
|
$request.Method = "GET"
|
||||||
|
$request.Timeout = 5000
|
||||||
|
$response = $request.GetResponse()
|
||||||
|
$response.Close()
|
||||||
|
return $true
|
||||||
|
} catch [System.Net.WebException] {
|
||||||
|
if ($_.Exception.Response -ne $null) {
|
||||||
|
$_.Exception.Response.Close()
|
||||||
|
return $true
|
||||||
|
}
|
||||||
|
return $false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function Get-TraceData {
|
||||||
|
param($TraceResponse)
|
||||||
|
|
||||||
|
if ($TraceResponse.PSObject.Properties.Name -contains "data") {
|
||||||
|
return $TraceResponse.data
|
||||||
|
}
|
||||||
|
return $TraceResponse
|
||||||
|
}
|
||||||
|
|
||||||
|
function Get-SelfEvaluation {
|
||||||
|
param($TraceData)
|
||||||
|
|
||||||
|
if ($null -eq $TraceData -or $null -eq $TraceData.session) {
|
||||||
|
return $null
|
||||||
|
}
|
||||||
|
return $TraceData.session.selfEvaluation
|
||||||
|
}
|
||||||
|
|
||||||
|
function Get-ToolNames {
|
||||||
|
param($TraceData)
|
||||||
|
|
||||||
|
if ($null -eq $TraceData -or $null -eq $TraceData.toolInvocations) {
|
||||||
|
return @()
|
||||||
|
}
|
||||||
|
return @($TraceData.toolInvocations | ForEach-Object { $_.toolName } | Where-Object { $_ } | Sort-Object -Unique)
|
||||||
|
}
|
||||||
|
|
||||||
|
New-Item -ItemType Directory -Force -Path $OutputDir | Out-Null
|
||||||
|
|
||||||
|
Write-Host "Running interview demo preflight..."
|
||||||
|
Write-Host "BaseUrl: $BaseUrl"
|
||||||
|
Write-Host "SessionId: $SessionId"
|
||||||
|
|
||||||
|
if (-not (Test-ServiceReachable -Url $BaseUrl)) {
|
||||||
|
throw "Service is not reachable: $BaseUrl. Start the app with mvp-demo profile first: mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo"
|
||||||
|
}
|
||||||
|
|
||||||
|
$request = Get-Content -Raw -Encoding UTF8 -Path $RequestFile | ConvertFrom-Json
|
||||||
|
$request.Id = $SessionId
|
||||||
|
$body = $request | ConvertTo-Json -Depth 8
|
||||||
|
|
||||||
|
$chatRequest = @{
|
||||||
|
Method = "Post"
|
||||||
|
Uri = "$BaseUrl/api/chat"
|
||||||
|
ContentType = "application/json; charset=utf-8"
|
||||||
|
Body = $body
|
||||||
|
}
|
||||||
|
$chat = Invoke-RestMethod @chatRequest
|
||||||
|
|
||||||
|
$chatPath = Join-Path $OutputDir "chat-response.json"
|
||||||
|
$chat | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $chatPath
|
||||||
|
|
||||||
|
$traceRequest = @{
|
||||||
|
Method = "Get"
|
||||||
|
Uri = "$BaseUrl/api/diagnosis/$SessionId/trace"
|
||||||
|
}
|
||||||
|
$trace = Invoke-RestMethod @traceRequest
|
||||||
|
|
||||||
|
$tracePath = Join-Path $OutputDir "trace-response.json"
|
||||||
|
$trace | ConvertTo-Json -Depth 80 | Set-Content -Encoding UTF8 -Path $tracePath
|
||||||
|
|
||||||
|
$feedbackBody = @{
|
||||||
|
sessionId = $SessionId
|
||||||
|
feedback = "useful"
|
||||||
|
} | ConvertTo-Json
|
||||||
|
|
||||||
|
$feedbackRequest = @{
|
||||||
|
Method = "Post"
|
||||||
|
Uri = "$BaseUrl/api/feedback"
|
||||||
|
ContentType = "application/json; charset=utf-8"
|
||||||
|
Body = $feedbackBody
|
||||||
|
}
|
||||||
|
$feedback = Invoke-RestMethod @feedbackRequest
|
||||||
|
|
||||||
|
$feedbackPath = Join-Path $OutputDir "feedback-response.json"
|
||||||
|
$feedback | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $feedbackPath
|
||||||
|
|
||||||
|
$traceData = Get-TraceData -TraceResponse $trace
|
||||||
|
$selfEvaluation = Get-SelfEvaluation -TraceData $traceData
|
||||||
|
$verifierEvaluation = $null
|
||||||
|
if ($null -ne $selfEvaluation) {
|
||||||
|
$verifierEvaluation = $selfEvaluation.verifier_evaluation
|
||||||
|
}
|
||||||
|
|
||||||
|
$gatekeeperResult = $null
|
||||||
|
$promptAudit = $null
|
||||||
|
if ($null -ne $verifierEvaluation) {
|
||||||
|
$gatekeeperResult = $verifierEvaluation.gatekeeper_result
|
||||||
|
$promptAudit = $verifierEvaluation.prompt_audit
|
||||||
|
}
|
||||||
|
|
||||||
|
$verdict = $null
|
||||||
|
$gatekeeperStatus = $null
|
||||||
|
$gatekeeperRuleSetVersion = $null
|
||||||
|
$promptAuditVersion = $null
|
||||||
|
if ($null -ne $verifierEvaluation) {
|
||||||
|
$verdict = $verifierEvaluation.verdict
|
||||||
|
}
|
||||||
|
if ($null -ne $gatekeeperResult) {
|
||||||
|
$gatekeeperStatus = $gatekeeperResult.status
|
||||||
|
$gatekeeperRuleSetVersion = $gatekeeperResult.rule_set_version
|
||||||
|
}
|
||||||
|
if ($null -ne $promptAudit) {
|
||||||
|
$promptAuditVersion = $promptAudit.version
|
||||||
|
}
|
||||||
|
$toolNames = Get-ToolNames -TraceData $traceData
|
||||||
|
$summaryPath = Join-Path $OutputDir "interview-demo-summary.json"
|
||||||
|
|
||||||
|
$summary = [ordered]@{
|
||||||
|
sessionId = $SessionId
|
||||||
|
baseUrl = $BaseUrl
|
||||||
|
chatSuccess = $chat.data.success
|
||||||
|
verdict = $verdict
|
||||||
|
gatekeeperStatus = $gatekeeperStatus
|
||||||
|
gatekeeperRuleSetVersion = $gatekeeperRuleSetVersion
|
||||||
|
promptAuditVersion = $promptAuditVersion
|
||||||
|
toolNames = $toolNames
|
||||||
|
paths = [ordered]@{
|
||||||
|
chat = $chatPath
|
||||||
|
trace = $tracePath
|
||||||
|
feedback = $feedbackPath
|
||||||
|
summary = $summaryPath
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
$summary | ConvertTo-Json -Depth 20 | Set-Content -Encoding UTF8 -Path $summaryPath
|
||||||
|
|
||||||
|
Write-Host ""
|
||||||
|
Write-Host "Interview demo preflight completed."
|
||||||
|
Write-Host "Verdict: $($summary.verdict)"
|
||||||
|
Write-Host "Gatekeeper rules: $($summary.gatekeeperRuleSetVersion)"
|
||||||
|
Write-Host "Prompt audit: $($summary.promptAuditVersion)"
|
||||||
|
Write-Host "Summary: $summaryPath"
|
||||||
@@ -37,7 +37,7 @@ http://localhost:9900
|
|||||||
推荐使用固定脚本:
|
推荐使用固定脚本:
|
||||||
|
|
||||||
```powershell
|
```powershell
|
||||||
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1
|
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1
|
||||||
```
|
```
|
||||||
|
|
||||||
脚本会写出:
|
脚本会写出:
|
||||||
@@ -46,6 +46,7 @@ powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-de
|
|||||||
mvp/demo/output/chat-response.json
|
mvp/demo/output/chat-response.json
|
||||||
mvp/demo/output/trace-response.json
|
mvp/demo/output/trace-response.json
|
||||||
mvp/demo/output/feedback-response.json
|
mvp/demo/output/feedback-response.json
|
||||||
|
mvp/demo/output/interview-demo-summary.json
|
||||||
```
|
```
|
||||||
|
|
||||||
现场话术:
|
现场话术:
|
||||||
@@ -98,6 +99,8 @@ data.toolInvocations[*].outputPreview
|
|||||||
data.toolInvocations[*].retrievalLayer
|
data.toolInvocations[*].retrievalLayer
|
||||||
data.toolInvocations[*].relevanceLevel
|
data.toolInvocations[*].relevanceLevel
|
||||||
data.summary.hasVerifierEvaluation
|
data.summary.hasVerifierEvaluation
|
||||||
|
data.session.selfEvaluation.verifier_evaluation.prompt_audit.version
|
||||||
|
data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version
|
||||||
```
|
```
|
||||||
|
|
||||||
现场话术:
|
现场话术:
|
||||||
@@ -155,6 +158,10 @@ Verifier 不做新检索,只看工具 trace 汇总。
|
|||||||
如果 PASS,就输出原答案。
|
如果 PASS,就输出原答案。
|
||||||
如果 LOW_CONFID,可以补证据或加低置信提示。
|
如果 LOW_CONFID,可以补证据或加低置信提示。
|
||||||
如果 REJECT,就降级输出,只保留已确认信息。
|
如果 REJECT,就降级输出,只保留已确认信息。
|
||||||
|
|
||||||
|
Prompt 和 Gatekeeper 的版本也会进入 trace。
|
||||||
|
`prompt_audit.version` 用于说明本次 Chat 使用哪套 Prompt 契约,`gatekeeper_result.rule_set_version` 用于说明引用验真的规则版本。
|
||||||
|
固定 fixture baseline 是回归判断来源,live demo 主要证明当前环境链路可跑通。
|
||||||
```
|
```
|
||||||
|
|
||||||
## 7. 展示反馈闭环
|
## 7. 展示反馈闭环
|
||||||
@@ -226,6 +233,7 @@ AIOps 有两个模式。
|
|||||||
mvp/demo/output/chat-response.json
|
mvp/demo/output/chat-response.json
|
||||||
mvp/demo/output/trace-response.json
|
mvp/demo/output/trace-response.json
|
||||||
mvp/demo/output/feedback-response.json
|
mvp/demo/output/feedback-response.json
|
||||||
|
mvp/demo/output/interview-demo-summary.json
|
||||||
```
|
```
|
||||||
|
|
||||||
降级话术:
|
降级话术:
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# Trace 检查清单
|
# Trace 检查清单
|
||||||
|
|
||||||
运行 `scripts/run-payment-timeout-demo.ps1` 后,用这份清单检查 `trace-response.json`。
|
运行 `scripts/run-interview-demo-check.ps1` 后,用这份清单检查 `trace-response.json` 和 `interview-demo-summary.json`。
|
||||||
|
|
||||||
## 1. Session
|
## 1. Session
|
||||||
|
|
||||||
@@ -11,6 +11,8 @@
|
|||||||
| `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace |
|
| `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace |
|
||||||
| `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 |
|
| `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 |
|
||||||
| `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 如果是 Chat V2 链路,是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 |
|
| `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 如果是 Chat V2 链路,是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 |
|
||||||
|
| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` | 如果是 Chat V2 链路,是否记录 Prompt 审计版本 | Prompt 变更可解释、可回归 |
|
||||||
|
| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.prompts[*].version` | 是否记录 planner / executor / verifier / composer 版本 | 便于定位 Prompt 变更影响 |
|
||||||
| `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在同一次诊断上 |
|
| `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在同一次诊断上 |
|
||||||
|
|
||||||
## 2. Agent 步骤
|
## 2. Agent 步骤
|
||||||
|
|||||||
+8
-4
@@ -29,10 +29,10 @@ The baseline evaluates saved trace fixtures. It does not start the application a
|
|||||||
The committed baseline currently contains:
|
The committed baseline currently contains:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
10 fixed cases
|
12 fixed cases
|
||||||
10 passing fixture evaluations
|
12 passing fixture evaluations
|
||||||
4 PASS verdicts
|
5 PASS verdicts
|
||||||
5 LOW_CONFID verdicts
|
6 LOW_CONFID verdicts
|
||||||
1 REJECT verdict
|
1 REJECT verdict
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -44,6 +44,8 @@ The V2 evidence-pipeline matrix covers:
|
|||||||
- Unsupported claim filtering before the final answer.
|
- Unsupported claim filtering before the final answer.
|
||||||
- Composer fallback rendering without raw Executor JSON leakage.
|
- Composer fallback rendering without raw Executor JSON leakage.
|
||||||
- Gatekeeper rule set version audit for new matrix fixtures.
|
- Gatekeeper rule set version audit for new matrix fixtures.
|
||||||
|
- Prompt audit version checks for planner, executor, verifier, and composer prompts.
|
||||||
|
- Gatekeeper rule metadata checks for enabled rule id and default severity.
|
||||||
|
|
||||||
## Verification
|
## Verification
|
||||||
|
|
||||||
@@ -80,6 +82,8 @@ Stage 5 adds these V2 checks:
|
|||||||
- `claim_checks` must be structurally auditable.
|
- `claim_checks` must be structurally auditable.
|
||||||
- Composer output must record whether normal parsing or fallback rendering was used.
|
- Composer output must record whether normal parsing or fallback rendering was used.
|
||||||
- Gatekeeper rule set version can be asserted per fixture.
|
- Gatekeeper rule set version can be asserted per fixture.
|
||||||
|
- Prompt audit version and per-prompt versions can be asserted per fixture.
|
||||||
|
- Gatekeeper rule metadata can be required per fixture.
|
||||||
- Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`.
|
- Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`.
|
||||||
- Configured unsupported claim keywords must not appear as confirmed final-answer content.
|
- Configured unsupported claim keywords must not appear as confirmed final-answer content.
|
||||||
|
|
||||||
|
|||||||
@@ -17,6 +17,33 @@
|
|||||||
"expectedComposerStatuses": ["valid"],
|
"expectedComposerStatuses": ["valid"],
|
||||||
"forbiddenConfirmedClaimKeywords": ["数据库连接池"]
|
"forbiddenConfirmedClaimKeywords": ["数据库连接池"]
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"id": "prompt-gatekeeper-audit-closure",
|
||||||
|
"title": "Prompt and Gatekeeper audit closure",
|
||||||
|
"question": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。",
|
||||||
|
"traceFixture": "prompt-gatekeeper-audit-closure-pass.json",
|
||||||
|
"expectedRootCauseKeywords": ["payment-service", "HighCPUUsage", "92%"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["query_metrics"],
|
||||||
|
"allowedVerdicts": ["PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["根因", "修复建议", "通常情况下"],
|
||||||
|
"requireV2AuditClosure": true,
|
||||||
|
"requireClaimChecks": true,
|
||||||
|
"requireComposerOutput": true,
|
||||||
|
"expectedGatekeeperStatuses": ["pass"],
|
||||||
|
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||||
|
"expectedComposerStatuses": ["valid"],
|
||||||
|
"forbiddenConfirmedClaimKeywords": ["数据库连接池"],
|
||||||
|
"requirePromptAudit": true,
|
||||||
|
"expectedPromptAuditVersion": "chat-prompts-v1",
|
||||||
|
"expectedPromptVersions": {
|
||||||
|
"chat_planner": "chat-planner-v1",
|
||||||
|
"chat_executor": "chat-executor-v2",
|
||||||
|
"chat_verifier": "chat-verifier-v2",
|
||||||
|
"chat_composer": "chat-composer-v1"
|
||||||
|
},
|
||||||
|
"requireGatekeeperRules": true
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"id": "hikari-no-evidence-negative-observation",
|
"id": "hikari-no-evidence-negative-observation",
|
||||||
"title": "Hikari no-evidence negative observation",
|
"title": "Hikari no-evidence negative observation",
|
||||||
@@ -124,6 +151,33 @@
|
|||||||
"expectedComposerStatuses": ["valid"],
|
"expectedComposerStatuses": ["valid"],
|
||||||
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"id": "audit-metadata-low-confid",
|
||||||
|
"title": "Audit metadata low confidence",
|
||||||
|
"question": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?",
|
||||||
|
"traceFixture": "audit-metadata-low-confid.json",
|
||||||
|
"expectedRootCauseKeywords": ["超时", "证据"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["query_logs"],
|
||||||
|
"allowedVerdicts": ["LOW_CONFID"],
|
||||||
|
"forbiddenAnswerKeywords": ["已经确认"],
|
||||||
|
"requireV2AuditClosure": true,
|
||||||
|
"requireClaimChecks": true,
|
||||||
|
"requireComposerOutput": true,
|
||||||
|
"expectedGatekeeperStatuses": ["pass"],
|
||||||
|
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||||
|
"expectedComposerStatuses": ["valid"],
|
||||||
|
"forbiddenConfirmedClaimKeywords": ["主库故障"],
|
||||||
|
"requirePromptAudit": true,
|
||||||
|
"expectedPromptAuditVersion": "chat-prompts-v1",
|
||||||
|
"expectedPromptVersions": {
|
||||||
|
"chat_planner": "chat-planner-v1",
|
||||||
|
"chat_executor": "chat-executor-v2",
|
||||||
|
"chat_verifier": "chat-verifier-v2",
|
||||||
|
"chat_composer": "chat-composer-v1"
|
||||||
|
},
|
||||||
|
"requireGatekeeperRules": true
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"id": "composer-fallback-no-raw-json",
|
"id": "composer-fallback-no-raw-json",
|
||||||
"title": "Composer fallback no raw JSON",
|
"title": "Composer fallback no raw JSON",
|
||||||
|
|||||||
@@ -0,0 +1,192 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-audit-metadata-low-confid",
|
||||||
|
"query": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 45000,
|
||||||
|
"toolCallCount": 1,
|
||||||
|
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "LOW_CONFID",
|
||||||
|
"groundedness_score": 0.42,
|
||||||
|
"critical_fact_count": 2,
|
||||||
|
"prompt_audit": {
|
||||||
|
"version": "chat-prompts-v1",
|
||||||
|
"prompts": [
|
||||||
|
{
|
||||||
|
"name": "chat_planner",
|
||||||
|
"version": "chat-planner-v1",
|
||||||
|
"resource": "prompts/chat-planner-prompt.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "chat_executor",
|
||||||
|
"version": "chat-executor-v2",
|
||||||
|
"resource": "prompts/chat-executor-prompt.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "chat_verifier",
|
||||||
|
"version": "chat-verifier-v2",
|
||||||
|
"resource": "prompts/chat-verifier-prompt.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "chat_composer",
|
||||||
|
"version": "chat-composer-v1",
|
||||||
|
"resource": "prompts/chat-composer-prompt.md"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"gatekeeper_result": {
|
||||||
|
"status": "pass",
|
||||||
|
"severity": "none",
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.invocation",
|
||||||
|
"description": "source_invocation_id must reference an existing tool invocation",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "evidence.excerpt",
|
||||||
|
"description": "evidence_excerpt must be supported by recorded evidence text",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"checked_bindings": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-timeout",
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"source_invocation_id": 22,
|
||||||
|
"raw_path": "$.logs[0]",
|
||||||
|
"matched_text": "order api timeout",
|
||||||
|
"status": "pass"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"failed_rules": [],
|
||||||
|
"warnings": [],
|
||||||
|
"errors": []
|
||||||
|
},
|
||||||
|
"executor_structured_output": {
|
||||||
|
"answer_version": "executor_evidence_v2",
|
||||||
|
"claims": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-timeout",
|
||||||
|
"claim_type": "symptom",
|
||||||
|
"claim_text": "订单接口出现超时",
|
||||||
|
"support_level": "direct",
|
||||||
|
"evidence_bindings": [
|
||||||
|
{
|
||||||
|
"source_type": "tool_trace",
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"source_invocation_id": 22,
|
||||||
|
"raw_path": "$.logs[0]",
|
||||||
|
"evidence_excerpt": "order api timeout"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"claim_id": "claim-db-primary",
|
||||||
|
"claim_type": "root_cause",
|
||||||
|
"claim_text": "数据库主库故障导致订单超时",
|
||||||
|
"support_level": "weak",
|
||||||
|
"evidence_bindings": [
|
||||||
|
{
|
||||||
|
"source_type": "tool_trace",
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"source_invocation_id": 22,
|
||||||
|
"raw_path": "$.logs[0]",
|
||||||
|
"evidence_excerpt": "order api timeout"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"hypotheses": [],
|
||||||
|
"recommended_actions": [
|
||||||
|
{
|
||||||
|
"action_text": "补充查询数据库主库状态和错误日志",
|
||||||
|
"reason": "当前只有订单接口超时日志"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"missing_info": ["数据库主库状态", "数据库错误日志"]
|
||||||
|
},
|
||||||
|
"claim_checks": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-timeout",
|
||||||
|
"claim_text": "订单接口出现超时",
|
||||||
|
"claim_type": "symptom",
|
||||||
|
"verification": "direct_observation",
|
||||||
|
"detail": "日志直接记录 order api timeout",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"source_invocation_id": 22,
|
||||||
|
"raw_path": "$.logs[0]"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"claim_id": "claim-db-primary",
|
||||||
|
"claim_text": "数据库主库故障导致订单超时",
|
||||||
|
"claim_type": "root_cause",
|
||||||
|
"verification": "unsupported",
|
||||||
|
"detail": "日志只能证明订单接口超时,不能证明数据库主库故障",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"source_invocation_id": 22,
|
||||||
|
"raw_path": "$.logs[0]"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"facts_checked": [],
|
||||||
|
"composer_output": {
|
||||||
|
"status": "valid",
|
||||||
|
"answer_summary": "日志显示订单接口超时,但数据库方向证据不足。",
|
||||||
|
"recommended_actions": [
|
||||||
|
{
|
||||||
|
"action_text": "补充查询数据库主库状态和错误日志",
|
||||||
|
"reason": "当前只有订单接口超时日志"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"user_facing_answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。"
|
||||||
|
},
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 22,
|
||||||
|
"sessionId": "eval-audit-metadata-low-confid",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"outputPreview": "order api timeout",
|
||||||
|
"retrievalDetails": {
|
||||||
|
"evidence_status": "supported",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"raw_path": "$.logs[0]",
|
||||||
|
"text": "order api timeout"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 1,
|
||||||
|
"returnedToolCallCount": 1,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,155 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-prompt-gatekeeper-audit-closure",
|
||||||
|
"query": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 19000,
|
||||||
|
"toolCallCount": 1,
|
||||||
|
"answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "PASS",
|
||||||
|
"groundedness_score": 1.0,
|
||||||
|
"critical_fact_count": 1,
|
||||||
|
"prompt_audit": {
|
||||||
|
"version": "chat-prompts-v1",
|
||||||
|
"prompts": [
|
||||||
|
{
|
||||||
|
"name": "chat_planner",
|
||||||
|
"version": "chat-planner-v1",
|
||||||
|
"resource": "prompts/chat-planner-prompt.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "chat_executor",
|
||||||
|
"version": "chat-executor-v2",
|
||||||
|
"resource": "prompts/chat-executor-prompt.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "chat_verifier",
|
||||||
|
"version": "chat-verifier-v2",
|
||||||
|
"resource": "prompts/chat-verifier-prompt.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "chat_composer",
|
||||||
|
"version": "chat-composer-v1",
|
||||||
|
"resource": "prompts/chat-composer-prompt.md"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"gatekeeper_result": {
|
||||||
|
"status": "pass",
|
||||||
|
"severity": "none",
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.invocation",
|
||||||
|
"description": "source_invocation_id must reference an existing tool invocation",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"checked_bindings": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"source_invocation_id": 21,
|
||||||
|
"raw_path": "$.alerts[0]",
|
||||||
|
"matched_text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
|
||||||
|
"status": "pass"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"failed_rules": [],
|
||||||
|
"warnings": [],
|
||||||
|
"errors": []
|
||||||
|
},
|
||||||
|
"executor_structured_output": {
|
||||||
|
"answer_version": "executor_evidence_v2",
|
||||||
|
"claims": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"claim_type": "observation",
|
||||||
|
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||||
|
"support_level": "direct",
|
||||||
|
"evidence_bindings": [
|
||||||
|
{
|
||||||
|
"source_type": "tool_trace",
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"source_invocation_id": 21,
|
||||||
|
"raw_path": "$.alerts[0]",
|
||||||
|
"evidence_excerpt": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"hypotheses": [],
|
||||||
|
"recommended_actions": [],
|
||||||
|
"missing_info": []
|
||||||
|
},
|
||||||
|
"claim_checks": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||||
|
"claim_type": "observation",
|
||||||
|
"verification": "direct_observation",
|
||||||
|
"detail": "已核验的指标证据直接包含服务名、告警名和 CPU 当前值。",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"source_invocation_id": 21,
|
||||||
|
"raw_path": "$.alerts[0]"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"facts_checked": [],
|
||||||
|
"composer_output": {
|
||||||
|
"status": "valid",
|
||||||
|
"answer_summary": "payment-service 当前存在 HighCPUUsage 告警。",
|
||||||
|
"recommended_actions": [],
|
||||||
|
"user_facing_answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。"
|
||||||
|
},
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"success": true,
|
||||||
|
"source_invocation_ids": [21],
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 21,
|
||||||
|
"sessionId": "eval-prompt-gatekeeper-audit-closure",
|
||||||
|
"toolName": "query_metrics",
|
||||||
|
"outputPreview": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
|
||||||
|
"retrievalDetails": {
|
||||||
|
"evidence_status": "supported",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"raw_path": "$.alerts[0]",
|
||||||
|
"text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 1,
|
||||||
|
"returnedToolCallCount": 1,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,14 +1,14 @@
|
|||||||
{
|
{
|
||||||
"totalCases" : 10,
|
"totalCases" : 12,
|
||||||
"passedCases" : 10,
|
"passedCases" : 12,
|
||||||
"passRate" : 1.0,
|
"passRate" : 1.0,
|
||||||
"verdictDistribution" : {
|
"verdictDistribution" : {
|
||||||
"PASS" : 4,
|
"PASS" : 5,
|
||||||
"LOW_CONFID" : 5,
|
"LOW_CONFID" : 6,
|
||||||
"REJECT" : 1
|
"REJECT" : 1
|
||||||
},
|
},
|
||||||
"averageToolCallCount" : 1.5,
|
"averageToolCallCount" : 1.4166666666666667,
|
||||||
"averageDurationMs" : 39800.0,
|
"averageDurationMs" : 38500.0,
|
||||||
"results" : [ {
|
"results" : [ {
|
||||||
"caseId" : "narrow-highcpu-observation",
|
"caseId" : "narrow-highcpu-observation",
|
||||||
"title" : "Narrow HighCPU observation",
|
"title" : "Narrow HighCPU observation",
|
||||||
@@ -22,10 +22,31 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : "pass",
|
"gatekeeperStatus" : "pass",
|
||||||
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : "valid",
|
"composerStatus" : "valid",
|
||||||
"claimCheckCount" : 1,
|
"claimCheckCount" : 1,
|
||||||
|
"gatekeeperRuleCount" : 1,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
"durationMs" : 18000
|
"durationMs" : 18000
|
||||||
|
}, {
|
||||||
|
"caseId" : "prompt-gatekeeper-audit-closure",
|
||||||
|
"title" : "Prompt and Gatekeeper audit closure",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "PASS",
|
||||||
|
"matchedKeywordCount" : 3,
|
||||||
|
"requiredKeywordCount" : 3,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_metrics" : true
|
||||||
|
},
|
||||||
|
"gatekeeperStatus" : "pass",
|
||||||
|
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||||
|
"promptAuditVersion" : "chat-prompts-v1",
|
||||||
|
"composerStatus" : "valid",
|
||||||
|
"claimCheckCount" : 1,
|
||||||
|
"gatekeeperRuleCount" : 2,
|
||||||
|
"toolCallCount" : 1,
|
||||||
|
"durationMs" : 19000
|
||||||
}, {
|
}, {
|
||||||
"caseId" : "hikari-no-evidence-negative-observation",
|
"caseId" : "hikari-no-evidence-negative-observation",
|
||||||
"title" : "Hikari no-evidence negative observation",
|
"title" : "Hikari no-evidence negative observation",
|
||||||
@@ -39,8 +60,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : "pass",
|
"gatekeeperStatus" : "pass",
|
||||||
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : "valid",
|
"composerStatus" : "valid",
|
||||||
"claimCheckCount" : 1,
|
"claimCheckCount" : 1,
|
||||||
|
"gatekeeperRuleCount" : 1,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
"durationMs" : 21000
|
"durationMs" : 21000
|
||||||
}, {
|
}, {
|
||||||
@@ -58,8 +81,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 3,
|
"toolCallCount" : 3,
|
||||||
"durationMs" : 42000
|
"durationMs" : 42000
|
||||||
}, {
|
}, {
|
||||||
@@ -76,8 +101,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 2,
|
"toolCallCount" : 2,
|
||||||
"durationMs" : 51000
|
"durationMs" : 51000
|
||||||
}, {
|
}, {
|
||||||
@@ -93,8 +120,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
"durationMs" : 36000
|
"durationMs" : 36000
|
||||||
}, {
|
}, {
|
||||||
@@ -111,8 +140,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 2,
|
"toolCallCount" : 2,
|
||||||
"durationMs" : 47000
|
"durationMs" : 47000
|
||||||
}, {
|
}, {
|
||||||
@@ -129,8 +160,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 2,
|
"toolCallCount" : 2,
|
||||||
"durationMs" : 53000
|
"durationMs" : 53000
|
||||||
}, {
|
}, {
|
||||||
@@ -146,8 +179,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : "fail",
|
"gatekeeperStatus" : "fail",
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : "valid",
|
"composerStatus" : "valid",
|
||||||
"claimCheckCount" : 1,
|
"claimCheckCount" : 1,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
"durationMs" : 39000
|
"durationMs" : 39000
|
||||||
}, {
|
}, {
|
||||||
@@ -163,10 +198,31 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : "pass",
|
"gatekeeperStatus" : "pass",
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : "valid",
|
"composerStatus" : "valid",
|
||||||
"claimCheckCount" : 2,
|
"claimCheckCount" : 2,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
"durationMs" : 44000
|
"durationMs" : 44000
|
||||||
|
}, {
|
||||||
|
"caseId" : "audit-metadata-low-confid",
|
||||||
|
"title" : "Audit metadata low confidence",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "LOW_CONFID",
|
||||||
|
"matchedKeywordCount" : 2,
|
||||||
|
"requiredKeywordCount" : 2,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_logs" : true
|
||||||
|
},
|
||||||
|
"gatekeeperStatus" : "pass",
|
||||||
|
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||||
|
"promptAuditVersion" : "chat-prompts-v1",
|
||||||
|
"composerStatus" : "valid",
|
||||||
|
"claimCheckCount" : 2,
|
||||||
|
"gatekeeperRuleCount" : 2,
|
||||||
|
"toolCallCount" : 1,
|
||||||
|
"durationMs" : 45000
|
||||||
}, {
|
}, {
|
||||||
"caseId" : "composer-fallback-no-raw-json",
|
"caseId" : "composer-fallback-no-raw-json",
|
||||||
"title" : "Composer fallback no raw JSON",
|
"title" : "Composer fallback no raw JSON",
|
||||||
@@ -180,8 +236,10 @@
|
|||||||
},
|
},
|
||||||
"gatekeeperStatus" : "pass",
|
"gatekeeperStatus" : "pass",
|
||||||
"gatekeeperRuleSetVersion" : null,
|
"gatekeeperRuleSetVersion" : null,
|
||||||
|
"promptAuditVersion" : null,
|
||||||
"composerStatus" : "composer_malformed",
|
"composerStatus" : "composer_malformed",
|
||||||
"claimCheckCount" : 2,
|
"claimCheckCount" : 2,
|
||||||
|
"gatekeeperRuleCount" : null,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
"durationMs" : 47000
|
"durationMs" : 47000
|
||||||
} ]
|
} ]
|
||||||
|
|||||||
@@ -1,28 +1,30 @@
|
|||||||
# Diagnosis Eval Report
|
# Diagnosis Eval Report
|
||||||
|
|
||||||
- Total cases: 10
|
- Total cases: 12
|
||||||
- Passed cases: 10
|
- Passed cases: 12
|
||||||
- Pass rate: 100.00%
|
- Pass rate: 100.00%
|
||||||
- Average tool calls: 1.50
|
- Average tool calls: 1.42
|
||||||
- Average duration ms: 39800.00
|
- Average duration ms: 38500.00
|
||||||
|
|
||||||
## Verdict Distribution
|
## Verdict Distribution
|
||||||
|
|
||||||
- PASS: 4
|
- PASS: 5
|
||||||
- LOW_CONFID: 5
|
- LOW_CONFID: 6
|
||||||
- REJECT: 1
|
- REJECT: 1
|
||||||
|
|
||||||
## Cases
|
## Cases
|
||||||
|
|
||||||
| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |
|
| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks |
|
||||||
| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |
|
| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- |
|
||||||
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 18000 | - |
|
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 18000 | - |
|
||||||
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 21000 | - |
|
| prompt-gatekeeper-audit-closure | PASS | PASS | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 1 | 2 | 3/3 | 1 | 19000 | - |
|
||||||
| payment-timeout | PASS | PASS | - | - | - | - | 3/3 | 3 | 42000 | - |
|
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 21000 | - |
|
||||||
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 51000 | - |
|
| payment-timeout | PASS | PASS | - | - | - | - | - | - | 3/3 | 3 | 42000 | - |
|
||||||
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | 2/2 | 1 | 36000 | - |
|
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 51000 | - |
|
||||||
| slow-response | PASS | PASS | - | - | - | - | 2/2 | 2 | 47000 | - |
|
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | - | - | 2/2 | 1 | 36000 | - |
|
||||||
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 53000 | - |
|
| slow-response | PASS | PASS | - | - | - | - | - | - | 2/2 | 2 | 47000 | - |
|
||||||
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | valid | 1 | 3/3 | 1 | 39000 | - |
|
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 53000 | - |
|
||||||
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | valid | 2 | 2/2 | 1 | 44000 | - |
|
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | - | valid | 1 | - | 3/3 | 1 | 39000 | - |
|
||||||
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | composer_malformed | 2 | 2/2 | 1 | 47000 | - |
|
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | - | valid | 2 | - | 2/2 | 1 | 44000 | - |
|
||||||
|
| audit-metadata-low-confid | PASS | LOW_CONFID | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 2 | 2 | 2/2 | 1 | 45000 | - |
|
||||||
|
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | - | composer_malformed | 2 | - | 2/2 | 1 | 47000 | - |
|
||||||
|
|||||||
+23
-1
@@ -34,7 +34,16 @@ baseline report:整套固定集当前认可的结果
|
|||||||
"expectedGatekeeperStatuses": ["pass"],
|
"expectedGatekeeperStatuses": ["pass"],
|
||||||
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||||
"expectedComposerStatuses": ["valid"],
|
"expectedComposerStatuses": ["valid"],
|
||||||
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
"forbiddenConfirmedClaimKeywords": ["主库故障"],
|
||||||
|
"requirePromptAudit": true,
|
||||||
|
"expectedPromptAuditVersion": "chat-prompts-v1",
|
||||||
|
"expectedPromptVersions": {
|
||||||
|
"chat_planner": "chat-planner-v1",
|
||||||
|
"chat_executor": "chat-executor-v2",
|
||||||
|
"chat_verifier": "chat-verifier-v2",
|
||||||
|
"chat_composer": "chat-composer-v1"
|
||||||
|
},
|
||||||
|
"requireGatekeeperRules": true
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -58,6 +67,10 @@ baseline report:整套固定集当前认可的结果
|
|||||||
| `expectedGatekeeperRuleSetVersion` | 期望的 Gatekeeper 规则集版本 | 配置后校验 `gatekeeper_result.rule_set_version` |
|
| `expectedGatekeeperRuleSetVersion` | 期望的 Gatekeeper 规则集版本 | 配置后校验 `gatekeeper_result.rule_set_version` |
|
||||||
| `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 |
|
| `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 |
|
||||||
| `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 |
|
| `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 |
|
||||||
|
| `requirePromptAudit` | 是否要求 Prompt 审计元数据 | 要求 `prompt_audit.version` 存在 |
|
||||||
|
| `expectedPromptAuditVersion` | 期望的 Prompt 审计目录版本 | 配置后校验 `prompt_audit.version` |
|
||||||
|
| `expectedPromptVersions` | 期望的各角色 Prompt 版本 | 校验 `prompt_audit.prompts[*].name/version` |
|
||||||
|
| `requireGatekeeperRules` | 是否要求 Gatekeeper 规则元数据 | 要求 `gatekeeper_result.rules` 非空,且每条规则有 `id`、`enabled`、`default_severity` |
|
||||||
|
|
||||||
## 2. Trace Fixture
|
## 2. Trace Fixture
|
||||||
|
|
||||||
@@ -72,6 +85,9 @@ fixture 是一次 Agent 运行后的 trace 快照。评测器只读取当前规
|
|||||||
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` |
|
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` |
|
||||||
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` |
|
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` |
|
||||||
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | Gatekeeper 规则集版本 | 新矩阵 case 可显式断言该版本 |
|
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | Gatekeeper 规则集版本 | 新矩阵 case 可显式断言该版本 |
|
||||||
|
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rules` | Gatekeeper 规则元数据摘要 | 审计 case 可要求规则列表非空且字段完整 |
|
||||||
|
| `session.selfEvaluation.verifier_evaluation.prompt_audit.version` | Chat Prompt 审计目录版本 | 审计 case 可显式断言该版本 |
|
||||||
|
| `session.selfEvaluation.verifier_evaluation.prompt_audit.prompts` | 各 Chat Prompt 名称、版本和资源路径 | 审计 case 可断言 planner、executor、verifier、composer 版本 |
|
||||||
| `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` |
|
| `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` |
|
||||||
| `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 |
|
| `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 |
|
||||||
| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 |
|
| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 |
|
||||||
@@ -95,8 +111,10 @@ Java 类型:`DiagnosisEvalResult`
|
|||||||
| `evidenceCoverage` | 每个必需工具是否出现 |
|
| `evidenceCoverage` | 每个必需工具是否出现 |
|
||||||
| `gatekeeperStatus` | 读到的 `gatekeeper_result.status` |
|
| `gatekeeperStatus` | 读到的 `gatekeeper_result.status` |
|
||||||
| `gatekeeperRuleSetVersion` | 读到的 `gatekeeper_result.rule_set_version` |
|
| `gatekeeperRuleSetVersion` | 读到的 `gatekeeper_result.rule_set_version` |
|
||||||
|
| `promptAuditVersion` | 读到的 `prompt_audit.version` |
|
||||||
| `composerStatus` | 读到的 `composer_output.status` |
|
| `composerStatus` | 读到的 `composer_output.status` |
|
||||||
| `claimCheckCount` | `claim_checks` 数量 |
|
| `claimCheckCount` | `claim_checks` 数量 |
|
||||||
|
| `gatekeeperRuleCount` | `gatekeeper_result.rules` 数量 |
|
||||||
| `toolCallCount` | trace 中工具调用总数 |
|
| `toolCallCount` | trace 中工具调用总数 |
|
||||||
| `durationMs` | trace 总耗时 |
|
| `durationMs` | trace 总耗时 |
|
||||||
|
|
||||||
@@ -130,6 +148,10 @@ Executor structured output
|
|||||||
- V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。
|
- V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。
|
||||||
- `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。
|
- `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。
|
||||||
- 配置 `expectedGatekeeperRuleSetVersion` 的 case 必须匹配 `gatekeeper_result.rule_set_version`。
|
- 配置 `expectedGatekeeperRuleSetVersion` 的 case 必须匹配 `gatekeeper_result.rule_set_version`。
|
||||||
|
- 配置 `requirePromptAudit` 的 case 必须包含 `prompt_audit.version`。
|
||||||
|
- 配置 `expectedPromptAuditVersion` 的 case 必须匹配 `prompt_audit.version`。
|
||||||
|
- 配置 `expectedPromptVersions` 的 case 必须能在 `prompt_audit.prompts` 中找到对应角色和版本。
|
||||||
|
- 配置 `requireGatekeeperRules` 的 case 必须包含非空 `gatekeeper_result.rules`,且每条规则有 `id`、`enabled`、`default_severity`。
|
||||||
- `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。
|
- `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。
|
||||||
- Composer 输出必须记录 `status`。
|
- Composer 输出必须记录 `status`。
|
||||||
- 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。
|
- 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。
|
||||||
|
|||||||
+55
-38
@@ -1,47 +1,64 @@
|
|||||||
# 已知问题记录
|
# MVP Issues 索引
|
||||||
|
|
||||||
| # | 标题 | 严重程度 | 状态 | 文件 |
|
**更新日期**:2026-07-09
|
||||||
|---|---|---|---|---|
|
**状态**:按活跃问题、设计笔记、RAG 问题集和已归档问题整理
|
||||||
| ISS-001 | Executor 重复召回同一文档 | 中 | 已修复 | [ISS-001-duplicate-retrieval.md](ISS-001-duplicate-retrieval.md) |
|
|
||||||
| ISS-002 | Executor 无约束重复调用 lookup_knowledge | 中 | 已修复 | [ISS-002-executor-unconstrained-lookup.md](ISS-002-executor-unconstrained-lookup.md) |
|
|
||||||
| ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [ISS-003-mvp-design-implementation-review.md](ISS-003-mvp-design-implementation-review.md) |
|
|
||||||
| ISS-004 | Executor 域级检索水位控制(Phase 2) | 低 | 待规划 | [ISS-004-executor-domain-hard-limit.md](ISS-004-executor-domain-hard-limit.md) |
|
|
||||||
| ISS-005 | 证据链补齐与降级契约收敛 | 高 | 已归档 | [ISS-005-evidence-trace-hardening.md](ISS-005-evidence-trace-hardening.md) |
|
|
||||||
| ISS-006 | 固定诊断评测集与回归 Harness | 高 | 已归档 | [ISS-006-diagnosis-eval-harness.md](ISS-006-diagnosis-eval-harness.md) |
|
|
||||||
| ISS-007 | Verifier 证据摘要保真与工具命中质量问题 | 高 | 已实施 | [ISS-007-verifier-evidence-summary-fidelity.md](ISS-007-verifier-evidence-summary-fidelity.md) |
|
|
||||||
| ISS-008 | Executor 窄范围查询越界 | 中 | 已修复 | [ISS-008-executor-narrow-scope-overreach.md](ISS-008-executor-narrow-scope-overreach.md) |
|
|
||||||
| ISS-009 | negative_observation 精确引用 no-evidence 结果 | 中 | 已修复 | [ISS-009-negative-observation-no-evidence-reference.md](ISS-009-negative-observation-no-evidence-reference.md) |
|
|
||||||
| executor-evidence-attribution-hallucination | Executor 证据归因幻觉 | 高 | 待规划 | [executor-evidence-attribution-hallucination.md](executor-evidence-attribution-hallucination.md) |
|
|
||||||
| executor-self-evidence-loop-design-note | Executor 自证循环与证据摘要链路设计记录 | 高 | 已形成方向 | [executor-self-evidence-loop-design-note.md](executor-self-evidence-loop-design-note.md) |
|
|
||||||
| expand-diagnosis-eval-fixtures | 补齐固定诊断评测 fixture 与 baseline | 中 | 已归档 | [expand-diagnosis-eval-fixtures.md](expand-diagnosis-eval-fixtures.md) |
|
|
||||||
| diagnosis-eval-baseline-diff | 诊断评测 baseline diff 与回归判断 | 中 | 已归档 | [diagnosis-eval-baseline-diff.md](diagnosis-eval-baseline-diff.md) |
|
|
||||||
| mvp-demo-interview-runbook | Plan C 面试可复现 Demo 包 | 中 | 已归档 | [mvp-demo-interview-runbook.md](mvp-demo-interview-runbook.md) |
|
|
||||||
|
|
||||||
## RAG 重构计划
|
## 目录约定
|
||||||
|
|
||||||
|
| 目录 | 用途 |
|
||||||
|
|---|---|
|
||||||
|
| [active/](active/) | 仍需要规划或实现的问题 |
|
||||||
|
| [design-notes/](design-notes/) | 已形成方向、用于指导后续实现的设计记录 |
|
||||||
|
| [rag/](rag/) | RAG 子问题集合;多数已合并到 RAG 重构计划 |
|
||||||
|
| [archived/](archived/) | 已修复、已实施或已归档的问题 |
|
||||||
|
|
||||||
|
## 活跃问题
|
||||||
|
|
||||||
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|---|
|
||||||
| rag-refactor-plan | RAG 检索重构计划 | 高 | 待规划 | [rag-refactor-plan.md](rag-refactor-plan.md) |
|
| ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [active/ISS-003-mvp-design-implementation-review.md](active/ISS-003-mvp-design-implementation-review.md) |
|
||||||
|
| ISS-004 | Executor 域级检索水位控制 | 低 | 待规划 | [active/ISS-004-executor-domain-hard-limit.md](active/ISS-004-executor-domain-hard-limit.md) |
|
||||||
|
| executor-evidence-attribution-hallucination | Executor 证据归因幻觉 | 高 | 待规划 | [active/executor-evidence-attribution-hallucination.md](active/executor-evidence-attribution-hallucination.md) |
|
||||||
|
| rag-refactor-plan | RAG 检索重构计划 | 高 | 待规划 | [active/rag-refactor-plan.md](active/rag-refactor-plan.md) |
|
||||||
|
|
||||||
## RAG 检索问题
|
## 设计笔记
|
||||||
|
|
||||||
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|
| 名称 | 标题 | 状态 | 文件 |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|
|
||||||
| chunk-context-reconstruction | RAG 切片上下文重建缺失 | 高 | 已合并到重构计划 | [rag-chunk-context-reconstruction.md](rag-chunk-context-reconstruction.md) |
|
| executor-self-evidence-loop-design-note | Executor 自证循环与证据摘要链路设计记录 | 已形成方向 | [design-notes/executor-self-evidence-loop-design-note.md](design-notes/executor-self-evidence-loop-design-note.md) |
|
||||||
| breadcrumb-embedding-gap | RAG breadcrumb 未参与向量语义 | 高 | 已合并到重构计划 | [rag-breadcrumb-embedding-gap.md](rag-breadcrumb-embedding-gap.md) |
|
| executor-structured-output-v2 | Executor 结构化输出 V2 阶段设计 | 部分已实施,保留为后续改造参考 | [design-notes/executor-structured-output-v2.md](design-notes/executor-structured-output-v2.md) |
|
||||||
| l0-l1-fusion-ranking | RAG L0 和 L1 未真正融合排序 | 中 | 已合并到重构计划 | [rag-l0-l1-fusion-ranking.md](rag-l0-l1-fusion-ranking.md) |
|
|
||||||
| l0-keyword-matching-quality | RAG L0 关键词匹配质量不足 | 中 | 已合并到重构计划 | [rag-l0-keyword-matching-quality.md](rag-l0-keyword-matching-quality.md) |
|
|
||||||
| l1-score-calibration | RAG L1 分数阈值未校准 | 中 | 已合并到重构计划 | [rag-l1-score-calibration.md](rag-l1-score-calibration.md) |
|
|
||||||
| context-packing-and-reranking | RAG 缺少上下文打包和 Rerank | 中 | 已合并到重构计划 | [rag-context-packing-and-reranking.md](rag-context-packing-and-reranking.md) |
|
|
||||||
| upload-chunk-parameter-drift | RAG 上传切片参数未真正生效 | 低 | 已合并到重构计划 | [rag-upload-chunk-parameter-drift.md](rag-upload-chunk-parameter-drift.md) |
|
|
||||||
| query-rewrite-gap | RAG 查询改写能力薄弱 | 中 | 已合并到重构计划 | [rag-query-rewrite-gap.md](rag-query-rewrite-gap.md) |
|
|
||||||
|
|
||||||
## RAG 框架化改造
|
## RAG 问题集
|
||||||
|
|
||||||
| 名称 | 标题 | 严重程度 | 状态 | 文件 |
|
这些问题已经收敛到 [active/rag-refactor-plan.md](active/rag-refactor-plan.md),单个文件保留用于追溯原始问题和设计背景。
|
||||||
|---|---|---|---|---|
|
|
||||||
| spring-ai-vectorstore-migration | RAG 迁移到 Spring AI VectorStore 检索抽象 | 高 | 已合并到重构计划 | [rag-spring-ai-vectorstore-migration.md](rag-spring-ai-vectorstore-migration.md) |
|
| 名称 | 标题 | 状态 | 文件 |
|
||||||
| spring-ai-query-transformer | RAG 接入 Spring AI Query Transformer | 中 | 已合并到重构计划 | [rag-spring-ai-query-transformer.md](rag-spring-ai-query-transformer.md) |
|
|---|---|---|---|
|
||||||
| spring-ai-document-postprocessor | RAG 使用 DocumentPostProcessor 做后处理 | 中 | 已合并到重构计划 | [rag-spring-ai-document-postprocessor.md](rag-spring-ai-document-postprocessor.md) |
|
| breadcrumb-embedding-gap | RAG breadcrumb 未参与向量语义 | 已合并到重构计划 | [rag/rag-breadcrumb-embedding-gap.md](rag/rag-breadcrumb-embedding-gap.md) |
|
||||||
| l0-domain-entity-hint | RAG 将 L0 降级为领域和实体 Hint | 中 | 已合并到重构计划 | [rag-l0-domain-entity-hint.md](rag-l0-domain-entity-hint.md) |
|
| chunk-context-reconstruction | RAG 切片上下文重建缺失 | 已合并到重构计划 | [rag/rag-chunk-context-reconstruction.md](rag/rag-chunk-context-reconstruction.md) |
|
||||||
| spring-ai-advisor-boundary | RAG 明确 Spring AI Advisor 与 Agent Tool 的边界 | 中 | 已合并到重构计划 | [rag-spring-ai-advisor-boundary.md](rag-spring-ai-advisor-boundary.md) |
|
| context-packing-and-reranking | RAG 缺少上下文打包和 Rerank | 已合并到重构计划 | [rag/rag-context-packing-and-reranking.md](rag/rag-context-packing-and-reranking.md) |
|
||||||
|
| l0-domain-entity-hint | RAG 将 L0 降级为领域和实体 Hint | 已合并到重构计划 | [rag/rag-l0-domain-entity-hint.md](rag/rag-l0-domain-entity-hint.md) |
|
||||||
|
| l0-keyword-matching-quality | RAG L0 关键词匹配质量不足 | 已合并到重构计划 | [rag/rag-l0-keyword-matching-quality.md](rag/rag-l0-keyword-matching-quality.md) |
|
||||||
|
| l0-l1-fusion-ranking | RAG L0 和 L1 未真正融合排序 | 已合并到重构计划 | [rag/rag-l0-l1-fusion-ranking.md](rag/rag-l0-l1-fusion-ranking.md) |
|
||||||
|
| l1-score-calibration | RAG L1 分数阈值未校准 | 已合并到重构计划 | [rag/rag-l1-score-calibration.md](rag/rag-l1-score-calibration.md) |
|
||||||
|
| query-rewrite-gap | RAG 查询改写能力薄弱 | 已合并到重构计划 | [rag/rag-query-rewrite-gap.md](rag/rag-query-rewrite-gap.md) |
|
||||||
|
| spring-ai-advisor-boundary | Spring AI Advisor 与 Agent Tool 边界 | 已合并到重构计划 | [rag/rag-spring-ai-advisor-boundary.md](rag/rag-spring-ai-advisor-boundary.md) |
|
||||||
|
| spring-ai-document-postprocessor | 使用 DocumentPostProcessor 做后处理 | 已合并到重构计划 | [rag/rag-spring-ai-document-postprocessor.md](rag/rag-spring-ai-document-postprocessor.md) |
|
||||||
|
| spring-ai-query-transformer | 接入 Spring AI Query Transformer | 已合并到重构计划 | [rag/rag-spring-ai-query-transformer.md](rag/rag-spring-ai-query-transformer.md) |
|
||||||
|
| spring-ai-vectorstore-migration | 迁移到 Spring AI VectorStore 检索抽象 | 已合并到重构计划 | [rag/rag-spring-ai-vectorstore-migration.md](rag/rag-spring-ai-vectorstore-migration.md) |
|
||||||
|
| upload-chunk-parameter-drift | 上传切片参数未真正生效 | 已合并到重构计划 | [rag/rag-upload-chunk-parameter-drift.md](rag/rag-upload-chunk-parameter-drift.md) |
|
||||||
|
|
||||||
|
## 已归档问题
|
||||||
|
|
||||||
|
| 名称 | 标题 | 状态 | 文件 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| ISS-001 | Executor 重复召回同一文档 | 已修复 | [archived/ISS-001-duplicate-retrieval.md](archived/ISS-001-duplicate-retrieval.md) |
|
||||||
|
| ISS-002 | Executor 无约束重复调用 lookup_knowledge | 已修复 | [archived/ISS-002-executor-unconstrained-lookup.md](archived/ISS-002-executor-unconstrained-lookup.md) |
|
||||||
|
| ISS-005 | 证据链补齐与降级契约收敛 | 已归档 | [archived/ISS-005-evidence-trace-hardening.md](archived/ISS-005-evidence-trace-hardening.md) |
|
||||||
|
| ISS-006 | 固定诊断评测集与回归 Harness | 已归档 | [archived/ISS-006-diagnosis-eval-harness.md](archived/ISS-006-diagnosis-eval-harness.md) |
|
||||||
|
| ISS-007 | Verifier 证据摘要保真与工具命中质量问题 | 已实施 | [archived/ISS-007-verifier-evidence-summary-fidelity.md](archived/ISS-007-verifier-evidence-summary-fidelity.md) |
|
||||||
|
| ISS-008 | Executor 窄范围查询越界 | 已修复 | [archived/ISS-008-executor-narrow-scope-overreach.md](archived/ISS-008-executor-narrow-scope-overreach.md) |
|
||||||
|
| ISS-009 | negative_observation 精确引用 no-evidence 结果 | 已修复 | [archived/ISS-009-negative-observation-no-evidence-reference.md](archived/ISS-009-negative-observation-no-evidence-reference.md) |
|
||||||
|
| diagnosis-eval-baseline-diff | 诊断评测 baseline diff 与回归判断 | 已归档 | [archived/diagnosis-eval-baseline-diff.md](archived/diagnosis-eval-baseline-diff.md) |
|
||||||
|
| expand-diagnosis-eval-fixtures | 补齐固定诊断评测 fixture 与 baseline | 已归档 | [archived/expand-diagnosis-eval-fixtures.md](archived/expand-diagnosis-eval-fixtures.md) |
|
||||||
|
| mvp-demo-interview-runbook | Plan C 面试可复现 Demo 包 | 已归档 | [archived/mvp-demo-interview-runbook.md](archived/mvp-demo-interview-runbook.md) |
|
||||||
|
|||||||
@@ -347,19 +347,19 @@ RAG、Agent、AIOps、数据库记录互相关联,必须分阶段推进,每
|
|||||||
|
|
||||||
本计划合并以下问题和改造方向:
|
本计划合并以下问题和改造方向:
|
||||||
|
|
||||||
- [rag-chunk-context-reconstruction.md](rag-chunk-context-reconstruction.md)
|
- [rag-chunk-context-reconstruction.md](../rag/rag-chunk-context-reconstruction.md)
|
||||||
- [rag-breadcrumb-embedding-gap.md](rag-breadcrumb-embedding-gap.md)
|
- [rag-breadcrumb-embedding-gap.md](../rag/rag-breadcrumb-embedding-gap.md)
|
||||||
- [rag-l0-l1-fusion-ranking.md](rag-l0-l1-fusion-ranking.md)
|
- [rag-l0-l1-fusion-ranking.md](../rag/rag-l0-l1-fusion-ranking.md)
|
||||||
- [rag-l0-keyword-matching-quality.md](rag-l0-keyword-matching-quality.md)
|
- [rag-l0-keyword-matching-quality.md](../rag/rag-l0-keyword-matching-quality.md)
|
||||||
- [rag-l1-score-calibration.md](rag-l1-score-calibration.md)
|
- [rag-l1-score-calibration.md](../rag/rag-l1-score-calibration.md)
|
||||||
- [rag-context-packing-and-reranking.md](rag-context-packing-and-reranking.md)
|
- [rag-context-packing-and-reranking.md](../rag/rag-context-packing-and-reranking.md)
|
||||||
- [rag-upload-chunk-parameter-drift.md](rag-upload-chunk-parameter-drift.md)
|
- [rag-upload-chunk-parameter-drift.md](../rag/rag-upload-chunk-parameter-drift.md)
|
||||||
- [rag-query-rewrite-gap.md](rag-query-rewrite-gap.md)
|
- [rag-query-rewrite-gap.md](../rag/rag-query-rewrite-gap.md)
|
||||||
- [rag-spring-ai-vectorstore-migration.md](rag-spring-ai-vectorstore-migration.md)
|
- [rag-spring-ai-vectorstore-migration.md](../rag/rag-spring-ai-vectorstore-migration.md)
|
||||||
- [rag-spring-ai-query-transformer.md](rag-spring-ai-query-transformer.md)
|
- [rag-spring-ai-query-transformer.md](../rag/rag-spring-ai-query-transformer.md)
|
||||||
- [rag-spring-ai-document-postprocessor.md](rag-spring-ai-document-postprocessor.md)
|
- [rag-spring-ai-document-postprocessor.md](../rag/rag-spring-ai-document-postprocessor.md)
|
||||||
- [rag-l0-domain-entity-hint.md](rag-l0-domain-entity-hint.md)
|
- [rag-l0-domain-entity-hint.md](../rag/rag-l0-domain-entity-hint.md)
|
||||||
- [rag-spring-ai-advisor-boundary.md](rag-spring-ai-advisor-boundary.md)
|
- [rag-spring-ai-advisor-boundary.md](../rag/rag-spring-ai-advisor-boundary.md)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
+1
-1
@@ -4,7 +4,7 @@
|
|||||||
**严重程度**:中(影响 token 消耗和上下文质量,不影响功能正确性)
|
**严重程度**:中(影响 token 消耗和上下文质量,不影响功能正确性)
|
||||||
**发现时间**:2026-06-30
|
**发现时间**:2026-06-30
|
||||||
**修复版本**:session-dedup-knowledge-map
|
**修复版本**:session-dedup-knowledge-map
|
||||||
**历史架构文档**:[会话级去重与知识域地图](../architecture/archive/2026-07-05-legacy/session-dedup-knowledge-map.md)
|
**历史架构文档**:[会话级去重与知识域地图](../../architecture/archive/2026-07-05-legacy/session-dedup-knowledge-map.md)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
# Agent 步骤表:agent_step
|
||||||
|
|
||||||
|
**状态**:当前表
|
||||||
|
**来源**:`V005__create_session_storage.sql`、`V006__fix_agent_step_json_to_text.sql`、`AgentStep`
|
||||||
|
|
||||||
|
## 定位
|
||||||
|
|
||||||
|
`agent_step` 记录一次诊断过程中每个 Agent 步骤的模型输入、输出、耗时和 Token 消耗。页面展示执行链路时应优先按 `step_index` 排序。
|
||||||
|
|
||||||
|
## 字段
|
||||||
|
|
||||||
|
| 字段 | 类型 | 必填 | 说明 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `id` | BIGINT | 是 | 自增主键 |
|
||||||
|
| `session_id` | VARCHAR(64) | 是 | 关联 `diagnosis_session.session_id` |
|
||||||
|
| `step_index` | INT | 是 | 步骤序号,从 0 开始 |
|
||||||
|
| `agent_name` | VARCHAR(32) | 是 | Agent 名称,例如 planner、executor、verifier、composer |
|
||||||
|
| `model_input` | TEXT | 否 | 模型输入摘要;`V006` 已从 JSON 改为 TEXT |
|
||||||
|
| `model_output` | TEXT | 否 | 模型输出摘要;`V006` 已从 JSON 改为 TEXT |
|
||||||
|
| `thought` | TEXT | 否 | Agent 思考过程或调试摘要 |
|
||||||
|
| `has_tool_call` | BOOLEAN | 否 | 本步骤是否触发工具调用 |
|
||||||
|
| `duration_ms` | INT | 否 | 本步骤耗时 |
|
||||||
|
| `token_count` | INT | 否 | 本步骤 Token 消耗 |
|
||||||
|
| `created_at` | DATETIME | 是 | 创建时间 |
|
||||||
|
|
||||||
|
## 索引
|
||||||
|
|
||||||
|
| 索引 | 字段 | 用途 |
|
||||||
|
|---|---|---|
|
||||||
|
| `idx_session_step` | `session_id, step_index` | Trace 页面按会话和步骤顺序查询 |
|
||||||
|
| `idx_agent_name` | `agent_name` | 按 Agent 类型筛选 |
|
||||||
|
|
||||||
|
## 关系
|
||||||
|
|
||||||
|
- `agent_step.session_id` 逻辑关联 `diagnosis_session.session_id`。
|
||||||
|
- `tool_invocation.step_id` 可关联 `agent_step.id`,但当前允许为空且不强制外键。
|
||||||
|
|
||||||
|
## 注意点
|
||||||
|
|
||||||
|
- 前端展示步骤时应按 `step_index` 排序,而不是按 `created_at` 或数据库返回顺序。
|
||||||
|
- Verifier 应在 Executor 循环完成后出现;如果 `step_index` 中 Verifier 提前,通常意味着编排或记录顺序有问题。
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
# MVP 数据表索引
|
||||||
|
|
||||||
|
**更新日期**:2026-07-09
|
||||||
|
**状态**:当前表文档入口
|
||||||
|
|
||||||
|
本目录保存当前 MVP 使用的数据表说明。详细结构以 Flyway migration 和实体类为准;本目录用于面试讲解、排查索引和快速理解数据流。
|
||||||
|
|
||||||
|
## 当前表
|
||||||
|
|
||||||
|
| 表 | 用途 | 文档 |
|
||||||
|
|---|---|---|
|
||||||
|
| `diagnosis_session` | 会话级主记录,保存 query、状态、最终答案和自评估 | [诊断会话表-diagnosis_session.md](诊断会话表-diagnosis_session.md) |
|
||||||
|
| `agent_step` | Agent 步骤记录,按 `step_index` 回放执行链路 | [Agent步骤表-agent_step.md](Agent步骤表-agent_step.md) |
|
||||||
|
| `tool_invocation` | 工具调用记录,支撑 Trace、Verifier 和评测 | [工具调用表-tool_invocation.md](工具调用表-tool_invocation.md) |
|
||||||
|
| `api_document` | 知识库文档元数据,和向量库 chunk 通过 `doc_id` 关联 | [文档元数据表-api_document.md](文档元数据表-api_document.md) |
|
||||||
|
| `knowledge_domain` | 知识域元数据,支撑 RAG domain hint 和检索策略 | [知识域表-knowledge_domain.md](知识域表-knowledge_domain.md) |
|
||||||
|
| `case_library` | 用户反馈沉淀出的高质量诊断案例 | [案例库表-case_library.md](案例库表-case_library.md) |
|
||||||
|
|
||||||
|
## 已归档表
|
||||||
|
|
||||||
|
| 表 | 归档原因 | 文档 |
|
||||||
|
|---|---|---|
|
||||||
|
| `diagnosis_record` | 已由 `V007` 删除,被 `diagnosis_session + agent_step + tool_invocation` 替代 | [archive/2026-07-09-doc-cleanup/旧诊断记录表-diagnosis_record.md](archive/2026-07-09-doc-cleanup/旧诊断记录表-diagnosis_record.md) |
|
||||||
|
|
||||||
|
## 核心关系
|
||||||
|
|
||||||
|
```text
|
||||||
|
diagnosis_session.session_id
|
||||||
|
-> agent_step.session_id
|
||||||
|
-> tool_invocation.session_id
|
||||||
|
-> case_library.diagnosis_id
|
||||||
|
|
||||||
|
api_document.doc_id
|
||||||
|
-> vector chunk metadata.docId / doc_id
|
||||||
|
|
||||||
|
knowledge_domain.domain_id
|
||||||
|
-> api_document metadata.category / vector chunk metadata.category
|
||||||
|
```
|
||||||
|
|
||||||
|
当前实现主要使用逻辑关联,不依赖数据库外键。
|
||||||
@@ -1,332 +0,0 @@
|
|||||||
# api_document - 文档元数据表
|
|
||||||
|
|
||||||
## 表定位
|
|
||||||
|
|
||||||
**文档管理表**:管理接口文档的元信息,不负责文档检索(检索由 Milvus 负责)
|
|
||||||
|
|
||||||
## 设计理念
|
|
||||||
|
|
||||||
### 文档管理,不是文档检索
|
|
||||||
|
|
||||||
**核心定位**:
|
|
||||||
- MySQL 负责文档元数据管理(状态、版本、去重)
|
|
||||||
- Milvus 负责文档内容存储和检索
|
|
||||||
- 通过 doc_id 关联两者
|
|
||||||
|
|
||||||
**MVP版本原则**:
|
|
||||||
- ✅ 最简字段,满足基本管理需求
|
|
||||||
- ✅ 文件去重(基于 file_hash)
|
|
||||||
- ✅ 状态追踪(索引进度)
|
|
||||||
- ✅ 硬删除(同步删除 Milvus 数据)
|
|
||||||
- ❌ 暂不支持:软删除、启用开关、版本管理(Phase 2)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 表结构(MVP版)
|
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE TABLE api_document (
|
|
||||||
-- 主键
|
|
||||||
id BIGINT PRIMARY KEY AUTO_INCREMENT,
|
|
||||||
doc_id VARCHAR(64) UNIQUE NOT NULL COMMENT '文档唯一ID(UUID),关联Milvus',
|
|
||||||
|
|
||||||
-- 文档分类
|
|
||||||
fault_category VARCHAR(32) DEFAULT 'EXTERNAL_API' COMMENT '文档类别',
|
|
||||||
fault_source VARCHAR(128) COMMENT '文档归属(省份/服务名)',
|
|
||||||
api_name VARCHAR(128) COMMENT '接口名称',
|
|
||||||
version VARCHAR(32) DEFAULT 'v1.0' COMMENT '文档版本',
|
|
||||||
|
|
||||||
-- 文件信息
|
|
||||||
file_name VARCHAR(256) NOT NULL COMMENT '原始文件名',
|
|
||||||
file_path VARCHAR(512) COMMENT '文件存储路径',
|
|
||||||
file_hash VARCHAR(64) COMMENT '文件MD5 hash(用于去重)',
|
|
||||||
file_size BIGINT COMMENT '文件大小(字节)',
|
|
||||||
|
|
||||||
-- 索引状态
|
|
||||||
status VARCHAR(16) DEFAULT 'PENDING' COMMENT '索引状态(PENDING/PROCESSING/INDEXED/FAILED)',
|
|
||||||
chunk_count INT DEFAULT 0 COMMENT '分块数量',
|
|
||||||
error_message TEXT COMMENT '失败原因',
|
|
||||||
|
|
||||||
-- 时间字段
|
|
||||||
indexed_at DATETIME COMMENT '索引完成时间',
|
|
||||||
created_at DATETIME DEFAULT CURRENT_TIMESTAMP,
|
|
||||||
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
|
|
||||||
|
|
||||||
-- 索引
|
|
||||||
UNIQUE INDEX uk_file_hash (file_hash),
|
|
||||||
INDEX idx_doc_id (doc_id),
|
|
||||||
INDEX idx_fault_source (fault_source),
|
|
||||||
INDEX idx_status (status),
|
|
||||||
INDEX idx_created_at (created_at)
|
|
||||||
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COMMENT='文档元数据表(MVP版)';
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 字段说明
|
|
||||||
|
|
||||||
| 字段 | 类型 | 必填 | 说明 |
|
|
||||||
|------|------|------|------|
|
|
||||||
| doc_id | VARCHAR(64) | 是 | **核心**:文档唯一ID,关联 Milvus |
|
|
||||||
| fault_category | VARCHAR(32) | 否 | 文档类别 |
|
|
||||||
| fault_source | VARCHAR(128) | 否 | 文档归属(省份/服务名)|
|
|
||||||
| api_name | VARCHAR(128) | 否 | 接口名称 |
|
|
||||||
| version | VARCHAR(32) | 否 | 文档版本 |
|
|
||||||
| file_name | VARCHAR(256) | 是 | 原始文件名 |
|
|
||||||
| file_path | VARCHAR(512) | 否 | 文件存储路径 |
|
|
||||||
| file_hash | VARCHAR(64) | 否 | **去重关键**:文件MD5 |
|
|
||||||
| file_size | BIGINT | 否 | 文件大小 |
|
|
||||||
| status | VARCHAR(16) | 是 | **状态追踪**:PENDING/PROCESSING/INDEXED/FAILED |
|
|
||||||
| chunk_count | INT | 否 | 分块数量 |
|
|
||||||
| error_message | TEXT | 否 | 失败原因 |
|
|
||||||
| indexed_at | DATETIME | 否 | 索引完成时间 |
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 核心设计决策
|
|
||||||
|
|
||||||
### 1. doc_id:MySQL 与 Milvus 的桥梁
|
|
||||||
|
|
||||||
```
|
|
||||||
作用:
|
|
||||||
- MySQL:通过 doc_id 管理文档元数据
|
|
||||||
- Milvus:每个 chunk 的 metadata 中携带 doc_id
|
|
||||||
|
|
||||||
关联关系:
|
|
||||||
api_document (MySQL)
|
|
||||||
doc_id: doc-001
|
|
||||||
↓ 1:N
|
|
||||||
Milvus chunks
|
|
||||||
chunk_1: {doc_id: 'doc-001', text: '...', vector: [...]}
|
|
||||||
chunk_2: {doc_id: 'doc-001', text: '...', vector: [...]}
|
|
||||||
|
|
||||||
管理操作:
|
|
||||||
- 删除文档:
|
|
||||||
DELETE FROM milvus_collection WHERE metadata["doc_id"] == 'doc-001';
|
|
||||||
DELETE FROM api_document WHERE doc_id = 'doc-001';
|
|
||||||
```
|
|
||||||
|
|
||||||
### 2. file_hash:文件去重
|
|
||||||
|
|
||||||
```
|
|
||||||
去重流程:
|
|
||||||
1. 用户上传文件
|
|
||||||
↓
|
|
||||||
2. 计算文件 MD5
|
|
||||||
file_hash = md5(file_content)
|
|
||||||
↓
|
|
||||||
3. 检查是否已存在
|
|
||||||
SELECT * FROM api_document WHERE file_hash = 'abc123...';
|
|
||||||
↓
|
|
||||||
4a. 如果存在 → 提示"文档已存在"
|
|
||||||
4b. 如果不存在 → 继续导入
|
|
||||||
|
|
||||||
唯一约束:UNIQUE INDEX uk_file_hash (file_hash)
|
|
||||||
```
|
|
||||||
|
|
||||||
### 3. status:状态追踪
|
|
||||||
|
|
||||||
```
|
|
||||||
状态流转:
|
|
||||||
PENDING (待处理)
|
|
||||||
↓
|
|
||||||
PROCESSING (处理中)
|
|
||||||
↓ 成功
|
|
||||||
INDEXED (已索引)
|
|
||||||
↓ 失败
|
|
||||||
FAILED (失败)
|
|
||||||
|
|
||||||
用途:
|
|
||||||
- 批量导入时监控进度
|
|
||||||
- 失败重试
|
|
||||||
- 统计索引成功率
|
|
||||||
```
|
|
||||||
|
|
||||||
### 4. 硬删除策略(MVP)
|
|
||||||
|
|
||||||
```
|
|
||||||
删除文档时:
|
|
||||||
1. 删除 Milvus 中的所有分块
|
|
||||||
2. 删除 MySQL 元数据
|
|
||||||
3. 可选:删除原始文件
|
|
||||||
|
|
||||||
特点:
|
|
||||||
- 简单直接
|
|
||||||
- 数据彻底删除
|
|
||||||
- 不可恢复(需谨慎)
|
|
||||||
|
|
||||||
Phase 2 可增强:
|
|
||||||
- 软删除(archived_at)
|
|
||||||
- 启用开关(enabled)
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 数据流
|
|
||||||
|
|
||||||
### 场景1:导入新文档
|
|
||||||
|
|
||||||
```
|
|
||||||
1. 用户上传文件
|
|
||||||
↓
|
|
||||||
2. 计算 hash
|
|
||||||
↓
|
|
||||||
3. 检查去重(MySQL)
|
|
||||||
↓
|
|
||||||
4. 插入元数据(status=PROCESSING)
|
|
||||||
↓
|
|
||||||
5. 后台处理:解析 → 分块 → 向量化 → 存入 Milvus
|
|
||||||
↓
|
|
||||||
6. 更新状态(status=INDEXED, chunk_count=15)
|
|
||||||
```
|
|
||||||
|
|
||||||
### 场景2:删除文档
|
|
||||||
|
|
||||||
```
|
|
||||||
1. 用户删除文档
|
|
||||||
↓
|
|
||||||
2. 删除 Milvus 数据(WHERE metadata["doc_id"] == 'xxx')
|
|
||||||
↓
|
|
||||||
3. 删除 MySQL 元数据
|
|
||||||
↓
|
|
||||||
4. 可选:删除原始文件
|
|
||||||
```
|
|
||||||
|
|
||||||
### 场景3:重新索引
|
|
||||||
|
|
||||||
```
|
|
||||||
1. 删除旧数据(Milvus + MySQL)
|
|
||||||
↓
|
|
||||||
2. 重新导入(同场景1)
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 典型查询
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- 查看文档列表
|
|
||||||
SELECT doc_id, file_name, version, status, chunk_count, indexed_at
|
|
||||||
FROM api_document
|
|
||||||
WHERE fault_source = '广东'
|
|
||||||
AND status = 'INDEXED'
|
|
||||||
ORDER BY indexed_at DESC;
|
|
||||||
|
|
||||||
-- 查询失败的文档
|
|
||||||
SELECT doc_id, file_name, error_message
|
|
||||||
FROM api_document
|
|
||||||
WHERE status = 'FAILED';
|
|
||||||
|
|
||||||
-- 统计各状态文档数量
|
|
||||||
SELECT status, COUNT(*) as count
|
|
||||||
FROM api_document
|
|
||||||
GROUP BY status;
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 与 Milvus 的协作
|
|
||||||
|
|
||||||
### Milvus Collection Schema
|
|
||||||
|
|
||||||
```python
|
|
||||||
{
|
|
||||||
"collection_name": "api_doc_collection",
|
|
||||||
"fields": [
|
|
||||||
{"name": "id", "type": "VARCHAR", "is_primary": true},
|
|
||||||
{"name": "content", "type": "VARCHAR"},
|
|
||||||
{"name": "vector", "type": "FLOAT_VECTOR", "dim": 1536},
|
|
||||||
{"name": "metadata", "type": "JSON"}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
|
|
||||||
# metadata 结构
|
|
||||||
{
|
|
||||||
"doc_id": "doc-001", # 关联 MySQL
|
|
||||||
"_source": "/path/to/file",
|
|
||||||
"_file_name": "xxx.docx",
|
|
||||||
"chunkIndex": 0,
|
|
||||||
"totalChunks": 15
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
### Java 代码示例
|
|
||||||
|
|
||||||
```java
|
|
||||||
// 插入时携带 doc_id
|
|
||||||
Map<String, Object> metadata = new HashMap<>();
|
|
||||||
metadata.put("doc_id", docId); // 关联 MySQL
|
|
||||||
metadata.put("_source", filePath);
|
|
||||||
metadata.put("chunkIndex", chunkIndex);
|
|
||||||
|
|
||||||
// 删除文档的所有分块
|
|
||||||
String expr = String.format("metadata[\"doc_id\"] == \"%s\"", docId);
|
|
||||||
milvusClient.delete(DeleteParam.newBuilder()
|
|
||||||
.withCollectionName(COLLECTION_NAME)
|
|
||||||
.withExpr(expr)
|
|
||||||
.build());
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 数据示例
|
|
||||||
|
|
||||||
```sql
|
|
||||||
-- 外部接口文档
|
|
||||||
INSERT INTO api_document VALUES
|
|
||||||
(1, 'doc-001', 'EXTERNAL_API', '广东', '社保查询', 'v2.1',
|
|
||||||
'广东社保查询v2.1.docx', '/docs/guangdong/social-v2.1.docx',
|
|
||||||
'abc123...', 1048576,
|
|
||||||
'INDEXED', 15, NULL, '2024-06-15 10:30:00', NOW(), NOW());
|
|
||||||
|
|
||||||
-- 内部服务文档
|
|
||||||
INSERT INTO api_document VALUES
|
|
||||||
(2, 'doc-002', 'INTERNAL_ERROR', 'order-service', '订单服务API', 'v1.0',
|
|
||||||
'订单服务API文档.pdf', '/docs/internal/order-service-api.pdf',
|
|
||||||
'def456...', 2097152,
|
|
||||||
'INDEXED', 20, NULL, '2024-06-14 15:20:00', NOW(), NOW());
|
|
||||||
|
|
||||||
-- 处理失败的文档
|
|
||||||
INSERT INTO api_document VALUES
|
|
||||||
(3, 'doc-003', 'EXTERNAL_API', '江苏', '公积金查询', 'v1.5',
|
|
||||||
'江苏公积金查询.html', '/docs/jiangsu/fund-v1.5.html',
|
|
||||||
'ghi789...', 512000,
|
|
||||||
'FAILED', 0, '不支持HTML格式', NULL, NOW(), NOW());
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 数据量预估
|
|
||||||
|
|
||||||
```
|
|
||||||
预估:100-200 条
|
|
||||||
- 外部接口文档:50-100 条
|
|
||||||
- 内部服务文档:20-50 条
|
|
||||||
- 其他文档:30-50 条
|
|
||||||
|
|
||||||
存储:
|
|
||||||
- 单条记录:约 1KB
|
|
||||||
- 200 条:约 200KB
|
|
||||||
|
|
||||||
结论:数据量很小
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## MVP 版本的简化
|
|
||||||
|
|
||||||
```
|
|
||||||
Phase 1(当前):
|
|
||||||
✅ 基础字段和表结构
|
|
||||||
✅ 文件去重(file_hash)
|
|
||||||
✅ 状态追踪(status)
|
|
||||||
✅ 硬删除
|
|
||||||
✅ 通过 doc_id 关联 Milvus
|
|
||||||
|
|
||||||
Phase 2(未来增强):
|
|
||||||
❌ enabled(启用开关)
|
|
||||||
❌ archived_at(软删除)
|
|
||||||
❌ batch_id(批次管理)
|
|
||||||
❌ status 细化
|
|
||||||
❌ tags(标签分类)
|
|
||||||
```
|
|
||||||
+3
-1
@@ -1,4 +1,6 @@
|
|||||||
# diagnosis_record - 诊断记录表
|
# diagnosis_record - 旧诊断记录表
|
||||||
|
|
||||||
|
> 归档说明:`diagnosis_record` 已在 `V007__drop_diagnosis_record.sql` 中删除,当前主模型是 `diagnosis_session + agent_step + tool_invocation`。本文只用于追溯早期设计。
|
||||||
|
|
||||||
## 表定位
|
## 表定位
|
||||||
|
|
||||||
@@ -1,265 +0,0 @@
|
|||||||
# case_library - 案例库表
|
|
||||||
|
|
||||||
## 表定位
|
|
||||||
|
|
||||||
**知识沉淀表**:存储高质量诊断案例,支持相似案例推荐
|
|
||||||
|
|
||||||
## 设计理念
|
|
||||||
|
|
||||||
### 知识沉淀,系统越用越智能
|
|
||||||
|
|
||||||
**核心价值**:
|
|
||||||
- 质量过滤:只存储高质量案例(成功诊断 + 用户反馈有用)
|
|
||||||
- 知识沉淀:历史诊断经验可复用
|
|
||||||
- 提升准确率:相似问题提供历史参考
|
|
||||||
- 加速诊断:快速推荐相似案例
|
|
||||||
|
|
||||||
**MVP版本设计原则**:
|
|
||||||
- ✅ 能用:满足基本案例推荐功能
|
|
||||||
- ✅ 简单:字段不多,逻辑清晰
|
|
||||||
- ✅ 可扩展:后续可增加字段
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 表结构(MVP版)
|
|
||||||
|
|
||||||
```sql
|
|
||||||
CREATE TABLE case_library (
|
|
||||||
-- 主键
|
|
||||||
id BIGINT PRIMARY KEY AUTO_INCREMENT,
|
|
||||||
case_id VARCHAR(64) UNIQUE NOT NULL COMMENT '案例唯一ID(UUID)',
|
|
||||||
|
|
||||||
-- 来源关联
|
|
||||||
diagnosis_id VARCHAR(64) COMMENT '关联诊断记录(可选,人工录入时为空)',
|
|
||||||
source_type VARCHAR(16) DEFAULT 'AUTO' COMMENT '来源类型(AUTO:自动生成/MANUAL:人工录入)',
|
|
||||||
|
|
||||||
-- 案例分类
|
|
||||||
fault_category VARCHAR(32) COMMENT '故障类别(EXTERNAL_API/INTERNAL_ERROR/DATABASE...)',
|
|
||||||
fault_source VARCHAR(128) COMMENT '故障源(省份/服务名/类名...)',
|
|
||||||
fault_target VARCHAR(256) COMMENT '故障目标(接口URL/方法名/SQL...)',
|
|
||||||
error_code VARCHAR(64) COMMENT '错误码',
|
|
||||||
|
|
||||||
-- 案例内容
|
|
||||||
title VARCHAR(256) NOT NULL COMMENT '案例标题(简短描述)',
|
|
||||||
root_cause TEXT NOT NULL COMMENT '根因分析',
|
|
||||||
solution TEXT NOT NULL COMMENT '解决方案',
|
|
||||||
|
|
||||||
-- 简单统计
|
|
||||||
reference_count INT DEFAULT 0 COMMENT '引用次数(被推荐的次数)',
|
|
||||||
|
|
||||||
-- 元数据
|
|
||||||
created_by VARCHAR(64) COMMENT '创建人',
|
|
||||||
created_at DATETIME DEFAULT CURRENT_TIMESTAMP,
|
|
||||||
updated_at DATETIME DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
|
|
||||||
|
|
||||||
-- 索引
|
|
||||||
INDEX idx_fault_category (fault_category),
|
|
||||||
INDEX idx_error_code (error_code),
|
|
||||||
INDEX idx_fault_source (fault_source),
|
|
||||||
INDEX idx_fault_target (fault_target(100)),
|
|
||||||
INDEX idx_diagnosis_id (diagnosis_id),
|
|
||||||
INDEX idx_reference_count (reference_count),
|
|
||||||
INDEX idx_created_at (created_at)
|
|
||||||
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COMMENT='案例库表(MVP版)';
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 字段说明
|
|
||||||
|
|
||||||
| 字段 | 类型 | 必填 | 说明 |
|
|
||||||
|------|------|------|------|
|
|
||||||
| case_id | VARCHAR(64) | 是 | 案例唯一标识(UUID)|
|
|
||||||
| diagnosis_id | VARCHAR(64) | 否 | 关联诊断记录(人工录入时为空)|
|
|
||||||
| source_type | VARCHAR(16) | 是 | 来源:AUTO(自动)/MANUAL(人工)|
|
|
||||||
| fault_category | VARCHAR(32) | 否 | 故障类别 |
|
|
||||||
| fault_source | VARCHAR(128) | 否 | 故障源 |
|
|
||||||
| fault_target | VARCHAR(256) | 否 | 故障目标(与 diagnosis_record 一致)|
|
|
||||||
| error_code | VARCHAR(64) | 否 | 错误码 |
|
|
||||||
| title | VARCHAR(256) | 是 | 案例标题 |
|
|
||||||
| root_cause | TEXT | 是 | 根因分析(核心内容)|
|
|
||||||
| solution | TEXT | 是 | 解决方案(核心内容)|
|
|
||||||
| reference_count | INT | 是 | 引用次数(用于排序)|
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 核心设计决策
|
|
||||||
|
|
||||||
### 1. 案例来源
|
|
||||||
|
|
||||||
```
|
|
||||||
来源1:自动生成(source_type=AUTO)
|
|
||||||
├─ 触发条件:诊断成功 + 用户反馈"有用"
|
|
||||||
├─ 关联诊断:diagnosis_id 不为空
|
|
||||||
└─ 质量保证:用户验证过
|
|
||||||
|
|
||||||
来源2:人工录入(source_type=MANUAL)
|
|
||||||
├─ 运维团队总结的经典案例
|
|
||||||
├─ diagnosis_id 为空
|
|
||||||
└─ 质量最高
|
|
||||||
|
|
||||||
注意:诊断失败或用户反馈"无用"的不自动生成案例
|
|
||||||
```
|
|
||||||
|
|
||||||
### 2. 简化的评分机制(MVP)
|
|
||||||
|
|
||||||
```
|
|
||||||
MVP版本:只按 reference_count 排序
|
|
||||||
- 引用次数多的排前面
|
|
||||||
- 简单有效
|
|
||||||
|
|
||||||
Phase 2 可增强:
|
|
||||||
- 增加 useful_count(用户反馈有用次数)
|
|
||||||
- 增加 score(综合评分)
|
|
||||||
- 增加 is_featured(人工标记的经典案例)
|
|
||||||
```
|
|
||||||
|
|
||||||
### 3. 与 diagnosis_record 的关系
|
|
||||||
|
|
||||||
```
|
|
||||||
关系:一对一(可选)
|
|
||||||
- 一次诊断 → 可以生成一个案例
|
|
||||||
- 通过 diagnosis_id 关联
|
|
||||||
- diagnosis_id 可为空(人工录入案例)
|
|
||||||
|
|
||||||
流程:
|
|
||||||
diagnosis_record(成功)
|
|
||||||
↓
|
|
||||||
用户反馈"有用"
|
|
||||||
↓
|
|
||||||
自动生成 case_library
|
|
||||||
↓
|
|
||||||
后续可人工修正、合并相似案例
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 数据示例
|
|
||||||
|
|
||||||
### 示例1:外部接口故障案例
|
|
||||||
```sql
|
|
||||||
INSERT INTO case_library VALUES
|
|
||||||
(1, 'case-001', 'diag-001', 'AUTO', 'EXTERNAL_API', '广东', '/api/v1/guangdong/social-security', '40003',
|
|
||||||
'广东社保查询idCard字段缺失',
|
|
||||||
'请求报文中未传入idCard字段,导致参数校验失败',
|
|
||||||
'前端表单增加idCard必填校验;后端增加参数校验提示',
|
|
||||||
15, 'system', NOW(), NOW());
|
|
||||||
```
|
|
||||||
|
|
||||||
### 示例2:内部错误案例
|
|
||||||
```sql
|
|
||||||
INSERT INTO case_library VALUES
|
|
||||||
(2, 'case-002', 'diag-045', 'AUTO', 'INTERNAL_ERROR', 'order-service', 'OrderController.createOrder()', 'NullPointerException',
|
|
||||||
'订单服务创建订单空指针异常',
|
|
||||||
'OrderController.createOrder()方法中user对象为null,未做空判断',
|
|
||||||
'在第45行添加空判断:if (user == null) throw new BizException("用户信息不存在")',
|
|
||||||
8, 'system', NOW(), NOW());
|
|
||||||
```
|
|
||||||
|
|
||||||
### 示例3:人工录入案例
|
|
||||||
```sql
|
|
||||||
INSERT INTO case_library VALUES
|
|
||||||
(3, 'case-003', NULL, 'MANUAL', 'DATABASE', 'mysql-master-01', 'UPDATE orders SET status=? WHERE order_id=?', '1213',
|
|
||||||
'订单库存更新死锁通用处理',
|
|
||||||
'两个事务互相等待对方释放锁',
|
|
||||||
'调整事务加锁顺序:统一先锁订单,再锁库存;或使用乐观锁',
|
|
||||||
3, 'admin', NOW(), NOW());
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 典型查询
|
|
||||||
|
|
||||||
### 精确匹配查询
|
|
||||||
```sql
|
|
||||||
-- 按错误码查询
|
|
||||||
SELECT * FROM case_library
|
|
||||||
WHERE error_code = '40003'
|
|
||||||
ORDER BY reference_count DESC
|
|
||||||
LIMIT 5;
|
|
||||||
|
|
||||||
-- 按故障类别 + 错误码 + 故障目标查询
|
|
||||||
SELECT * FROM case_library
|
|
||||||
WHERE fault_category = 'INTERNAL_ERROR'
|
|
||||||
AND error_code = 'NullPointerException'
|
|
||||||
AND fault_target = 'OrderController.createOrder()'
|
|
||||||
ORDER BY reference_count DESC
|
|
||||||
LIMIT 5;
|
|
||||||
```
|
|
||||||
|
|
||||||
### 统计分析
|
|
||||||
```sql
|
|
||||||
-- 统计案例分布
|
|
||||||
SELECT
|
|
||||||
fault_category,
|
|
||||||
COUNT(*) as count,
|
|
||||||
AVG(reference_count) as avg_reference
|
|
||||||
FROM case_library
|
|
||||||
GROUP BY fault_category
|
|
||||||
ORDER BY count DESC;
|
|
||||||
|
|
||||||
-- Top 引用案例
|
|
||||||
SELECT title, reference_count, created_at
|
|
||||||
FROM case_library
|
|
||||||
ORDER BY reference_count DESC
|
|
||||||
LIMIT 10;
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 与 Milvus 的配合
|
|
||||||
|
|
||||||
### 混合检索策略
|
|
||||||
|
|
||||||
```
|
|
||||||
1. 精确匹配(MySQL)
|
|
||||||
- 按 error_code 查询
|
|
||||||
- 按 fault_category + fault_source 查询
|
|
||||||
- 优点:快速、准确
|
|
||||||
|
|
||||||
2. 语义检索(Milvus)
|
|
||||||
- 将案例内容向量化
|
|
||||||
- 按语义相似度查询
|
|
||||||
- 优点:能找到相似但不同错误码的案例
|
|
||||||
|
|
||||||
3. 混合策略(推荐)
|
|
||||||
Step 1: 先精确匹配(MySQL)
|
|
||||||
Step 2: 如果结果 < 3 个,补充语义检索(Milvus)
|
|
||||||
Step 3: 合并去重,按 reference_count 排序
|
|
||||||
Step 4: 返回 Top 5
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## 数据量预估
|
|
||||||
|
|
||||||
```
|
|
||||||
预估:500-1000 条
|
|
||||||
- 初期:每月新增 10-20 条
|
|
||||||
- 稳定期:每月新增 5-10 条
|
|
||||||
- 总量:1-2 年达到稳定
|
|
||||||
|
|
||||||
存储:
|
|
||||||
- 单条记录:约 2KB
|
|
||||||
- 1000 条:约 2MB
|
|
||||||
|
|
||||||
结论:数据量很小
|
|
||||||
```
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## MVP 版本的简化
|
|
||||||
|
|
||||||
```
|
|
||||||
Phase 1(当前):
|
|
||||||
✅ 基础字段和表结构
|
|
||||||
✅ 自动生成案例
|
|
||||||
✅ 人工录入案例
|
|
||||||
✅ 按 reference_count 简单排序
|
|
||||||
|
|
||||||
Phase 2(未来增强):
|
|
||||||
❌ useful_count + score(复杂评分)
|
|
||||||
❌ 版本管理
|
|
||||||
❌ 标签分类(tags)
|
|
||||||
❌ 案例合并功能
|
|
||||||
```
|
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
# 工具调用表:tool_invocation
|
||||||
|
|
||||||
|
**状态**:当前表
|
||||||
|
**来源**:`V005__create_session_storage.sql`、`V010__add_relevance_level_to_tool_invocation.sql`、`ToolInvocation`
|
||||||
|
|
||||||
|
## 定位
|
||||||
|
|
||||||
|
`tool_invocation` 记录 Agent 显式调用工具的事实,包括工具名、入参、输出摘要、检索层级、证据引用和失败信息。它是 Trace、Verifier、评测和人工排查的共同数据源。
|
||||||
|
|
||||||
|
## 字段
|
||||||
|
|
||||||
|
| 字段 | 类型 | 必填 | 说明 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `id` | BIGINT | 是 | 自增主键 |
|
||||||
|
| `session_id` | VARCHAR(64) | 是 | 关联 `diagnosis_session.session_id` |
|
||||||
|
| `step_id` | BIGINT | 否 | 可关联 `agent_step.id` |
|
||||||
|
| `tool_name` | VARCHAR(64) | 是 | 工具名称,例如 `lookup_knowledge`、日志查询、指标查询 |
|
||||||
|
| `input_params` | JSON | 是 | 工具入参 |
|
||||||
|
| `output_preview` | TEXT | 否 | 工具输出摘要或前缀 |
|
||||||
|
| `output_length` | INT | 否 | 工具输出字符数 |
|
||||||
|
| `retrieval_layer` | VARCHAR(8) | 否 | 检索层级,例如 `L0`、`L1`、`L0+L1` |
|
||||||
|
| `l0_match_count` | INT | 否 | L0 命中数量 |
|
||||||
|
| `l1_match_count` | INT | 否 | L1 命中数量 |
|
||||||
|
| `is_truncated` | BOOLEAN | 否 | 输出是否被截断 |
|
||||||
|
| `relevance_level` | VARCHAR(20) | 否 | 归一化质量等级:`PRECISE`、`HIGHLY_RELEVANT`、`REFERENCE`、`DEDUPED` |
|
||||||
|
| `dedup_reason` | VARCHAR(32) | 否 | 去重原因,例如 `doc_retrieved`、`domain_retrieved` |
|
||||||
|
| `retrieval_details` | JSON | 否 | 检索明细、证据引用、Gatekeeper 可用导航信息 |
|
||||||
|
| `duration_ms` | INT | 否 | 工具耗时 |
|
||||||
|
| `success` | BOOLEAN | 否 | 工具是否成功 |
|
||||||
|
| `error_message` | TEXT | 否 | 失败原因 |
|
||||||
|
| `created_at` | DATETIME | 是 | 创建时间 |
|
||||||
|
|
||||||
|
## 索引
|
||||||
|
|
||||||
|
| 索引 | 字段 | 用途 |
|
||||||
|
|---|---|---|
|
||||||
|
| `idx_session_id` | `session_id` | 按会话查询工具调用 |
|
||||||
|
| `idx_tool_name` | `tool_name` | 按工具类型排查 |
|
||||||
|
| `idx_retrieval_layer` | `retrieval_layer` | 观察 RAG L0/L1 行为 |
|
||||||
|
|
||||||
|
## 关系
|
||||||
|
|
||||||
|
- `tool_invocation.session_id` 逻辑关联 `diagnosis_session.session_id`。
|
||||||
|
- `tool_invocation.step_id` 可关联 `agent_step.id`,但当前不强制。
|
||||||
|
|
||||||
|
## 关键 JSON
|
||||||
|
|
||||||
|
`retrieval_details` 是扩展字段。当前重要结构包括:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"evidence_status": "supported",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"raw_path": "$.logs[0]",
|
||||||
|
"text": "工具返回中可核对的最小证据文本"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## 注意点
|
||||||
|
|
||||||
|
- Verifier 不应只信任 RAG 证据;所有工具只要能提供 `evidence_refs`,都应该进入可校验证据链。
|
||||||
|
- `output_preview` 只适合展示和排查,不应被当成完整原始输出。
|
||||||
|
- `$.no_evidence` 只代表“本次工具未命中证据”,不能推导为“故障不存在”。
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
# 文档元数据表:api_document
|
||||||
|
|
||||||
|
**状态**:当前表
|
||||||
|
**来源**:`V003__create_api_document.sql`、`V004__add_metadata_to_api_document.sql`、`ApiDocument`
|
||||||
|
|
||||||
|
## 定位
|
||||||
|
|
||||||
|
`api_document` 是知识库文档的 MySQL 元数据表。它不保存向量正文,正文切片和向量检索由 Milvus/Zilliz collection 承担;两侧通过 `doc_id` 和 chunk metadata 逻辑关联。
|
||||||
|
|
||||||
|
## 字段
|
||||||
|
|
||||||
|
| 字段 | 类型 | 必填 | 说明 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `id` | BIGINT | 是 | 自增主键 |
|
||||||
|
| `doc_id` | VARCHAR(64) | 是 | 文档唯一 ID,关联向量库 chunk metadata |
|
||||||
|
| `fault_category` | VARCHAR(32) | 否 | 文档类别,默认 `EXTERNAL_API`;实体侧使用 `FaultCategory` |
|
||||||
|
| `fault_source` | VARCHAR(128) | 否 | 文档归属,例如服务名、省份或系统来源 |
|
||||||
|
| `api_name` | VARCHAR(128) | 否 | 接口或文档主题名称 |
|
||||||
|
| `version` | VARCHAR(32) | 否 | 文档版本,默认 `v1.0` |
|
||||||
|
| `file_name` | VARCHAR(256) | 是 | 原始文件名 |
|
||||||
|
| `file_path` | VARCHAR(512) | 否 | 文件存储路径 |
|
||||||
|
| `file_hash` | VARCHAR(64) | 否 | 文件 MD5,用于去重 |
|
||||||
|
| `file_size` | BIGINT | 否 | 文件大小,单位字节 |
|
||||||
|
| `status` | VARCHAR(16) | 否 | 索引状态:`PENDING`、`PROCESSING`、`INDEXED`、`FAILED` |
|
||||||
|
| `chunk_count` | INT | 否 | 向量库切片数量 |
|
||||||
|
| `error_message` | TEXT | 否 | 索引失败原因 |
|
||||||
|
| `metadata` | TEXT | 否 | frontmatter 元数据 JSON 字符串 |
|
||||||
|
| `indexed_at` | DATETIME | 否 | 索引完成时间 |
|
||||||
|
| `created_at` | DATETIME | 是 | 创建时间 |
|
||||||
|
| `updated_at` | DATETIME | 是 | 更新时间 |
|
||||||
|
|
||||||
|
## 索引
|
||||||
|
|
||||||
|
| 索引 | 字段 | 用途 |
|
||||||
|
|---|---|---|
|
||||||
|
| `uk_file_hash` | `file_hash` | 文件去重 |
|
||||||
|
| `idx_doc_id` | `doc_id` | 按文档 ID 查询 |
|
||||||
|
| `idx_fault_source` | `fault_source` | 按来源筛选 |
|
||||||
|
| `idx_status` | `status` | 查看索引状态 |
|
||||||
|
| `idx_created_at` | `created_at` | 按上传时间排序 |
|
||||||
|
|
||||||
|
## 关系
|
||||||
|
|
||||||
|
- `api_document.doc_id` 与向量库 chunk metadata 中的 `docId` / `doc_id` 逻辑关联。
|
||||||
|
- `knowledge_domain.domain_id` 与文档 metadata 中的 `category` 形成领域聚合关系;当前没有数据库外键。
|
||||||
|
|
||||||
|
## 注意点
|
||||||
|
|
||||||
|
- 删除文档时需要同时处理 MySQL 元数据和向量库 chunk。
|
||||||
|
- `metadata` 是 JSON 字符串,不是 MySQL JSON 列。
|
||||||
|
- 表字段以 Flyway 为准;实体默认值和枚举可能与迁移脚本的 SQL 默认值存在历史差异,排查时优先看实际迁移和数据库结构。
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
# 案例库表:case_library
|
||||||
|
|
||||||
|
**状态**:当前表
|
||||||
|
**来源**:`V002__create_case_library.sql`、`CaseLibrary`
|
||||||
|
|
||||||
|
## 定位
|
||||||
|
|
||||||
|
`case_library` 保存高质量诊断案例,用于后续相似案例推荐和知识沉淀。当前自动沉淀路径来自 `useful` 用户反馈:系统把 `diagnosis_session` 中的 query 和 answer 映射为案例内容。
|
||||||
|
|
||||||
|
## 字段
|
||||||
|
|
||||||
|
| 字段 | 类型 | 必填 | 说明 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `id` | BIGINT | 是 | 自增主键 |
|
||||||
|
| `case_id` | VARCHAR(64) | 是 | 案例唯一 ID |
|
||||||
|
| `diagnosis_id` | VARCHAR(64) | 否 | 关联诊断会话;当前自动生成时存 `diagnosis_session.session_id` |
|
||||||
|
| `source_type` | VARCHAR(16) | 否 | 来源类型:`AUTO` 或 `MANUAL` |
|
||||||
|
| `fault_category` | VARCHAR(32) | 否 | 故障类别,实体侧使用 `FaultCategory` |
|
||||||
|
| `fault_source` | VARCHAR(128) | 否 | 故障源,例如服务、系统或省份 |
|
||||||
|
| `fault_target` | VARCHAR(256) | 否 | 故障目标,例如接口、方法、SQL 或组件 |
|
||||||
|
| `error_code` | VARCHAR(64) | 否 | 错误码或异常类型 |
|
||||||
|
| `title` | VARCHAR(256) | 是 | 案例标题 |
|
||||||
|
| `root_cause` | TEXT | 是 | 根因分析 |
|
||||||
|
| `solution` | TEXT | 是 | 解决方案 |
|
||||||
|
| `reference_count` | INT | 否 | 被推荐次数 |
|
||||||
|
| `created_by` | VARCHAR(64) | 否 | 创建人 |
|
||||||
|
| `created_at` | DATETIME | 是 | 创建时间 |
|
||||||
|
| `updated_at` | DATETIME | 是 | 更新时间 |
|
||||||
|
|
||||||
|
## 索引
|
||||||
|
|
||||||
|
| 索引 | 字段 | 用途 |
|
||||||
|
|---|---|---|
|
||||||
|
| `idx_fault_category` | `fault_category` | 按故障类别筛选 |
|
||||||
|
| `idx_error_code` | `error_code` | 按错误码精确匹配 |
|
||||||
|
| `idx_fault_source` | `fault_source` | 按故障源筛选 |
|
||||||
|
| `idx_fault_target` | `fault_target(100)` | 按故障目标筛选 |
|
||||||
|
| `idx_diagnosis_id` | `diagnosis_id` | 追溯来源会话 |
|
||||||
|
| `idx_reference_count` | `reference_count` | 推荐排序 |
|
||||||
|
| `idx_created_at` | `created_at` | 时间排序 |
|
||||||
|
|
||||||
|
## 关系
|
||||||
|
|
||||||
|
- `case_library.diagnosis_id` 当前逻辑关联 `diagnosis_session.session_id`,不是旧的 `diagnosis_record`。
|
||||||
|
- 人工录入案例可以不填写 `diagnosis_id`。
|
||||||
|
|
||||||
|
## 注意点
|
||||||
|
|
||||||
|
- 旧文档里提到的 `diagnosis_record` 已被 `V007` 删除,不再是当前主模型。
|
||||||
|
- 当前自动沉淀仍比较粗:`root_cause` 和 `solution` 都可能来自完整 answer。后续可从结构化结论中拆分根因、证据和修复建议。
|
||||||
@@ -0,0 +1,36 @@
|
|||||||
|
# 知识域表:knowledge_domain
|
||||||
|
|
||||||
|
**状态**:当前表
|
||||||
|
**来源**:`V009__add_knowledge_domain.sql`、`KnowledgeDomain`
|
||||||
|
|
||||||
|
## 定位
|
||||||
|
|
||||||
|
`knowledge_domain` 保存知识库领域级元数据,用来帮助 Planner/Executor 判断什么时候检索某一类知识,并为 RAG 的 domain hint、去重和可观测性提供基础信息。
|
||||||
|
|
||||||
|
## 字段
|
||||||
|
|
||||||
|
| 字段 | 类型 | 必填 | 说明 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `id` | BIGINT | 是 | 自增主键 |
|
||||||
|
| `domain_id` | VARCHAR(64) | 是 | 领域 ID,通常对应文档 category,例如 `payment`、`infrastructure` |
|
||||||
|
| `description` | VARCHAR(256) | 否 | 领域描述 |
|
||||||
|
| `when_to_retrieve` | TEXT | 否 | 何时检索该领域的提示说明 |
|
||||||
|
| `document_count` | INT | 是 | 当前领域文档数量 |
|
||||||
|
| `created_at` | DATETIME | 是 | 创建时间 |
|
||||||
|
| `updated_at` | DATETIME | 是 | 更新时间 |
|
||||||
|
|
||||||
|
## 索引
|
||||||
|
|
||||||
|
| 索引 | 字段 | 用途 |
|
||||||
|
|---|---|---|
|
||||||
|
| unique | `domain_id` | 保证领域 ID 唯一 |
|
||||||
|
|
||||||
|
## 关系
|
||||||
|
|
||||||
|
- `knowledge_domain.domain_id` 与 `api_document.metadata` 或向量库 chunk metadata 中的 `category` 逻辑关联。
|
||||||
|
- 当前没有数据库外键,领域文档数量由服务逻辑维护。
|
||||||
|
|
||||||
|
## 注意点
|
||||||
|
|
||||||
|
- `when_to_retrieve` 是检索策略提示,不是事实证据。
|
||||||
|
- Executor / Verifier 不能把领域描述当作诊断结论依据;事实仍应来自工具返回的证据块或证据引用。
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
# 诊断会话表:diagnosis_session
|
||||||
|
|
||||||
|
**状态**:当前主表
|
||||||
|
**来源**:`V005__create_session_storage.sql`、`V008__add_answer_to_diagnosis_session.sql`、`DiagnosisSession`
|
||||||
|
|
||||||
|
## 定位
|
||||||
|
|
||||||
|
`diagnosis_session` 是一次 Chat 或 AIOps 诊断的会话级主记录,负责保存用户问题、执行状态、最终答案、总体统计和自评估结果。
|
||||||
|
|
||||||
|
## 字段
|
||||||
|
|
||||||
|
| 字段 | 类型 | 必填 | 说明 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `id` | BIGINT | 是 | 自增主键 |
|
||||||
|
| `session_id` | VARCHAR(64) | 是 | 会话唯一 ID,Trace API 和反馈接口使用它 |
|
||||||
|
| `query` | TEXT | 是 | 用户原始问题或 AIOps 输入摘要 |
|
||||||
|
| `status` | VARCHAR(16) | 否 | `PENDING`、`RUNNING`、`SUCCESS`、`FAILED` |
|
||||||
|
| `agent_flow` | VARCHAR(32) | 否 | `CHAT` 或 `AI_OPS` |
|
||||||
|
| `total_duration_ms` | INT | 否 | 总耗时,单位毫秒 |
|
||||||
|
| `total_token_count` | INT | 否 | 总 Token 消耗 |
|
||||||
|
| `step_count` | INT | 否 | Agent 步骤数 |
|
||||||
|
| `tool_call_count` | INT | 否 | 工具调用次数 |
|
||||||
|
| `answer` | LONGTEXT | 否 | 返回给用户的最终答案 |
|
||||||
|
| `self_evaluation` | JSON | 否 | rule、verifier、aiops 等自评估结果容器 |
|
||||||
|
| `feedback` | VARCHAR(16) | 否 | 用户反馈:`useful`、`not_useful` 或空 |
|
||||||
|
| `created_at` | DATETIME | 是 | 创建时间 |
|
||||||
|
| `updated_at` | DATETIME | 是 | 更新时间 |
|
||||||
|
|
||||||
|
## 索引
|
||||||
|
|
||||||
|
| 索引 | 字段 | 用途 |
|
||||||
|
|---|---|---|
|
||||||
|
| `session_id` unique | `session_id` | 会话唯一约束 |
|
||||||
|
| `idx_created_at` | `created_at` | 按时间查询 |
|
||||||
|
| `idx_status` | `status` | 按状态筛选 |
|
||||||
|
| `idx_agent_flow` | `agent_flow` | 区分 Chat / AIOps |
|
||||||
|
|
||||||
|
## 关系
|
||||||
|
|
||||||
|
- `agent_step.session_id` 逻辑关联 `diagnosis_session.session_id`。
|
||||||
|
- `tool_invocation.session_id` 逻辑关联 `diagnosis_session.session_id`。
|
||||||
|
- `case_library.diagnosis_id` 在自动生成案例时保存 `diagnosis_session.session_id`。
|
||||||
|
|
||||||
|
## 注意点
|
||||||
|
|
||||||
|
- 当前没有数据库外键,Trace 聚合依赖 `session_id`。
|
||||||
|
- `self_evaluation` 是扩展容器,里面可能包含 `rule_evaluation`、`verifier_evaluation`、`aiops_rule_evaluation`。
|
||||||
@@ -30,7 +30,7 @@
|
|||||||
|
|
||||||
| Source | Alignment |
|
| Source | Alignment |
|
||||||
|---|---|
|
|---|---|
|
||||||
| Issue objective | Stage four in `mvp/issues/executor-structured-output-v2.md` requires Composer output final answer from Verifier-allowed material. Covered by proposal, design, specs, and tasks. |
|
| Issue objective | Stage four in `mvp/issues/design-notes/executor-structured-output-v2.md` requires Composer output final answer from Verifier-allowed material. Covered by proposal, design, specs, and tasks. |
|
||||||
| Proposal -> design | Proposal says Composer owns final expression; design defines input filtering, output parsing, fallback, and audit. |
|
| Proposal -> design | Proposal says Composer owns final expression; design defines input filtering, output parsing, fallback, and audit. |
|
||||||
| Design -> specs | Design decisions are reflected in `chat-composer-agent` requirements and modified `chat-verifier-agent` routing requirements. |
|
| Design -> specs | Design decisions are reflected in `chat-composer-agent` requirements and modified `chat-verifier-agent` routing requirements. |
|
||||||
| Specs -> tasks | Each required behavior has implementation and test tasks, including malformed fallback and no raw output leakage. |
|
| Specs -> tasks | Each required behavior has implementation and test tasks, including malformed fallback and no raw output leakage. |
|
||||||
|
|||||||
+1
-1
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
## Discover Context
|
## Discover Context
|
||||||
|
|
||||||
- Source issue: `mvp/issues/ISS-007-verifier-evidence-summary-fidelity.md`.
|
- Source issue: `mvp/issues/archived/ISS-007-verifier-evidence-summary-fidelity.md`.
|
||||||
- Related existing specs: `chat-verifier-agent`, `evidence-trace-hardening`.
|
- Related existing specs: `chat-verifier-agent`, `evidence-trace-hardening`.
|
||||||
- Related devflow records: `executor-gatekeeper-hook`, `executor-verifier-claim-checks`, `executor-composer-final-answer`, `evidence-trace-hardening`.
|
- Related devflow records: `executor-gatekeeper-hook`, `executor-verifier-claim-checks`, `executor-composer-final-answer`, `evidence-trace-hardening`.
|
||||||
- Current repo instruction requested semantic code search and LSP confirmation before code changes; those tools are not exposed in this environment, so implementation will use `rg`, direct code reading, and focused tests as fallback evidence.
|
- Current repo instruction requested semantic code search and LSP confirmation before code changes; those tools are not exposed in this environment, so implementation will use `rg`, direct code reading, and focused tests as fallback evidence.
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
ready
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
committed
|
||||||
@@ -0,0 +1,117 @@
|
|||||||
|
# Design: Interview Demo Quality Audit
|
||||||
|
|
||||||
|
## Overview
|
||||||
|
|
||||||
|
The change adds auditability and demo readiness without changing Agent routing.
|
||||||
|
|
||||||
|
```text
|
||||||
|
ChatService prompt resources
|
||||||
|
-> PromptAuditService
|
||||||
|
-> verifier_evaluation.prompt_audit
|
||||||
|
-> Trace API
|
||||||
|
-> DiagnosisTraceEvaluator
|
||||||
|
-> baseline report
|
||||||
|
|
||||||
|
mvp/demo/scripts/run-interview-demo-check.ps1
|
||||||
|
-> health/readiness check
|
||||||
|
-> payment timeout chat
|
||||||
|
-> trace fetch
|
||||||
|
-> feedback
|
||||||
|
-> demo output bundle
|
||||||
|
```
|
||||||
|
|
||||||
|
## Prompt Audit
|
||||||
|
|
||||||
|
Add a compact `prompt_audit` object under:
|
||||||
|
|
||||||
|
```text
|
||||||
|
diagnosis_session.self_evaluation.verifier_evaluation.prompt_audit
|
||||||
|
```
|
||||||
|
|
||||||
|
Shape:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"version": "chat-prompts-v1",
|
||||||
|
"prompts": [
|
||||||
|
{
|
||||||
|
"name": "chat_planner",
|
||||||
|
"version": "chat-planner-v1",
|
||||||
|
"resource": "prompts/chat-planner-prompt.md"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Design choices:
|
||||||
|
|
||||||
|
- Use explicit local metadata, not full prompt hashes, because the interview goal is explainable version audit rather than cryptographic integrity.
|
||||||
|
- Keep this metadata in code or a small resource catalog near prompt loading.
|
||||||
|
- Persist prompt audit with every Chat verifier evaluation, including fallback/degraded paths.
|
||||||
|
- Do not include full prompt text in trace.
|
||||||
|
|
||||||
|
## Eval Expansion
|
||||||
|
|
||||||
|
Extend `DiagnosisEvalCase` with optional fields:
|
||||||
|
|
||||||
|
- `requirePromptAudit`
|
||||||
|
- `expectedPromptAuditVersion`
|
||||||
|
- `expectedPromptVersions`
|
||||||
|
- `requireGatekeeperRules`
|
||||||
|
|
||||||
|
Evaluator behavior:
|
||||||
|
|
||||||
|
- If `requirePromptAudit=true`, `verifier_evaluation.prompt_audit.version` must exist.
|
||||||
|
- If `expectedPromptAuditVersion` is set, it must match.
|
||||||
|
- If `expectedPromptVersions` is set, each listed prompt name/version pair must exist.
|
||||||
|
- If `requireGatekeeperRules=true`, `gatekeeper_result.rules` must be a non-empty list and each item must include `id`, `enabled`, and `default_severity`.
|
||||||
|
|
||||||
|
Add at least two fixture-backed cases:
|
||||||
|
|
||||||
|
- A positive audit closure case that requires prompt audit + Gatekeeper rules.
|
||||||
|
- A metadata-gap negative case represented as `LOW_CONFID`/safe final answer, used to prove the evaluator catches missing audit metadata when configured.
|
||||||
|
|
||||||
|
The baseline must remain fully passing after fixtures are updated.
|
||||||
|
|
||||||
|
## Demo Stabilization
|
||||||
|
|
||||||
|
Add a PowerShell script:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mvp/demo/scripts/run-interview-demo-check.ps1
|
||||||
|
```
|
||||||
|
|
||||||
|
Responsibilities:
|
||||||
|
|
||||||
|
- Accept base URL and session id parameters.
|
||||||
|
- Check that the service is reachable.
|
||||||
|
- Run the existing payment-timeout chat request.
|
||||||
|
- Fetch trace for the same session id.
|
||||||
|
- Submit useful feedback.
|
||||||
|
- Write outputs under `mvp/demo/output/`.
|
||||||
|
- Emit a concise summary with session id, verdict, Gatekeeper rule version, prompt audit version, and output paths.
|
||||||
|
|
||||||
|
The script should fail fast with actionable messages when the service is unavailable.
|
||||||
|
|
||||||
|
## Documentation
|
||||||
|
|
||||||
|
Add/update:
|
||||||
|
|
||||||
|
- `mvp/demo/README.md`: mention the preflight script.
|
||||||
|
- `mvp/demo/ten-minute-interview-demo.md`: use the preflight script as the recommended path.
|
||||||
|
- `mvp/demo/interview-q-and-a.md`: concise interview answers for Agent engineering tradeoffs.
|
||||||
|
- `mvp/architecture/harness-quality-gates.md`: record prompt audit as part of the quality gate.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
Required:
|
||||||
|
|
||||||
|
- Targeted unit/eval tests for prompt audit persistence and evaluator checks.
|
||||||
|
- Regenerated baseline JSON/Markdown reports.
|
||||||
|
- OpenSpec validation.
|
||||||
|
|
||||||
|
E2E:
|
||||||
|
|
||||||
|
- If local dependencies are available, run Spring Boot with `mvp-demo` profile and execute the new preflight script.
|
||||||
|
- If unavailable, record the reason and rely on deterministic unit/eval evidence.
|
||||||
|
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
# Change: Interview Demo Quality Audit
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
SuperBizAgent MVP is now strong enough to demonstrate traceable Agent engineering, but the interview path still has three gaps:
|
||||||
|
|
||||||
|
- The live demo has a run script, but no preflight command that checks service readiness and produces a concise interview evidence bundle.
|
||||||
|
- The diagnosis eval baseline covers the V2 evidence pipeline, but it does not yet assert prompt-version audit data and has limited coverage for audit metadata gaps.
|
||||||
|
- Gatekeeper already exposes `rule_set_version`, but prompt versions are not persisted with the verifier evaluation, making prompt changes harder to explain, compare, and roll back in an interview.
|
||||||
|
|
||||||
|
This change stabilizes the MVP as an interview artifact rather than adding a new diagnosis architecture.
|
||||||
|
|
||||||
|
## Proposed Solution
|
||||||
|
|
||||||
|
Implement a small internal quality/audit increment:
|
||||||
|
|
||||||
|
1. Add prompt version audit metadata to Chat verifier evaluation.
|
||||||
|
2. Extend deterministic diagnosis eval cases/fixtures to assert prompt audit metadata and audit-metadata failures.
|
||||||
|
3. Add an interview demo preflight script and documentation that can be run before or during a demo to verify service readiness, execute the payment timeout path, fetch trace, and record key audit fields.
|
||||||
|
4. Add/update MVP documentation for interview Q&A and the new audit/preflight workflow.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
In scope:
|
||||||
|
|
||||||
|
- Internal `diagnosis_session.self_evaluation.verifier_evaluation` audit JSON.
|
||||||
|
- Diagnosis eval case schema, evaluator checks, fixtures, and baseline reports.
|
||||||
|
- MVP demo scripts/docs.
|
||||||
|
- Architecture/demo documentation for prompt and Gatekeeper version audit.
|
||||||
|
|
||||||
|
Out of scope:
|
||||||
|
|
||||||
|
- Public HTTP API changes.
|
||||||
|
- Database schema changes.
|
||||||
|
- New Agent roles, MCP tool server migration, process isolation, or AIOps LLM Verifier.
|
||||||
|
- Replacing existing `Planner -> Executor -> Gatekeeper -> Verifier -> Composer` orchestration.
|
||||||
|
- Guaranteeing live LLM `PASS` for every demo run. Live demo compatibility and deterministic fixture regression are separate acceptance paths.
|
||||||
|
|
||||||
|
## Context Constraints From devflow
|
||||||
|
|
||||||
|
- Evidence Tools produce incident facts and must be recorded in `tool_invocation`.
|
||||||
|
- Chat quality gates are layered: Gatekeeper verifies evidence references, Verifier judges derivability, Composer controls expression.
|
||||||
|
- `diagnosis eval` is deterministic and fixture-backed; no LLM-as-judge.
|
||||||
|
- Demo assets should be runnable, but interview safety should not depend solely on live LLM behavior.
|
||||||
|
- Gatekeeper rule metadata is metadata-only; dynamic rule execution is out of scope.
|
||||||
|
|
||||||
|
## Interface Impact
|
||||||
|
|
||||||
|
Level: L2 internal contract change.
|
||||||
|
|
||||||
|
Reason: `verifier_evaluation` gains a compact `prompt_audit` object. Existing public API shape remains the same, and the value is exposed only through already-existing trace/self-evaluation JSON.
|
||||||
|
|
||||||
|
## Risks
|
||||||
|
|
||||||
|
- Baseline report churn is expected when adding cases; JSON and Markdown reports must be regenerated together.
|
||||||
|
- Prompt audit must be deterministic and stable enough for eval fixtures; avoid hashing full prompt text with environment-specific content.
|
||||||
|
- Demo preflight must not hardcode secrets and must tolerate local service unavailability with clear failure messages.
|
||||||
|
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
## MODIFIED Requirements
|
||||||
|
|
||||||
|
### Requirement: Verifier SHALL be observable
|
||||||
|
The Verifier's verdict SHALL be persisted for observability.
|
||||||
|
|
||||||
|
#### Scenario: prompt audit written to verifier evaluation
|
||||||
|
- **WHEN** the Chat verifier evaluation is persisted
|
||||||
|
- **THEN** the system SHALL include a `prompt_audit` object under `diagnosis_session.self_evaluation.verifier_evaluation`
|
||||||
|
- **AND** `prompt_audit.version` SHALL identify the Chat prompt audit catalog version
|
||||||
|
- **AND** `prompt_audit.prompts` SHALL include the planner, executor, verifier, and composer prompt names and versions
|
||||||
|
- **AND** full prompt text SHALL NOT be persisted in `prompt_audit`
|
||||||
|
|
||||||
|
#### Scenario: prompt audit available on fallback paths
|
||||||
|
- **WHEN** Chat verifier parsing fails, Composer parsing fails, or Chat produces a degraded answer
|
||||||
|
- **THEN** the persisted verifier evaluation SHALL still include `prompt_audit`
|
||||||
|
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
## MODIFIED Requirements
|
||||||
|
|
||||||
|
### Requirement: Diagnosis eval SHALL validate trace fixtures deterministically
|
||||||
|
The diagnosis eval harness SHALL evaluate saved trace fixtures without invoking an LLM judge.
|
||||||
|
|
||||||
|
#### Scenario: prompt audit assertions are enforced
|
||||||
|
- **WHEN** an eval case sets `requirePromptAudit=true`
|
||||||
|
- **THEN** the evaluator SHALL require `verifier_evaluation.prompt_audit.version`
|
||||||
|
- **AND** when `expectedPromptAuditVersion` is configured, it SHALL match exactly
|
||||||
|
- **AND** when `expectedPromptVersions` is configured, each configured prompt name SHALL appear with the expected version
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper rule metadata assertions are enforced
|
||||||
|
- **WHEN** an eval case sets `requireGatekeeperRules=true`
|
||||||
|
- **THEN** the evaluator SHALL require `verifier_evaluation.gatekeeper_result.rules` to be non-empty
|
||||||
|
- **AND** each rule item SHALL include `id`, `enabled`, and `default_severity`
|
||||||
|
|
||||||
|
#### Scenario: expanded baseline remains passing
|
||||||
|
- **WHEN** the committed fixture set is evaluated
|
||||||
|
- **THEN** every case SHALL pass
|
||||||
|
- **AND** baseline JSON and Markdown reports SHALL reflect the expanded case count and verdict distribution
|
||||||
|
|
||||||
+22
@@ -0,0 +1,22 @@
|
|||||||
|
## MODIFIED Requirements
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL be reproducible for interviews
|
||||||
|
The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, verifier evaluation, and feedback.
|
||||||
|
|
||||||
|
#### Scenario: interview demo check script records an evidence bundle
|
||||||
|
- **WHEN** the user runs the interview demo check script against a running `mvp-demo` service
|
||||||
|
- **THEN** the script SHALL submit a fixed Chat diagnosis request
|
||||||
|
- **AND** it SHALL fetch the trace for the same session id
|
||||||
|
- **AND** it SHALL submit useful feedback for that session
|
||||||
|
- **AND** it SHALL write chat, trace, feedback, and summary outputs under `mvp/demo/output/`
|
||||||
|
|
||||||
|
#### Scenario: interview demo check fails with actionable readiness output
|
||||||
|
- **WHEN** the target service is not reachable
|
||||||
|
- **THEN** the script SHALL fail before issuing diagnosis requests
|
||||||
|
- **AND** the failure message SHALL name the base URL and the expected startup profile
|
||||||
|
|
||||||
|
#### Scenario: interview documentation explains audit fields
|
||||||
|
- **WHEN** an interviewer asks how prompt or Gatekeeper changes are audited
|
||||||
|
- **THEN** the demo documentation SHALL point to `prompt_audit.version` and `gatekeeper_result.rule_set_version`
|
||||||
|
- **AND** it SHALL explain that deterministic eval fixtures are the regression source of truth
|
||||||
|
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
## 1. OpenSpec And devflow
|
||||||
|
|
||||||
|
- [x] 1.1 Create OpenSpec proposal/design/spec/tasks for `interview-demo-quality-audit`.
|
||||||
|
- [x] 1.2 Record context, question pool, interface impact, audit, and verification plan in devflow decisions.
|
||||||
|
- [x] 1.3 Pass OpenSpec validation and create `.committed`.
|
||||||
|
|
||||||
|
## 2. Prompt/Gatekeeper Version Audit
|
||||||
|
|
||||||
|
- [x] 2.1 Add compact Chat prompt audit metadata for planner, executor, verifier, and composer prompts.
|
||||||
|
- [x] 2.2 Persist `prompt_audit` under `verifier_evaluation` for Chat verifier/composer outcomes.
|
||||||
|
- [x] 2.3 Add focused tests proving prompt audit appears in persisted verifier evaluation.
|
||||||
|
- [x] 2.4 Extend eval checks for prompt audit and Gatekeeper rule metadata.
|
||||||
|
|
||||||
|
## 3. Eval Expansion
|
||||||
|
|
||||||
|
- [x] 3.1 Extend diagnosis eval case/result schema for prompt audit fields.
|
||||||
|
- [x] 3.2 Add fixture-backed cases for audit closure coverage.
|
||||||
|
- [x] 3.3 Regenerate baseline JSON and Markdown reports.
|
||||||
|
- [x] 3.4 Update eval docs/schema.
|
||||||
|
|
||||||
|
## 4. Interview Demo Stabilization
|
||||||
|
|
||||||
|
- [x] 4.1 Add `run-interview-demo-check.ps1` with service preflight, chat, trace, feedback, and summary output.
|
||||||
|
- [x] 4.2 Update demo README and 10-minute script to use the preflight path.
|
||||||
|
- [x] 4.3 Add interview Q&A documentation focused on Agent engineering tradeoffs.
|
||||||
|
|
||||||
|
## 5. Verification And Archive
|
||||||
|
|
||||||
|
- [x] 5.1 Run targeted tests for ChatService/prompt audit and diagnosis eval.
|
||||||
|
- [x] 5.2 Run relevant broader regression tests.
|
||||||
|
- [x] 5.3 Run E2E demo check with `mvp-demo` profile if dependencies are available; otherwise record the blocker.
|
||||||
|
- [x] 5.4 Archive the OpenSpec change, update devflow artifacts, and commit implementation + archive.
|
||||||
@@ -145,6 +145,17 @@ The Verifier's verdict and downstream final-answer composition SHALL be persiste
|
|||||||
- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `gatekeeper_result`
|
- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `gatekeeper_result`
|
||||||
- **AND** existing verifier fields such as `verdict`, `facts_checked`, `executor_output_parse_status`, and `tool_trace_summary` SHALL be preserved
|
- **AND** existing verifier fields such as `verdict`, `facts_checked`, `executor_output_parse_status`, and `tool_trace_summary` SHALL be preserved
|
||||||
|
|
||||||
|
#### Scenario: prompt audit written to verifier evaluation
|
||||||
|
- **WHEN** the Chat verifier evaluation is persisted
|
||||||
|
- **THEN** the system SHALL include a `prompt_audit` object under `diagnosis_session.self_evaluation.verifier_evaluation`
|
||||||
|
- **AND** `prompt_audit.version` SHALL identify the Chat prompt audit catalog version
|
||||||
|
- **AND** `prompt_audit.prompts` SHALL include the planner, executor, verifier, and composer prompt names and versions
|
||||||
|
- **AND** full prompt text SHALL NOT be persisted in `prompt_audit`
|
||||||
|
|
||||||
|
#### Scenario: prompt audit available on fallback paths
|
||||||
|
- **WHEN** Chat verifier parsing fails, Composer parsing fails, or Chat produces a degraded answer
|
||||||
|
- **THEN** the persisted verifier evaluation SHALL still include `prompt_audit`
|
||||||
|
|
||||||
### Requirement: self_evaluation SHALL be a container object
|
### Requirement: self_evaluation SHALL be a container object
|
||||||
The `diagnosis_session.self_evaluation` field SHALL store multiple evaluation channels in one JSON object.
|
The `diagnosis_session.self_evaluation` field SHALL store multiple evaluation channels in one JSON object.
|
||||||
|
|
||||||
|
|||||||
@@ -195,3 +195,22 @@ The evaluation harness SHALL be able to assert the Gatekeeper rule set version r
|
|||||||
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
|
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
|
||||||
- **AND** the fixture has a different or missing rule set version
|
- **AND** the fixture has a different or missing rule set version
|
||||||
- **THEN** the case SHALL fail with a clear failed check
|
- **THEN** the case SHALL fail with a clear failed check
|
||||||
|
|
||||||
|
### Requirement: Diagnosis eval SHALL validate trace fixtures deterministically
|
||||||
|
The diagnosis eval harness SHALL evaluate saved trace fixtures without invoking an LLM judge.
|
||||||
|
|
||||||
|
#### Scenario: prompt audit assertions are enforced
|
||||||
|
- **WHEN** an eval case sets `requirePromptAudit=true`
|
||||||
|
- **THEN** the evaluator SHALL require `verifier_evaluation.prompt_audit.version`
|
||||||
|
- **AND** when `expectedPromptAuditVersion` is configured, it SHALL match exactly
|
||||||
|
- **AND** when `expectedPromptVersions` is configured, each configured prompt name SHALL appear with the expected version
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper rule metadata assertions are enforced
|
||||||
|
- **WHEN** an eval case sets `requireGatekeeperRules=true`
|
||||||
|
- **THEN** the evaluator SHALL require `verifier_evaluation.gatekeeper_result.rules` to be non-empty
|
||||||
|
- **AND** each rule item SHALL include `id`, `enabled`, and `default_severity`
|
||||||
|
|
||||||
|
#### Scenario: expanded baseline remains passing
|
||||||
|
- **WHEN** the committed fixture set is evaluated
|
||||||
|
- **THEN** every case SHALL pass
|
||||||
|
- **AND** baseline JSON and Markdown reports SHALL reflect the expanded case count and verdict distribution
|
||||||
|
|||||||
@@ -56,6 +56,26 @@ The MVP demo SHALL provide scripts and request payloads for running the payment-
|
|||||||
- **WHEN** the demo script finishes successfully
|
- **WHEN** the demo script finishes successfully
|
||||||
- **THEN** it SHALL write chat, trace, and feedback responses under a demo output directory
|
- **THEN** it SHALL write chat, trace, and feedback responses under a demo output directory
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL be reproducible for interviews
|
||||||
|
The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, verifier evaluation, and feedback.
|
||||||
|
|
||||||
|
#### Scenario: interview demo check script records an evidence bundle
|
||||||
|
- **WHEN** the user runs the interview demo check script against a running `mvp-demo` service
|
||||||
|
- **THEN** the script SHALL submit a fixed Chat diagnosis request
|
||||||
|
- **AND** it SHALL fetch the trace for the same session id
|
||||||
|
- **AND** it SHALL submit useful feedback for that session
|
||||||
|
- **AND** it SHALL write chat, trace, feedback, and summary outputs under `mvp/demo/output/`
|
||||||
|
|
||||||
|
#### Scenario: interview demo check fails with actionable readiness output
|
||||||
|
- **WHEN** the target service is not reachable
|
||||||
|
- **THEN** the script SHALL fail before issuing diagnosis requests
|
||||||
|
- **AND** the failure message SHALL name the base URL and the expected startup profile
|
||||||
|
|
||||||
|
#### Scenario: interview documentation explains audit fields
|
||||||
|
- **WHEN** an interviewer asks how prompt or Gatekeeper changes are audited
|
||||||
|
- **THEN** the demo documentation SHALL point to `prompt_audit.version` and `gatekeeper_result.rule_set_version`
|
||||||
|
- **AND** it SHALL explain that deterministic eval fixtures are the regression source of truth
|
||||||
|
|
||||||
### Requirement: MVP demo SHALL provide a trace inspection checklist
|
### Requirement: MVP demo SHALL provide a trace inspection checklist
|
||||||
The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and session-level auditability.
|
The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and session-level auditability.
|
||||||
|
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import lombok.Data;
|
|||||||
import lombok.NoArgsConstructor;
|
import lombok.NoArgsConstructor;
|
||||||
|
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
@Data
|
@Data
|
||||||
@Builder
|
@Builder
|
||||||
@@ -29,4 +30,8 @@ public class DiagnosisEvalCase {
|
|||||||
private String expectedGatekeeperRuleSetVersion;
|
private String expectedGatekeeperRuleSetVersion;
|
||||||
private List<String> expectedComposerStatuses;
|
private List<String> expectedComposerStatuses;
|
||||||
private List<String> forbiddenConfirmedClaimKeywords;
|
private List<String> forbiddenConfirmedClaimKeywords;
|
||||||
|
private Boolean requirePromptAudit;
|
||||||
|
private String expectedPromptAuditVersion;
|
||||||
|
private Map<String, String> expectedPromptVersions;
|
||||||
|
private Boolean requireGatekeeperRules;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -46,8 +46,8 @@ public class DiagnosisEvalReportWriter {
|
|||||||
}
|
}
|
||||||
|
|
||||||
builder.append("## Cases\n\n");
|
builder.append("## Cases\n\n");
|
||||||
builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
|
builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
|
||||||
builder.append("| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |\n");
|
builder.append("| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- |\n");
|
||||||
for (DiagnosisEvalResult result : report.getResults()) {
|
for (DiagnosisEvalResult result : report.getResults()) {
|
||||||
builder.append("| ")
|
builder.append("| ")
|
||||||
.append(result.getCaseId())
|
.append(result.getCaseId())
|
||||||
@@ -60,10 +60,14 @@ public class DiagnosisEvalReportWriter {
|
|||||||
.append(" | ")
|
.append(" | ")
|
||||||
.append(valueOrDash(result.getGatekeeperRuleSetVersion()))
|
.append(valueOrDash(result.getGatekeeperRuleSetVersion()))
|
||||||
.append(" | ")
|
.append(" | ")
|
||||||
|
.append(valueOrDash(result.getPromptAuditVersion()))
|
||||||
|
.append(" | ")
|
||||||
.append(valueOrDash(result.getComposerStatus()))
|
.append(valueOrDash(result.getComposerStatus()))
|
||||||
.append(" | ")
|
.append(" | ")
|
||||||
.append(result.getClaimCheckCount() == null ? "-" : result.getClaimCheckCount())
|
.append(result.getClaimCheckCount() == null ? "-" : result.getClaimCheckCount())
|
||||||
.append(" | ")
|
.append(" | ")
|
||||||
|
.append(result.getGatekeeperRuleCount() == null ? "-" : result.getGatekeeperRuleCount())
|
||||||
|
.append(" | ")
|
||||||
.append(result.getMatchedKeywordCount()).append("/").append(result.getRequiredKeywordCount())
|
.append(result.getMatchedKeywordCount()).append("/").append(result.getRequiredKeywordCount())
|
||||||
.append(" | ")
|
.append(" | ")
|
||||||
.append(result.getToolCallCount() == null ? "-" : result.getToolCallCount())
|
.append(result.getToolCallCount() == null ? "-" : result.getToolCallCount())
|
||||||
|
|||||||
@@ -24,8 +24,10 @@ public class DiagnosisEvalResult {
|
|||||||
private Map<String, Boolean> evidenceCoverage;
|
private Map<String, Boolean> evidenceCoverage;
|
||||||
private String gatekeeperStatus;
|
private String gatekeeperStatus;
|
||||||
private String gatekeeperRuleSetVersion;
|
private String gatekeeperRuleSetVersion;
|
||||||
|
private String promptAuditVersion;
|
||||||
private String composerStatus;
|
private String composerStatus;
|
||||||
private Integer claimCheckCount;
|
private Integer claimCheckCount;
|
||||||
|
private Integer gatekeeperRuleCount;
|
||||||
private Integer toolCallCount;
|
private Integer toolCallCount;
|
||||||
private Integer durationMs;
|
private Integer durationMs;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -67,8 +67,10 @@ public class DiagnosisTraceEvaluator {
|
|||||||
.evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools()))
|
.evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools()))
|
||||||
.gatekeeperStatus(null)
|
.gatekeeperStatus(null)
|
||||||
.gatekeeperRuleSetVersion(null)
|
.gatekeeperRuleSetVersion(null)
|
||||||
|
.promptAuditVersion(null)
|
||||||
.composerStatus(null)
|
.composerStatus(null)
|
||||||
.claimCheckCount(null)
|
.claimCheckCount(null)
|
||||||
|
.gatekeeperRuleCount(null)
|
||||||
.toolCallCount(null)
|
.toolCallCount(null)
|
||||||
.durationMs(null)
|
.durationMs(null)
|
||||||
.build());
|
.build());
|
||||||
@@ -123,10 +125,13 @@ public class DiagnosisTraceEvaluator {
|
|||||||
String gatekeeperStatus = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "status");
|
String gatekeeperStatus = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "status");
|
||||||
String gatekeeperRuleSetVersion = extractNestedString(trace, "verifier_evaluation",
|
String gatekeeperRuleSetVersion = extractNestedString(trace, "verifier_evaluation",
|
||||||
"gatekeeper_result", "rule_set_version");
|
"gatekeeper_result", "rule_set_version");
|
||||||
|
String promptAuditVersion = extractNestedString(trace, "verifier_evaluation",
|
||||||
|
"prompt_audit", "version");
|
||||||
String composerStatus = extractNestedString(trace, "verifier_evaluation", "composer_output", "status");
|
String composerStatus = extractNestedString(trace, "verifier_evaluation", "composer_output", "status");
|
||||||
Integer claimCheckCount = countList(trace, "verifier_evaluation", "claim_checks");
|
Integer claimCheckCount = countList(trace, "verifier_evaluation", "claim_checks");
|
||||||
|
Integer gatekeeperRuleCount = countNestedList(trace, "verifier_evaluation", "gatekeeper_result", "rules");
|
||||||
failedChecks.addAll(validateV2AuditClosure(evalCase, trace, normalizedAnswer, verdict,
|
failedChecks.addAll(validateV2AuditClosure(evalCase, trace, normalizedAnswer, verdict,
|
||||||
gatekeeperStatus, gatekeeperRuleSetVersion, composerStatus));
|
gatekeeperStatus, gatekeeperRuleSetVersion, promptAuditVersion, composerStatus));
|
||||||
|
|
||||||
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
|
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
|
||||||
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
|
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
|
||||||
@@ -142,8 +147,10 @@ public class DiagnosisTraceEvaluator {
|
|||||||
.evidenceCoverage(evidenceCoverage)
|
.evidenceCoverage(evidenceCoverage)
|
||||||
.gatekeeperStatus(gatekeeperStatus)
|
.gatekeeperStatus(gatekeeperStatus)
|
||||||
.gatekeeperRuleSetVersion(gatekeeperRuleSetVersion)
|
.gatekeeperRuleSetVersion(gatekeeperRuleSetVersion)
|
||||||
|
.promptAuditVersion(promptAuditVersion)
|
||||||
.composerStatus(composerStatus)
|
.composerStatus(composerStatus)
|
||||||
.claimCheckCount(claimCheckCount)
|
.claimCheckCount(claimCheckCount)
|
||||||
|
.gatekeeperRuleCount(gatekeeperRuleCount)
|
||||||
.toolCallCount(toolCallCount)
|
.toolCallCount(toolCallCount)
|
||||||
.durationMs(durationMs)
|
.durationMs(durationMs)
|
||||||
.build();
|
.build();
|
||||||
@@ -237,6 +244,7 @@ public class DiagnosisTraceEvaluator {
|
|||||||
String verdict,
|
String verdict,
|
||||||
String gatekeeperStatus,
|
String gatekeeperStatus,
|
||||||
String gatekeeperRuleSetVersion,
|
String gatekeeperRuleSetVersion,
|
||||||
|
String promptAuditVersion,
|
||||||
String composerStatus) {
|
String composerStatus) {
|
||||||
List<String> failedChecks = new ArrayList<>();
|
List<String> failedChecks = new ArrayList<>();
|
||||||
boolean requireV2AuditClosure = Boolean.TRUE.equals(evalCase.getRequireV2AuditClosure());
|
boolean requireV2AuditClosure = Boolean.TRUE.equals(evalCase.getRequireV2AuditClosure());
|
||||||
@@ -259,6 +267,10 @@ public class DiagnosisTraceEvaluator {
|
|||||||
failedChecks.add("gatekeeper rule set version not expected: "
|
failedChecks.add("gatekeeper rule set version not expected: "
|
||||||
+ valueOrMissing(gatekeeperRuleSetVersion));
|
+ valueOrMissing(gatekeeperRuleSetVersion));
|
||||||
}
|
}
|
||||||
|
if (Boolean.TRUE.equals(evalCase.getRequireGatekeeperRules())) {
|
||||||
|
failedChecks.addAll(validateGatekeeperRules(trace));
|
||||||
|
}
|
||||||
|
failedChecks.addAll(validatePromptAudit(evalCase, trace, promptAuditVersion));
|
||||||
|
|
||||||
failedChecks.addAll(validateClaimChecks(trace, requireClaimChecks));
|
failedChecks.addAll(validateClaimChecks(trace, requireClaimChecks));
|
||||||
|
|
||||||
@@ -290,6 +302,78 @@ public class DiagnosisTraceEvaluator {
|
|||||||
return failedChecks;
|
return failedChecks;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private List<String> validatePromptAudit(DiagnosisEvalCase evalCase,
|
||||||
|
DiagnosisTraceResponse trace,
|
||||||
|
String promptAuditVersion) {
|
||||||
|
List<String> failedChecks = new ArrayList<>();
|
||||||
|
Object promptAudit = nestedValue(trace, "verifier_evaluation", "prompt_audit");
|
||||||
|
if (Boolean.TRUE.equals(evalCase.getRequirePromptAudit()) && !(promptAudit instanceof Map<?, ?>)) {
|
||||||
|
failedChecks.add("missing prompt_audit");
|
||||||
|
return failedChecks;
|
||||||
|
}
|
||||||
|
if (!isBlank(evalCase.getExpectedPromptAuditVersion())
|
||||||
|
&& !evalCase.getExpectedPromptAuditVersion().equals(promptAuditVersion)) {
|
||||||
|
failedChecks.add("prompt audit version not expected: " + valueOrMissing(promptAuditVersion));
|
||||||
|
}
|
||||||
|
if (evalCase.getExpectedPromptVersions() == null || evalCase.getExpectedPromptVersions().isEmpty()) {
|
||||||
|
return failedChecks;
|
||||||
|
}
|
||||||
|
if (!(promptAudit instanceof Map<?, ?> audit)) {
|
||||||
|
failedChecks.add("missing prompt_audit");
|
||||||
|
return failedChecks;
|
||||||
|
}
|
||||||
|
Object promptsValue = audit.get("prompts");
|
||||||
|
if (!(promptsValue instanceof List<?> prompts)) {
|
||||||
|
failedChecks.add("prompt_audit missing prompts");
|
||||||
|
return failedChecks;
|
||||||
|
}
|
||||||
|
Map<String, String> actualVersions = new LinkedHashMap<>();
|
||||||
|
for (Object promptValue : prompts) {
|
||||||
|
if (promptValue instanceof Map<?, ?> prompt) {
|
||||||
|
String name = stringValue(prompt.get("name"));
|
||||||
|
String version = stringValue(prompt.get("version"));
|
||||||
|
if (!isBlank(name)) {
|
||||||
|
actualVersions.put(name, version);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (Map.Entry<String, String> expected : evalCase.getExpectedPromptVersions().entrySet()) {
|
||||||
|
String actual = actualVersions.get(expected.getKey());
|
||||||
|
if (!expected.getValue().equals(actual)) {
|
||||||
|
failedChecks.add("prompt version not expected: "
|
||||||
|
+ expected.getKey() + "=" + valueOrMissing(actual));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return failedChecks;
|
||||||
|
}
|
||||||
|
|
||||||
|
private List<String> validateGatekeeperRules(DiagnosisTraceResponse trace) {
|
||||||
|
Object rulesValue = nestedNestedValue(trace, "verifier_evaluation", "gatekeeper_result", "rules");
|
||||||
|
if (!(rulesValue instanceof List<?> rules) || rules.isEmpty()) {
|
||||||
|
return List.of("gatekeeper_result missing rules");
|
||||||
|
}
|
||||||
|
List<String> failedChecks = new ArrayList<>();
|
||||||
|
for (Object item : rules) {
|
||||||
|
if (!(item instanceof Map<?, ?> rule)) {
|
||||||
|
failedChecks.add("gatekeeper rule metadata is not an object");
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
String id = stringValue(rule.get("id"));
|
||||||
|
Object enabled = rule.get("enabled");
|
||||||
|
String severity = stringValue(rule.get("default_severity"));
|
||||||
|
if (isBlank(id)) {
|
||||||
|
failedChecks.add("gatekeeper rule metadata missing id");
|
||||||
|
}
|
||||||
|
if (!(enabled instanceof Boolean)) {
|
||||||
|
failedChecks.add("gatekeeper rule metadata missing enabled: " + valueOrMissing(id));
|
||||||
|
}
|
||||||
|
if (isBlank(severity)) {
|
||||||
|
failedChecks.add("gatekeeper rule metadata missing default_severity: " + valueOrMissing(id));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return failedChecks;
|
||||||
|
}
|
||||||
|
|
||||||
private List<String> validateClaimChecks(DiagnosisTraceResponse trace, boolean required) {
|
private List<String> validateClaimChecks(DiagnosisTraceResponse trace, boolean required) {
|
||||||
Object claimChecks = nestedValue(trace, "verifier_evaluation", "claim_checks");
|
Object claimChecks = nestedValue(trace, "verifier_evaluation", "claim_checks");
|
||||||
if (!(claimChecks instanceof List<?> claimCheckList)) {
|
if (!(claimChecks instanceof List<?> claimCheckList)) {
|
||||||
@@ -347,6 +431,19 @@ public class DiagnosisTraceEvaluator {
|
|||||||
return value instanceof List<?> list ? list.size() : null;
|
return value instanceof List<?> list ? list.size() : null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private Integer countNestedList(DiagnosisTraceResponse trace, String firstKey, String secondKey, String thirdKey) {
|
||||||
|
Object value = nestedNestedValue(trace, firstKey, secondKey, thirdKey);
|
||||||
|
return value instanceof List<?> list ? list.size() : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
private Object nestedNestedValue(DiagnosisTraceResponse trace, String firstKey, String secondKey, String thirdKey) {
|
||||||
|
Object value = nestedValue(trace, firstKey, secondKey);
|
||||||
|
if (!(value instanceof Map<?, ?> map)) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
return map.get(thirdKey);
|
||||||
|
}
|
||||||
|
|
||||||
private int countMatches(String normalizedAnswer, List<String> keywords) {
|
private int countMatches(String normalizedAnswer, List<String> keywords) {
|
||||||
int count = 0;
|
int count = 0;
|
||||||
for (String keyword : safeList(keywords)) {
|
for (String keyword : safeList(keywords)) {
|
||||||
|
|||||||
@@ -60,6 +60,7 @@ public class ChatService {
|
|||||||
private static final Logger logger = LoggerFactory.getLogger(ChatService.class);
|
private static final Logger logger = LoggerFactory.getLogger(ChatService.class);
|
||||||
private static final String LOW_CONFID_DISCLAIMER = "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。";
|
private static final String LOW_CONFID_DISCLAIMER = "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。";
|
||||||
private static final String DEGRADED_PREFIX = "当前无法基于已获取证据生成可靠结论,建议人工介入。";
|
private static final String DEGRADED_PREFIX = "当前无法基于已获取证据生成可靠结论,建议人工介入。";
|
||||||
|
private static final String CHAT_PROMPT_AUDIT_VERSION = "chat-prompts-v1";
|
||||||
|
|
||||||
/** 封装 answer + 后端生成的 sessionId,用于 feedback 关联 */
|
/** 封装 answer + 后端生成的 sessionId,用于 feedback 关联 */
|
||||||
public record ChatResult(String answer, String sessionId) {}
|
public record ChatResult(String answer, String sessionId) {}
|
||||||
@@ -884,6 +885,7 @@ public class ChatService {
|
|||||||
verifierEvaluation.put("rationale", decision.rationale());
|
verifierEvaluation.put("rationale", decision.rationale());
|
||||||
verifierEvaluation.put("round", round);
|
verifierEvaluation.put("round", round);
|
||||||
verifierEvaluation.put("traceability_version", "v1");
|
verifierEvaluation.put("traceability_version", "v1");
|
||||||
|
verifierEvaluation.put("prompt_audit", promptAuditSnapshot());
|
||||||
verifierEvaluation.put("executor_output_parse_status",
|
verifierEvaluation.put("executor_output_parse_status",
|
||||||
Optional.ofNullable(VerifierContextHolder.getExecutorOutputParseStatus())
|
Optional.ofNullable(VerifierContextHolder.getExecutorOutputParseStatus())
|
||||||
.orElse(Map.of("status", "missing", "detail", "executor parse status unavailable")));
|
.orElse(Map.of("status", "missing", "detail", "executor parse status unavailable")));
|
||||||
@@ -902,6 +904,26 @@ public class ChatService {
|
|||||||
diagnosisSessionRepository.save(session);
|
diagnosisSessionRepository.save(session);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private Map<String, Object> promptAuditSnapshot() {
|
||||||
|
Map<String, Object> audit = new LinkedHashMap<>();
|
||||||
|
audit.put("version", CHAT_PROMPT_AUDIT_VERSION);
|
||||||
|
audit.put("prompts", List.of(
|
||||||
|
promptAuditItem("chat_planner", "chat-planner-v1", "prompts/chat-planner-prompt.md"),
|
||||||
|
promptAuditItem("chat_executor", "chat-executor-v2", "prompts/chat-executor-prompt.md"),
|
||||||
|
promptAuditItem("chat_verifier", "chat-verifier-v2", "prompts/chat-verifier-prompt.md"),
|
||||||
|
promptAuditItem("chat_composer", "chat-composer-v1", "prompts/chat-composer-prompt.md")
|
||||||
|
));
|
||||||
|
return audit;
|
||||||
|
}
|
||||||
|
|
||||||
|
private Map<String, Object> promptAuditItem(String name, String version, String resource) {
|
||||||
|
Map<String, Object> item = new LinkedHashMap<>();
|
||||||
|
item.put("name", name);
|
||||||
|
item.put("version", version);
|
||||||
|
item.put("resource", resource);
|
||||||
|
return item;
|
||||||
|
}
|
||||||
|
|
||||||
private Map<String, Object> defaultGatekeeperPass() {
|
private Map<String, Object> defaultGatekeeperPass() {
|
||||||
GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback();
|
GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback();
|
||||||
return Map.of(
|
return Map.of(
|
||||||
|
|||||||
@@ -100,13 +100,13 @@ class DiagnosisEvalBaselineDiffTest {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private void degradeRedisCase(DiagnosisEvalReport report) {
|
private void degradeRedisCase(DiagnosisEvalReport report) {
|
||||||
report.setPassedCases(9);
|
report.setPassedCases(11);
|
||||||
report.setPassRate(0.9);
|
report.setPassRate(11.0 / 12.0);
|
||||||
report.setAverageToolCallCount(3.0);
|
report.setAverageToolCallCount(3.0);
|
||||||
report.setAverageDurationMs(39800.0);
|
report.setAverageDurationMs(38500.0);
|
||||||
report.setVerdictDistribution(new LinkedHashMap<>());
|
report.setVerdictDistribution(new LinkedHashMap<>());
|
||||||
report.getVerdictDistribution().put("PASS", 4L);
|
report.getVerdictDistribution().put("PASS", 5L);
|
||||||
report.getVerdictDistribution().put("LOW_CONFID", 4L);
|
report.getVerdictDistribution().put("LOW_CONFID", 5L);
|
||||||
report.getVerdictDistribution().put("REJECT", 2L);
|
report.getVerdictDistribution().put("REJECT", 2L);
|
||||||
|
|
||||||
DiagnosisEvalResult redis = result(report, "redis-timeout");
|
DiagnosisEvalResult redis = result(report, "redis-timeout");
|
||||||
|
|||||||
@@ -24,11 +24,11 @@ class DiagnosisTraceEvaluatorTest {
|
|||||||
|
|
||||||
DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures"));
|
DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures"));
|
||||||
|
|
||||||
assertEquals(10, report.getTotalCases());
|
assertEquals(12, report.getTotalCases());
|
||||||
assertEquals(10, report.getPassedCases());
|
assertEquals(12, report.getPassedCases());
|
||||||
assertEquals(1.0, report.getPassRate(), 0.001);
|
assertEquals(1.0, report.getPassRate(), 0.001);
|
||||||
assertEquals(4L, report.getVerdictDistribution().get("PASS"));
|
assertEquals(5L, report.getVerdictDistribution().get("PASS"));
|
||||||
assertEquals(5L, report.getVerdictDistribution().get("LOW_CONFID"));
|
assertEquals(6L, report.getVerdictDistribution().get("LOW_CONFID"));
|
||||||
assertEquals(1L, report.getVerdictDistribution().get("REJECT"));
|
assertEquals(1L, report.getVerdictDistribution().get("REJECT"));
|
||||||
|
|
||||||
DiagnosisEvalResult narrowHighCpu = result(report, "narrow-highcpu-observation");
|
DiagnosisEvalResult narrowHighCpu = result(report, "narrow-highcpu-observation");
|
||||||
@@ -36,6 +36,12 @@ class DiagnosisTraceEvaluatorTest {
|
|||||||
assertEquals("gatekeeper-rules-v1", narrowHighCpu.getGatekeeperRuleSetVersion());
|
assertEquals("gatekeeper-rules-v1", narrowHighCpu.getGatekeeperRuleSetVersion());
|
||||||
assertEquals("pass", narrowHighCpu.getGatekeeperStatus());
|
assertEquals("pass", narrowHighCpu.getGatekeeperStatus());
|
||||||
|
|
||||||
|
DiagnosisEvalResult promptGatekeeperAudit = result(report, "prompt-gatekeeper-audit-closure");
|
||||||
|
assertTrue(promptGatekeeperAudit.isPassed());
|
||||||
|
assertEquals("chat-prompts-v1", promptGatekeeperAudit.getPromptAuditVersion());
|
||||||
|
assertEquals("gatekeeper-rules-v1", promptGatekeeperAudit.getGatekeeperRuleSetVersion());
|
||||||
|
assertEquals(2, promptGatekeeperAudit.getGatekeeperRuleCount());
|
||||||
|
|
||||||
DiagnosisEvalResult hikariNoEvidence = result(report, "hikari-no-evidence-negative-observation");
|
DiagnosisEvalResult hikariNoEvidence = result(report, "hikari-no-evidence-negative-observation");
|
||||||
assertTrue(hikariNoEvidence.isPassed());
|
assertTrue(hikariNoEvidence.isPassed());
|
||||||
assertEquals("gatekeeper-rules-v1", hikariNoEvidence.getGatekeeperRuleSetVersion());
|
assertEquals("gatekeeper-rules-v1", hikariNoEvidence.getGatekeeperRuleSetVersion());
|
||||||
@@ -60,6 +66,12 @@ class DiagnosisTraceEvaluatorTest {
|
|||||||
DiagnosisEvalResult composerFallback = result(report, "composer-fallback-no-raw-json");
|
DiagnosisEvalResult composerFallback = result(report, "composer-fallback-no-raw-json");
|
||||||
assertTrue(composerFallback.isPassed());
|
assertTrue(composerFallback.isPassed());
|
||||||
assertEquals("composer_malformed", composerFallback.getComposerStatus());
|
assertEquals("composer_malformed", composerFallback.getComposerStatus());
|
||||||
|
|
||||||
|
DiagnosisEvalResult auditMetadataLowConfid = result(report, "audit-metadata-low-confid");
|
||||||
|
assertTrue(auditMetadataLowConfid.isPassed());
|
||||||
|
assertEquals("LOW_CONFID", auditMetadataLowConfid.getVerdict());
|
||||||
|
assertEquals("chat-prompts-v1", auditMetadataLowConfid.getPromptAuditVersion());
|
||||||
|
assertEquals(2, auditMetadataLowConfid.getGatekeeperRuleCount());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -195,6 +207,76 @@ class DiagnosisTraceEvaluatorTest {
|
|||||||
"gatekeeper rule set version not expected: old-rules"));
|
"gatekeeper rule set version not expected: old-rules"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void evaluateFailsWhenPromptAuditMissingOrMismatches() {
|
||||||
|
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
|
||||||
|
.id("prompt-audit")
|
||||||
|
.title("Prompt audit")
|
||||||
|
.expectedRootCauseKeywords(List.of())
|
||||||
|
.requiredEvidenceTools(List.of())
|
||||||
|
.allowedVerdicts(List.of("PASS"))
|
||||||
|
.requirePromptAudit(true)
|
||||||
|
.expectedPromptAuditVersion("chat-prompts-v1")
|
||||||
|
.expectedPromptVersions(java.util.Map.of("chat_executor", "chat-executor-v2"))
|
||||||
|
.build();
|
||||||
|
DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder()
|
||||||
|
.session(DiagnosisTraceResponse.SessionTrace.builder()
|
||||||
|
.answer("安全回答")
|
||||||
|
.selfEvaluation(java.util.Map.of(
|
||||||
|
"verifier_evaluation", java.util.Map.of(
|
||||||
|
"verdict", "PASS",
|
||||||
|
"prompt_audit", java.util.Map.of(
|
||||||
|
"version", "old-prompts",
|
||||||
|
"prompts", java.util.List.of(java.util.Map.of(
|
||||||
|
"name", "chat_executor",
|
||||||
|
"version", "chat-executor-v1"
|
||||||
|
))
|
||||||
|
)
|
||||||
|
)))
|
||||||
|
.build())
|
||||||
|
.toolInvocations(List.of())
|
||||||
|
.build();
|
||||||
|
|
||||||
|
DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace);
|
||||||
|
|
||||||
|
assertFalse(result.isPassed());
|
||||||
|
assertTrue(result.getFailedChecks().contains(
|
||||||
|
"prompt audit version not expected: old-prompts"));
|
||||||
|
assertTrue(result.getFailedChecks().contains(
|
||||||
|
"prompt version not expected: chat_executor=chat-executor-v1"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void evaluateFailsWhenGatekeeperRulesAreMissing() {
|
||||||
|
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
|
||||||
|
.id("gatekeeper-rules")
|
||||||
|
.title("Gatekeeper rules")
|
||||||
|
.expectedRootCauseKeywords(List.of())
|
||||||
|
.requiredEvidenceTools(List.of())
|
||||||
|
.allowedVerdicts(List.of("PASS"))
|
||||||
|
.requireGatekeeperRules(true)
|
||||||
|
.build();
|
||||||
|
DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder()
|
||||||
|
.session(DiagnosisTraceResponse.SessionTrace.builder()
|
||||||
|
.answer("安全回答")
|
||||||
|
.selfEvaluation(java.util.Map.of(
|
||||||
|
"verifier_evaluation", java.util.Map.of(
|
||||||
|
"verdict", "PASS",
|
||||||
|
"gatekeeper_result", java.util.Map.of(
|
||||||
|
"status", "pass",
|
||||||
|
"rule_set_version", "gatekeeper-rules-v1"
|
||||||
|
)
|
||||||
|
)))
|
||||||
|
.build())
|
||||||
|
.toolInvocations(List.of())
|
||||||
|
.build();
|
||||||
|
|
||||||
|
DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace);
|
||||||
|
|
||||||
|
assertFalse(result.isPassed());
|
||||||
|
assertTrue(result.getFailedChecks().contains("gatekeeper_result missing rules"));
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void evaluateFailsWhenUnsupportedClaimLeaksIntoFinalAnswer() {
|
void evaluateFailsWhenUnsupportedClaimLeaksIntoFinalAnswer() {
|
||||||
|
|||||||
@@ -394,10 +394,20 @@ class ChatServiceSequentialAgentTest {
|
|||||||
verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture());
|
verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture());
|
||||||
Map<String, Object> verifierEvaluation = captor.getValue();
|
Map<String, Object> verifierEvaluation = captor.getValue();
|
||||||
assertTrue(verifierEvaluation.containsKey("gatekeeper_result"));
|
assertTrue(verifierEvaluation.containsKey("gatekeeper_result"));
|
||||||
|
assertTrue(verifierEvaluation.containsKey("prompt_audit"));
|
||||||
@SuppressWarnings("unchecked")
|
@SuppressWarnings("unchecked")
|
||||||
Map<String, Object> gatekeeperResult = (Map<String, Object>) verifierEvaluation.get("gatekeeper_result");
|
Map<String, Object> gatekeeperResult = (Map<String, Object>) verifierEvaluation.get("gatekeeper_result");
|
||||||
assertEquals("pass", gatekeeperResult.get("status"));
|
assertEquals("pass", gatekeeperResult.get("status"));
|
||||||
assertEquals("none", gatekeeperResult.get("severity"));
|
assertEquals("none", gatekeeperResult.get("severity"));
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
Map<String, Object> promptAudit = (Map<String, Object>) verifierEvaluation.get("prompt_audit");
|
||||||
|
assertEquals("chat-prompts-v1", promptAudit.get("version"));
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
List<Map<String, Object>> prompts = (List<Map<String, Object>>) promptAudit.get("prompts");
|
||||||
|
assertEquals(4, prompts.size());
|
||||||
|
assertTrue(prompts.stream().anyMatch(prompt ->
|
||||||
|
"chat_executor".equals(prompt.get("name"))
|
||||||
|
&& "chat-executor-v2".equals(prompt.get("version"))));
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
|
|||||||
Reference in New Issue
Block a user