diff --git a/devflow/index.md b/devflow/index.md index f7aaf71..8f44023 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -2,8 +2,9 @@ ## 项目 -| 日期 | slug | 领域 | 关键词 | 状态 | -|---|---|---|---|---| +| 日期 | slug | 领域 | 关键词 | 关联 OpenSpec | 状态 | +|---|---|---|---|---|---| +| 2026-07-09 | interview-demo-quality-audit | Agent eval/demo/Prompt audit | interview demo preflight, prompt_audit, gatekeeper rules, diagnosis baseline, 12 fixtures | openspec/changes/archive/2026-07-09-interview-demo-quality-audit | archived | | 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived | | 2026-07-08 | diagnosis-eval-demo-gatekeeper-closure | Agent eval/demo/Gatekeeper | diagnosis eval matrix, stable demo scenarios, Gatekeeper rule set version, audit metadata | openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure | archived | | 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived | diff --git a/devflow/projects/2026-07-09-interview-demo-quality-audit/acceptance.md b/devflow/projects/2026-07-09-interview-demo-quality-audit/acceptance.md index 1f05c0f..6e223d1 100644 --- a/devflow/projects/2026-07-09-interview-demo-quality-audit/acceptance.md +++ b/devflow/projects/2026-07-09-interview-demo-quality-audit/acceptance.md @@ -1,10 +1,46 @@ # Acceptance -Acceptance will be filled during Apply/Archive with concrete command output. +## Static Verification -Planned checks: +- `openspec validate interview-demo-quality-audit --strict` + - Result: passed. + - Coverage: OpenSpec proposal/design/spec/tasks consistency. +- PowerShell parser/runtime readiness check: + - Command: `powershell -NoProfile -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 -BaseUrl http://127.0.0.1:1 -OutputDir target/demo-check-syntax` + - Result: expected failure with actionable readiness message. + - Coverage: script parses under Windows PowerShell and fails before issuing diagnosis requests when service is unreachable. -- `openspec validate interview-demo-quality-audit --strict` - passed during commit gate. +## Script Verification + +- `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test` + - Result: passed. + - Coverage: 12/12 fixed eval fixtures, Prompt audit evaluator checks, Gatekeeper rule metadata checks, regenerated baseline reports. +- `mvn -q "-Dtest=ChatServiceSequentialAgentTest" test` + - Result: passed. + - Coverage: Chat verifier evaluation persists `prompt_audit`. - `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ChatServiceSequentialAgentTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest" test` + - Result: passed. + - Coverage: broader eval, baseline diff, Chat sequential flow, Gatekeeper, and Verifier input hook regression set. - `mvn -q -DskipTests compile` -- `powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1` against `mvp-demo` service if dependencies are available. + - Result: passed. + - Coverage: main source compilation. + +## E2E Verification + +- Started service with: + - `mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo` +- Ran: + - `powershell -NoProfile -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 -BaseUrl http://localhost:9900 -SessionId mvp-demo-interview-quality-audit-001` +- Result: passed. +- Summary: + - `chatSuccess=true` + - `verdict=LOW_CONFID` + - `gatekeeperStatus=fail` + - `gatekeeperRuleSetVersion=gatekeeper-rules-v1` + - `promptAuditVersion=chat-prompts-v1` + - tools included `lookup_knowledge`, `query_logs`, `query_metrics`, and `get_available_log_topics` +- Note: live E2E remains a compatibility check, not the deterministic PASS oracle. The fixed fixture baseline is the regression source of truth. + +## Not Verified + +- Browser UI inspection was not required for this change because the scope is backend trace/eval/demo script documentation, not frontend behavior. diff --git a/devflow/projects/2026-07-09-interview-demo-quality-audit/decisions.md b/devflow/projects/2026-07-09-interview-demo-quality-audit/decisions.md index bb2bcc7..fb2d7cd 100644 --- a/devflow/projects/2026-07-09-interview-demo-quality-audit/decisions.md +++ b/devflow/projects/2026-07-09-interview-demo-quality-audit/decisions.md @@ -89,3 +89,23 @@ Architecture risk assessment: - Commit completed. - Apply is authorized by the original objective: "完成后归档提交". + +## Pre-apply Research + +- Capability source: sm-flow built-in apply protocol. `openspec-apply-change` was not invoked directly in this session. +- Repository semantic search/LSP note: the requested `codebase-retrieval` and LSP tools were not available in the exposed toolset, so impact analysis used `rg`, direct file reads, OpenSpec/devflow artifacts, and targeted tests. +- Reference implementation and reuse: + - `ChatService.persistVerifierEvaluation(...)` is the single persistence point for Chat verifier/composer audit data; prompt audit was added there to cover normal, fallback, and degraded Composer paths. + - `DiagnosisTraceEvaluator` and `DiagnosisEvalReportWriter` are the deterministic eval extension points; no LLM judge was introduced. + - `mvp/demo/scripts/run-payment-timeout-demo.ps1` provided the request/trace/feedback flow reused by the new interview preflight script. +- Interface impact remains L2 internal trace contract: `verifier_evaluation.prompt_audit` and eval report fields are added; no public endpoint, table, or request DTO changed. + +## Apply Notes + +- Added compact Chat prompt audit metadata: `chat-prompts-v1`, with planner/executor/verifier/composer prompt versions and resource paths. +- Extended diagnosis eval schema, result reporting, baseline fixtures, JSON report, and Markdown report for Prompt audit and Gatekeeper rule metadata. +- Added two fixture-backed audit cases: + - `prompt-gatekeeper-audit-closure` + - `audit-metadata-low-confid` +- Added `mvp/demo/scripts/run-interview-demo-check.ps1` to run service readiness, Chat, Trace, feedback, and summary output. +- Updated MVP demo/eval/architecture docs to explain `prompt_audit.version`, `gatekeeper_result.rule_set_version`, and deterministic fixture baseline. diff --git a/devflow/projects/2026-07-09-interview-demo-quality-audit/evidence.md b/devflow/projects/2026-07-09-interview-demo-quality-audit/evidence.md index 76d6616..d242dbd 100644 --- a/devflow/projects/2026-07-09-interview-demo-quality-audit/evidence.md +++ b/devflow/projects/2026-07-09-interview-demo-quality-audit/evidence.md @@ -23,3 +23,36 @@ The required `codebase-retrieval` and LSP tools were not exposed in this session. Impact analysis used `rg`, direct file reads, existing OpenSpec/devflow artifacts, and targeted tests instead. +## Implementation Evidence + +- `src/main/java/com/superbiz/agent/service/ChatService.java` + - Adds `prompt_audit` under `verifier_evaluation` through the shared `persistVerifierEvaluation(...)` path. + - Uses compact metadata only: audit version, prompt names, prompt versions, and resource paths. +- `src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java` + - Adds deterministic checks for `requirePromptAudit`, `expectedPromptAuditVersion`, `expectedPromptVersions`, and `requireGatekeeperRules`. +- `src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java` + - Adds Prompt Audit and Gatekeeper rule count columns to Markdown reports. +- `mvp/eval/cases/diagnosis-cases.json` + - Expands fixed baseline to 12 fixture-backed cases. +- `mvp/eval/fixtures/prompt-gatekeeper-audit-closure-pass.json` + - Positive PASS fixture proving Prompt audit and Gatekeeper rule metadata closure. +- `mvp/eval/fixtures/audit-metadata-low-confid.json` + - LOW_CONFID fixture proving safe answer behavior while audit metadata remains present. +- `mvp/demo/scripts/run-interview-demo-check.ps1` + - Adds service readiness, Chat, Trace, feedback, and summary output for interview preflight. + +## Verification Evidence + +- OpenSpec: + - `openspec validate interview-demo-quality-audit --strict`: passed before archive. + - `openspec validate --specs --strict`: 10 specs passed after merging deltas into main specs. +- Unit/eval: + - `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`: passed. + - `mvn -q "-Dtest=ChatServiceSequentialAgentTest" test`: passed. + - `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ChatServiceSequentialAgentTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`: passed. +- Compile: + - `mvn -q -DskipTests compile`: passed. +- E2E: + - Started `mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo`. + - Ran `mvp/demo/scripts/run-interview-demo-check.ps1` against `http://localhost:9900`. + - Summary recorded `chatSuccess=true`, `verdict=LOW_CONFID`, `gatekeeperRuleSetVersion=gatekeeper-rules-v1`, and `promptAuditVersion=chat-prompts-v1`. diff --git a/mvp/architecture/harness-quality-gates.md b/mvp/architecture/harness-quality-gates.md index 0863280..2e075ab 100644 --- a/mvp/architecture/harness-quality-gates.md +++ b/mvp/architecture/harness-quality-gates.md @@ -82,6 +82,23 @@ Prompt 层当前承担的门禁: - Chat Composer 不允许补事实,尤其不能把 `$.no_evidence` 表达为“已排除/确认没有”。 - AIOps payload 模式必须聚焦输入告警。 +Chat 链路还会在 `verifier_evaluation.prompt_audit` 中持久化紧凑 Prompt 审计快照: + +```json +{ + "version": "chat-prompts-v1", + "prompts": [ + { + "name": "chat_executor", + "version": "chat-executor-v2", + "resource": "prompts/chat-executor-prompt.md" + } + ] +} +``` + +该快照只保存版本和资源路径,不保存完整 Prompt 文本。它用于面试演示、trace 回放和离线 baseline 解释“本次诊断使用了哪套 Prompt 契约”。 + ## 4. Trace Hooks `AgentLoggingHook` 是当前 Agent step 可观测性的核心。 @@ -196,7 +213,7 @@ Verifier 不再逐字核验 excerpt 真伪;这由 Gatekeeper 完成。Verifier diagnosis_session.self_evaluation.verifier_evaluation ``` -其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary` 和 `composer_output`,用于 Trace 回放。 +其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary`、`prompt_audit` 和 `composer_output`,用于 Trace 回放。 ## 7. AIOps 规则门禁 @@ -233,7 +250,7 @@ diagnosis_session.self_evaluation.aiops_rule_evaluation - 同一工具调用次数上限。 - 工具超时的统一熔断。 - Gatekeeper 规则远程化或三层分离:索引层、元数据层、规则实现层。 -- Prompt 版本记录和回滚。 +- Prompt 版本回滚和更细粒度变更审计。 - Verifier 对 AIOps 报告的 LLM 级事实校验。 这些应在评测集扩大后逐步加入,避免一次性把诊断流程卡得过死。 diff --git a/mvp/demo/README.md b/mvp/demo/README.md index d01b681..0743be5 100644 --- a/mvp/demo/README.md +++ b/mvp/demo/README.md @@ -8,7 +8,9 @@ - `interview-walkthrough.md`:面试讲解话术。 - `evidence-pipeline-scenarios.md`:PASS / LOW_CONFID / REJECT / no-evidence 场景矩阵。 - `trace-inspection-checklist.md`:Trace 字段检查清单。 +- `scripts/run-interview-demo-check.ps1`:面试预检脚本,包含服务可达性、Chat、Trace、反馈和 summary 输出。 - `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。 +- `interview-q-and-a.md`:面试追问回答,覆盖 Agent 工程取舍、审计和评测。 - `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。 - `requests/narrow-highcpu-chat.json`:窄范围正向观察请求。 - `requests/hikari-no-evidence-chat.json`:no-evidence 负向观察请求。 @@ -37,7 +39,7 @@ http://localhost:9900 最快方式: ```powershell -powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1 +powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 ``` 脚本会生成: @@ -46,6 +48,7 @@ powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-de mvp/demo/output/chat-response.json mvp/demo/output/trace-response.json mvp/demo/output/feedback-response.json +mvp/demo/output/interview-demo-summary.json ``` 手动请求: @@ -85,6 +88,8 @@ Invoke-RestMethod ` - `data.steps` 包含 planner / executor / verifier 等步骤 - `data.toolInvocations` 包含 `lookup_knowledge`、`query_logs`、`query_metrics` 等证据工具 - `data.session.selfEvaluation` 包含 verifier 或 rule evaluation +- Chat V2 链路中,`data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` 记录 Chat Prompt 审计版本 +- Chat V2 链路中,`data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` 记录 Gatekeeper 规则集版本 ## 5. 提交反馈 @@ -176,8 +181,8 @@ AIOps 主线: 面试时不要把所有安全场景都压到 live LLM 现场表现上。建议使用: -- `scripts/run-payment-timeout-demo.ps1` 跑主路径。 +- `scripts/run-interview-demo-check.ps1` 跑主路径和预检 summary。 - `evidence-pipeline-scenarios.md` 讲解 PASS / LOW_CONFID / REJECT / no-evidence 矩阵。 -- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 10/10 通过。 +- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 12/12 通过。 这样可以同时展示真实链路和确定性回归能力。 diff --git a/mvp/demo/interview-q-and-a.md b/mvp/demo/interview-q-and-a.md new file mode 100644 index 0000000..5d57085 --- /dev/null +++ b/mvp/demo/interview-q-and-a.md @@ -0,0 +1,33 @@ +# 面试追问 Q&A + +## 为什么不用普通 Chatbot? + +这个项目的重点不是生成一段诊断文本,而是把诊断拆成可审计链路:Planner 拆解问题,Executor 调工具拿证据,Gatekeeper 用代码核验证据引用,Verifier 判断可推导性,Composer 生成最终表达。每次运行都能通过同一个 `sessionId` 回放。 + +## 为什么 RAG 要做成显式工具? + +`lookup_knowledge` 保持显式工具调用,才能在 `tool_invocation` 里看到 Agent 查了什么、命中了什么、相关性等级是什么,以及最终答案是否真的使用了这些证据。隐式 Advisor 更方便,但不利于审计 Agent 决策。 + +## 怎么防止 Executor 幻觉? + +Executor 不直接负责最终用户答案,而是输出 `executor_evidence_v2` 的微观事实和证据引用。Gatekeeper 会校验 `source_invocation_id`、`raw_path`、`evidence_excerpt` 是否真实存在;Verifier 再判断 claim 是否能由已验真的证据推出;Composer 只表达 Verifier 允许输出的内容。 + +## LOW_CONFID 是失败吗? + +不是。`LOW_CONFID` 表示当前证据不足以支撑强结论,但系统仍然可以安全表达已确认事实和缺失信息。面试时可以把它作为“没有证据就不强答”的质量门禁,而不是模型能力失败。 + +## Prompt 改了怎么审计? + +Chat verifier evaluation 里会记录 `prompt_audit.version`,并列出 planner、executor、verifier、composer 的 Prompt 版本和资源路径。它不保存完整 Prompt 文本,只保留用于回放和回归解释的紧凑元数据。 + +## Gatekeeper 改了怎么审计? + +Gatekeeper 结果里记录 `gatekeeper_result.rule_set_version` 和已启用规则元数据摘要。规则执行仍是确定性 Java 代码,版本和规则元数据用于解释“这次引用验真用的是哪套规则”。 + +## 为什么现在不拆 SubAgent? + +当前 MVP 的主要风险不是 Agent 数量不够,而是证据、验证和回归是否稳定。文档里的演进路线把 SubAgent 放在 P2:等故障类型、工具权限和评测集足够明确后再拆,避免只是移动复杂度。 + +## 为什么 baseline 比 live demo 更重要? + +live demo 证明链路在当前环境能跑通,但 LLM 和外部依赖会波动。`mvp/eval` 的固定 fixture baseline 是确定性回归来源,用来判断 Prompt、工具、Gatekeeper、Verifier 或 Composer 的改动有没有让系统退化。 diff --git a/mvp/demo/scripts/run-interview-demo-check.ps1 b/mvp/demo/scripts/run-interview-demo-check.ps1 new file mode 100644 index 0000000..df11a41 --- /dev/null +++ b/mvp/demo/scripts/run-interview-demo-check.ps1 @@ -0,0 +1,161 @@ +param( + [string]$BaseUrl = "http://localhost:9900", + [string]$SessionId = "mvp-demo-interview-payment-timeout-001", + [string]$RequestFile = "$PSScriptRoot/../requests/payment-timeout-chat.json", + [string]$OutputDir = "$PSScriptRoot/../output" +) + +$ErrorActionPreference = "Stop" + +function Test-ServiceReachable { + param([string]$Url) + + try { + $request = [System.Net.WebRequest]::Create($Url) + $request.Method = "GET" + $request.Timeout = 5000 + $response = $request.GetResponse() + $response.Close() + return $true + } catch [System.Net.WebException] { + if ($_.Exception.Response -ne $null) { + $_.Exception.Response.Close() + return $true + } + return $false + } +} + +function Get-TraceData { + param($TraceResponse) + + if ($TraceResponse.PSObject.Properties.Name -contains "data") { + return $TraceResponse.data + } + return $TraceResponse +} + +function Get-SelfEvaluation { + param($TraceData) + + if ($null -eq $TraceData -or $null -eq $TraceData.session) { + return $null + } + return $TraceData.session.selfEvaluation +} + +function Get-ToolNames { + param($TraceData) + + if ($null -eq $TraceData -or $null -eq $TraceData.toolInvocations) { + return @() + } + return @($TraceData.toolInvocations | ForEach-Object { $_.toolName } | Where-Object { $_ } | Sort-Object -Unique) +} + +New-Item -ItemType Directory -Force -Path $OutputDir | Out-Null + +Write-Host "Running interview demo preflight..." +Write-Host "BaseUrl: $BaseUrl" +Write-Host "SessionId: $SessionId" + +if (-not (Test-ServiceReachable -Url $BaseUrl)) { + throw "Service is not reachable: $BaseUrl. Start the app with mvp-demo profile first: mvn spring-boot:run -Dspring-boot.run.profiles=mvp-demo" +} + +$request = Get-Content -Raw -Encoding UTF8 -Path $RequestFile | ConvertFrom-Json +$request.Id = $SessionId +$body = $request | ConvertTo-Json -Depth 8 + +$chatRequest = @{ + Method = "Post" + Uri = "$BaseUrl/api/chat" + ContentType = "application/json; charset=utf-8" + Body = $body +} +$chat = Invoke-RestMethod @chatRequest + +$chatPath = Join-Path $OutputDir "chat-response.json" +$chat | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $chatPath + +$traceRequest = @{ + Method = "Get" + Uri = "$BaseUrl/api/diagnosis/$SessionId/trace" +} +$trace = Invoke-RestMethod @traceRequest + +$tracePath = Join-Path $OutputDir "trace-response.json" +$trace | ConvertTo-Json -Depth 80 | Set-Content -Encoding UTF8 -Path $tracePath + +$feedbackBody = @{ + sessionId = $SessionId + feedback = "useful" +} | ConvertTo-Json + +$feedbackRequest = @{ + Method = "Post" + Uri = "$BaseUrl/api/feedback" + ContentType = "application/json; charset=utf-8" + Body = $feedbackBody +} +$feedback = Invoke-RestMethod @feedbackRequest + +$feedbackPath = Join-Path $OutputDir "feedback-response.json" +$feedback | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $feedbackPath + +$traceData = Get-TraceData -TraceResponse $trace +$selfEvaluation = Get-SelfEvaluation -TraceData $traceData +$verifierEvaluation = $null +if ($null -ne $selfEvaluation) { + $verifierEvaluation = $selfEvaluation.verifier_evaluation +} + +$gatekeeperResult = $null +$promptAudit = $null +if ($null -ne $verifierEvaluation) { + $gatekeeperResult = $verifierEvaluation.gatekeeper_result + $promptAudit = $verifierEvaluation.prompt_audit +} + +$verdict = $null +$gatekeeperStatus = $null +$gatekeeperRuleSetVersion = $null +$promptAuditVersion = $null +if ($null -ne $verifierEvaluation) { + $verdict = $verifierEvaluation.verdict +} +if ($null -ne $gatekeeperResult) { + $gatekeeperStatus = $gatekeeperResult.status + $gatekeeperRuleSetVersion = $gatekeeperResult.rule_set_version +} +if ($null -ne $promptAudit) { + $promptAuditVersion = $promptAudit.version +} +$toolNames = Get-ToolNames -TraceData $traceData +$summaryPath = Join-Path $OutputDir "interview-demo-summary.json" + +$summary = [ordered]@{ + sessionId = $SessionId + baseUrl = $BaseUrl + chatSuccess = $chat.data.success + verdict = $verdict + gatekeeperStatus = $gatekeeperStatus + gatekeeperRuleSetVersion = $gatekeeperRuleSetVersion + promptAuditVersion = $promptAuditVersion + toolNames = $toolNames + paths = [ordered]@{ + chat = $chatPath + trace = $tracePath + feedback = $feedbackPath + summary = $summaryPath + } +} + +$summary | ConvertTo-Json -Depth 20 | Set-Content -Encoding UTF8 -Path $summaryPath + +Write-Host "" +Write-Host "Interview demo preflight completed." +Write-Host "Verdict: $($summary.verdict)" +Write-Host "Gatekeeper rules: $($summary.gatekeeperRuleSetVersion)" +Write-Host "Prompt audit: $($summary.promptAuditVersion)" +Write-Host "Summary: $summaryPath" diff --git a/mvp/demo/ten-minute-interview-demo.md b/mvp/demo/ten-minute-interview-demo.md index e6f1cc7..9b54887 100644 --- a/mvp/demo/ten-minute-interview-demo.md +++ b/mvp/demo/ten-minute-interview-demo.md @@ -37,7 +37,7 @@ http://localhost:9900 推荐使用固定脚本: ```powershell -powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1 +powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-interview-demo-check.ps1 ``` 脚本会写出: @@ -46,6 +46,7 @@ powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-de mvp/demo/output/chat-response.json mvp/demo/output/trace-response.json mvp/demo/output/feedback-response.json +mvp/demo/output/interview-demo-summary.json ``` 现场话术: @@ -98,6 +99,8 @@ data.toolInvocations[*].outputPreview data.toolInvocations[*].retrievalLayer data.toolInvocations[*].relevanceLevel data.summary.hasVerifierEvaluation +data.session.selfEvaluation.verifier_evaluation.prompt_audit.version +data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version ``` 现场话术: @@ -155,6 +158,10 @@ Verifier 不做新检索,只看工具 trace 汇总。 如果 PASS,就输出原答案。 如果 LOW_CONFID,可以补证据或加低置信提示。 如果 REJECT,就降级输出,只保留已确认信息。 + +Prompt 和 Gatekeeper 的版本也会进入 trace。 +`prompt_audit.version` 用于说明本次 Chat 使用哪套 Prompt 契约,`gatekeeper_result.rule_set_version` 用于说明引用验真的规则版本。 +固定 fixture baseline 是回归判断来源,live demo 主要证明当前环境链路可跑通。 ``` ## 7. 展示反馈闭环 @@ -226,6 +233,7 @@ AIOps 有两个模式。 mvp/demo/output/chat-response.json mvp/demo/output/trace-response.json mvp/demo/output/feedback-response.json +mvp/demo/output/interview-demo-summary.json ``` 降级话术: diff --git a/mvp/demo/trace-inspection-checklist.md b/mvp/demo/trace-inspection-checklist.md index 4f451aa..9cdd08a 100644 --- a/mvp/demo/trace-inspection-checklist.md +++ b/mvp/demo/trace-inspection-checklist.md @@ -1,6 +1,6 @@ # Trace 检查清单 -运行 `scripts/run-payment-timeout-demo.ps1` 后,用这份清单检查 `trace-response.json`。 +运行 `scripts/run-interview-demo-check.ps1` 后,用这份清单检查 `trace-response.json` 和 `interview-demo-summary.json`。 ## 1. Session @@ -11,6 +11,8 @@ | `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace | | `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 | | `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 如果是 Chat V2 链路,是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 | +| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` | 如果是 Chat V2 链路,是否记录 Prompt 审计版本 | Prompt 变更可解释、可回归 | +| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.prompts[*].version` | 是否记录 planner / executor / verifier / composer 版本 | 便于定位 Prompt 变更影响 | | `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在同一次诊断上 | ## 2. Agent 步骤 diff --git a/mvp/eval/README.md b/mvp/eval/README.md index f838f95..27849a3 100644 --- a/mvp/eval/README.md +++ b/mvp/eval/README.md @@ -29,10 +29,10 @@ The baseline evaluates saved trace fixtures. It does not start the application a The committed baseline currently contains: ```text -10 fixed cases -10 passing fixture evaluations -4 PASS verdicts -5 LOW_CONFID verdicts +12 fixed cases +12 passing fixture evaluations +5 PASS verdicts +6 LOW_CONFID verdicts 1 REJECT verdict ``` @@ -44,6 +44,8 @@ The V2 evidence-pipeline matrix covers: - Unsupported claim filtering before the final answer. - Composer fallback rendering without raw Executor JSON leakage. - Gatekeeper rule set version audit for new matrix fixtures. +- Prompt audit version checks for planner, executor, verifier, and composer prompts. +- Gatekeeper rule metadata checks for enabled rule id and default severity. ## Verification @@ -80,6 +82,8 @@ Stage 5 adds these V2 checks: - `claim_checks` must be structurally auditable. - Composer output must record whether normal parsing or fallback rendering was used. - Gatekeeper rule set version can be asserted per fixture. +- Prompt audit version and per-prompt versions can be asserted per fixture. +- Gatekeeper rule metadata can be required per fixture. - Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`. - Configured unsupported claim keywords must not appear as confirmed final-answer content. diff --git a/mvp/eval/cases/diagnosis-cases.json b/mvp/eval/cases/diagnosis-cases.json index 8ec495e..9faa47a 100644 --- a/mvp/eval/cases/diagnosis-cases.json +++ b/mvp/eval/cases/diagnosis-cases.json @@ -17,6 +17,33 @@ "expectedComposerStatuses": ["valid"], "forbiddenConfirmedClaimKeywords": ["数据库连接池"] }, + { + "id": "prompt-gatekeeper-audit-closure", + "title": "Prompt and Gatekeeper audit closure", + "question": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。", + "traceFixture": "prompt-gatekeeper-audit-closure-pass.json", + "expectedRootCauseKeywords": ["payment-service", "HighCPUUsage", "92%"], + "minKeywordMatches": 2, + "requiredEvidenceTools": ["query_metrics"], + "allowedVerdicts": ["PASS"], + "forbiddenAnswerKeywords": ["根因", "修复建议", "通常情况下"], + "requireV2AuditClosure": true, + "requireClaimChecks": true, + "requireComposerOutput": true, + "expectedGatekeeperStatuses": ["pass"], + "expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1", + "expectedComposerStatuses": ["valid"], + "forbiddenConfirmedClaimKeywords": ["数据库连接池"], + "requirePromptAudit": true, + "expectedPromptAuditVersion": "chat-prompts-v1", + "expectedPromptVersions": { + "chat_planner": "chat-planner-v1", + "chat_executor": "chat-executor-v2", + "chat_verifier": "chat-verifier-v2", + "chat_composer": "chat-composer-v1" + }, + "requireGatekeeperRules": true + }, { "id": "hikari-no-evidence-negative-observation", "title": "Hikari no-evidence negative observation", @@ -124,6 +151,33 @@ "expectedComposerStatuses": ["valid"], "forbiddenConfirmedClaimKeywords": ["主库故障"] }, + { + "id": "audit-metadata-low-confid", + "title": "Audit metadata low confidence", + "question": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?", + "traceFixture": "audit-metadata-low-confid.json", + "expectedRootCauseKeywords": ["超时", "证据"], + "minKeywordMatches": 2, + "requiredEvidenceTools": ["query_logs"], + "allowedVerdicts": ["LOW_CONFID"], + "forbiddenAnswerKeywords": ["已经确认"], + "requireV2AuditClosure": true, + "requireClaimChecks": true, + "requireComposerOutput": true, + "expectedGatekeeperStatuses": ["pass"], + "expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1", + "expectedComposerStatuses": ["valid"], + "forbiddenConfirmedClaimKeywords": ["主库故障"], + "requirePromptAudit": true, + "expectedPromptAuditVersion": "chat-prompts-v1", + "expectedPromptVersions": { + "chat_planner": "chat-planner-v1", + "chat_executor": "chat-executor-v2", + "chat_verifier": "chat-verifier-v2", + "chat_composer": "chat-composer-v1" + }, + "requireGatekeeperRules": true + }, { "id": "composer-fallback-no-raw-json", "title": "Composer fallback no raw JSON", diff --git a/mvp/eval/fixtures/audit-metadata-low-confid.json b/mvp/eval/fixtures/audit-metadata-low-confid.json new file mode 100644 index 0000000..eef04f7 --- /dev/null +++ b/mvp/eval/fixtures/audit-metadata-low-confid.json @@ -0,0 +1,192 @@ +{ + "session": { + "sessionId": "eval-audit-metadata-low-confid", + "query": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?", + "status": "SUCCESS", + "agentFlow": "CHAT", + "totalDurationMs": 45000, + "toolCallCount": 1, + "answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。", + "selfEvaluation": { + "verifier_evaluation": { + "verdict": "LOW_CONFID", + "groundedness_score": 0.42, + "critical_fact_count": 2, + "prompt_audit": { + "version": "chat-prompts-v1", + "prompts": [ + { + "name": "chat_planner", + "version": "chat-planner-v1", + "resource": "prompts/chat-planner-prompt.md" + }, + { + "name": "chat_executor", + "version": "chat-executor-v2", + "resource": "prompts/chat-executor-prompt.md" + }, + { + "name": "chat_verifier", + "version": "chat-verifier-v2", + "resource": "prompts/chat-verifier-prompt.md" + }, + { + "name": "chat_composer", + "version": "chat-composer-v1", + "resource": "prompts/chat-composer-prompt.md" + } + ] + }, + "gatekeeper_result": { + "status": "pass", + "severity": "none", + "rule_set_version": "gatekeeper-rules-v1", + "rules": [ + { + "id": "evidence.invocation", + "description": "source_invocation_id must reference an existing tool invocation", + "enabled": true, + "default_severity": "reject" + }, + { + "id": "evidence.excerpt", + "description": "evidence_excerpt must be supported by recorded evidence text", + "enabled": true, + "default_severity": "reject" + } + ], + "checked_bindings": [ + { + "claim_id": "claim-timeout", + "tool_name": "query_logs", + "source_invocation_id": 22, + "raw_path": "$.logs[0]", + "matched_text": "order api timeout", + "status": "pass" + } + ], + "failed_rules": [], + "warnings": [], + "errors": [] + }, + "executor_structured_output": { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-timeout", + "claim_type": "symptom", + "claim_text": "订单接口出现超时", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "tool_name": "query_logs", + "source_invocation_id": 22, + "raw_path": "$.logs[0]", + "evidence_excerpt": "order api timeout" + } + ] + }, + { + "claim_id": "claim-db-primary", + "claim_type": "root_cause", + "claim_text": "数据库主库故障导致订单超时", + "support_level": "weak", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "tool_name": "query_logs", + "source_invocation_id": 22, + "raw_path": "$.logs[0]", + "evidence_excerpt": "order api timeout" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [ + { + "action_text": "补充查询数据库主库状态和错误日志", + "reason": "当前只有订单接口超时日志" + } + ], + "missing_info": ["数据库主库状态", "数据库错误日志"] + }, + "claim_checks": [ + { + "claim_id": "claim-timeout", + "claim_text": "订单接口出现超时", + "claim_type": "symptom", + "verification": "direct_observation", + "detail": "日志直接记录 order api timeout", + "evidence_refs": [ + { + "source_invocation_id": 22, + "raw_path": "$.logs[0]" + } + ] + }, + { + "claim_id": "claim-db-primary", + "claim_text": "数据库主库故障导致订单超时", + "claim_type": "root_cause", + "verification": "unsupported", + "detail": "日志只能证明订单接口超时,不能证明数据库主库故障", + "evidence_refs": [ + { + "source_invocation_id": 22, + "raw_path": "$.logs[0]" + } + ] + } + ], + "facts_checked": [], + "composer_output": { + "status": "valid", + "answer_summary": "日志显示订单接口超时,但数据库方向证据不足。", + "recommended_actions": [ + { + "action_text": "补充查询数据库主库状态和错误日志", + "reason": "当前只有订单接口超时日志" + } + ], + "user_facing_answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。" + }, + "tool_trace_summary": [ + { + "tool_name": "query_logs", + "success": true, + "evidence_level": "direct" + } + ] + } + } + }, + "steps": [], + "toolInvocations": [ + { + "id": 22, + "sessionId": "eval-audit-metadata-low-confid", + "toolName": "query_logs", + "outputPreview": "order api timeout", + "retrievalDetails": { + "evidence_status": "supported", + "evidence_refs": [ + { + "raw_path": "$.logs[0]", + "text": "order api timeout" + } + ] + }, + "success": true + } + ], + "summary": { + "persistedStepCount": 3, + "returnedStepCount": 3, + "persistedToolCallCount": 1, + "returnedToolCallCount": 1, + "hasVerifierEvaluation": true, + "hasFeedback": false + } +} diff --git a/mvp/eval/fixtures/prompt-gatekeeper-audit-closure-pass.json b/mvp/eval/fixtures/prompt-gatekeeper-audit-closure-pass.json new file mode 100644 index 0000000..13e2bd5 --- /dev/null +++ b/mvp/eval/fixtures/prompt-gatekeeper-audit-closure-pass.json @@ -0,0 +1,155 @@ +{ + "session": { + "sessionId": "eval-prompt-gatekeeper-audit-closure", + "query": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。", + "status": "SUCCESS", + "agentFlow": "CHAT", + "totalDurationMs": 19000, + "toolCallCount": 1, + "answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。", + "selfEvaluation": { + "verifier_evaluation": { + "verdict": "PASS", + "groundedness_score": 1.0, + "critical_fact_count": 1, + "prompt_audit": { + "version": "chat-prompts-v1", + "prompts": [ + { + "name": "chat_planner", + "version": "chat-planner-v1", + "resource": "prompts/chat-planner-prompt.md" + }, + { + "name": "chat_executor", + "version": "chat-executor-v2", + "resource": "prompts/chat-executor-prompt.md" + }, + { + "name": "chat_verifier", + "version": "chat-verifier-v2", + "resource": "prompts/chat-verifier-prompt.md" + }, + { + "name": "chat_composer", + "version": "chat-composer-v1", + "resource": "prompts/chat-composer-prompt.md" + } + ] + }, + "gatekeeper_result": { + "status": "pass", + "severity": "none", + "rule_set_version": "gatekeeper-rules-v1", + "rules": [ + { + "id": "evidence.invocation", + "description": "source_invocation_id must reference an existing tool invocation", + "enabled": true, + "default_severity": "reject" + }, + { + "id": "evidence.raw_path", + "description": "raw_path must exist in retrieval_details.evidence_refs", + "enabled": true, + "default_severity": "reject" + } + ], + "checked_bindings": [ + { + "claim_id": "claim-1", + "tool_name": "query_metrics", + "source_invocation_id": 21, + "raw_path": "$.alerts[0]", + "matched_text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m", + "status": "pass" + } + ], + "failed_rules": [], + "warnings": [], + "errors": [] + }, + "executor_structured_output": { + "answer_version": "executor_evidence_v2", + "claims": [ + { + "claim_id": "claim-1", + "claim_type": "observation", + "claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。", + "support_level": "direct", + "evidence_bindings": [ + { + "source_type": "tool_trace", + "tool_name": "query_metrics", + "source_invocation_id": 21, + "raw_path": "$.alerts[0]", + "evidence_excerpt": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m" + } + ] + } + ], + "hypotheses": [], + "recommended_actions": [], + "missing_info": [] + }, + "claim_checks": [ + { + "claim_id": "claim-1", + "claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。", + "claim_type": "observation", + "verification": "direct_observation", + "detail": "已核验的指标证据直接包含服务名、告警名和 CPU 当前值。", + "evidence_refs": [ + { + "source_invocation_id": 21, + "raw_path": "$.alerts[0]" + } + ] + } + ], + "facts_checked": [], + "composer_output": { + "status": "valid", + "answer_summary": "payment-service 当前存在 HighCPUUsage 告警。", + "recommended_actions": [], + "user_facing_answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。" + }, + "tool_trace_summary": [ + { + "tool_name": "query_metrics", + "success": true, + "source_invocation_ids": [21], + "evidence_level": "direct" + } + ] + } + } + }, + "steps": [], + "toolInvocations": [ + { + "id": 21, + "sessionId": "eval-prompt-gatekeeper-audit-closure", + "toolName": "query_metrics", + "outputPreview": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m", + "retrievalDetails": { + "evidence_status": "supported", + "evidence_refs": [ + { + "raw_path": "$.alerts[0]", + "text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m" + } + ] + }, + "success": true + } + ], + "summary": { + "persistedStepCount": 3, + "returnedStepCount": 3, + "persistedToolCallCount": 1, + "returnedToolCallCount": 1, + "hasVerifierEvaluation": true, + "hasFeedback": false + } +} diff --git a/mvp/eval/reports/baseline-report.json b/mvp/eval/reports/baseline-report.json index 43a7dbe..356861f 100644 --- a/mvp/eval/reports/baseline-report.json +++ b/mvp/eval/reports/baseline-report.json @@ -1,14 +1,14 @@ { - "totalCases" : 10, - "passedCases" : 10, + "totalCases" : 12, + "passedCases" : 12, "passRate" : 1.0, "verdictDistribution" : { - "PASS" : 4, - "LOW_CONFID" : 5, + "PASS" : 5, + "LOW_CONFID" : 6, "REJECT" : 1 }, - "averageToolCallCount" : 1.5, - "averageDurationMs" : 39800.0, + "averageToolCallCount" : 1.4166666666666667, + "averageDurationMs" : 38500.0, "results" : [ { "caseId" : "narrow-highcpu-observation", "title" : "Narrow HighCPU observation", @@ -22,10 +22,31 @@ }, "gatekeeperStatus" : "pass", "gatekeeperRuleSetVersion" : "gatekeeper-rules-v1", + "promptAuditVersion" : null, "composerStatus" : "valid", "claimCheckCount" : 1, + "gatekeeperRuleCount" : 1, "toolCallCount" : 1, "durationMs" : 18000 + }, { + "caseId" : "prompt-gatekeeper-audit-closure", + "title" : "Prompt and Gatekeeper audit closure", + "passed" : true, + "failedChecks" : [ ], + "verdict" : "PASS", + "matchedKeywordCount" : 3, + "requiredKeywordCount" : 3, + "evidenceCoverage" : { + "query_metrics" : true + }, + "gatekeeperStatus" : "pass", + "gatekeeperRuleSetVersion" : "gatekeeper-rules-v1", + "promptAuditVersion" : "chat-prompts-v1", + "composerStatus" : "valid", + "claimCheckCount" : 1, + "gatekeeperRuleCount" : 2, + "toolCallCount" : 1, + "durationMs" : 19000 }, { "caseId" : "hikari-no-evidence-negative-observation", "title" : "Hikari no-evidence negative observation", @@ -39,8 +60,10 @@ }, "gatekeeperStatus" : "pass", "gatekeeperRuleSetVersion" : "gatekeeper-rules-v1", + "promptAuditVersion" : null, "composerStatus" : "valid", "claimCheckCount" : 1, + "gatekeeperRuleCount" : 1, "toolCallCount" : 1, "durationMs" : 21000 }, { @@ -58,8 +81,10 @@ }, "gatekeeperStatus" : null, "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : null, "claimCheckCount" : null, + "gatekeeperRuleCount" : null, "toolCallCount" : 3, "durationMs" : 42000 }, { @@ -76,8 +101,10 @@ }, "gatekeeperStatus" : null, "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : null, "claimCheckCount" : null, + "gatekeeperRuleCount" : null, "toolCallCount" : 2, "durationMs" : 51000 }, { @@ -93,8 +120,10 @@ }, "gatekeeperStatus" : null, "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : null, "claimCheckCount" : null, + "gatekeeperRuleCount" : null, "toolCallCount" : 1, "durationMs" : 36000 }, { @@ -111,8 +140,10 @@ }, "gatekeeperStatus" : null, "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : null, "claimCheckCount" : null, + "gatekeeperRuleCount" : null, "toolCallCount" : 2, "durationMs" : 47000 }, { @@ -129,8 +160,10 @@ }, "gatekeeperStatus" : null, "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : null, "claimCheckCount" : null, + "gatekeeperRuleCount" : null, "toolCallCount" : 2, "durationMs" : 53000 }, { @@ -146,8 +179,10 @@ }, "gatekeeperStatus" : "fail", "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : "valid", "claimCheckCount" : 1, + "gatekeeperRuleCount" : null, "toolCallCount" : 1, "durationMs" : 39000 }, { @@ -163,10 +198,31 @@ }, "gatekeeperStatus" : "pass", "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : "valid", "claimCheckCount" : 2, + "gatekeeperRuleCount" : null, "toolCallCount" : 1, "durationMs" : 44000 + }, { + "caseId" : "audit-metadata-low-confid", + "title" : "Audit metadata low confidence", + "passed" : true, + "failedChecks" : [ ], + "verdict" : "LOW_CONFID", + "matchedKeywordCount" : 2, + "requiredKeywordCount" : 2, + "evidenceCoverage" : { + "query_logs" : true + }, + "gatekeeperStatus" : "pass", + "gatekeeperRuleSetVersion" : "gatekeeper-rules-v1", + "promptAuditVersion" : "chat-prompts-v1", + "composerStatus" : "valid", + "claimCheckCount" : 2, + "gatekeeperRuleCount" : 2, + "toolCallCount" : 1, + "durationMs" : 45000 }, { "caseId" : "composer-fallback-no-raw-json", "title" : "Composer fallback no raw JSON", @@ -180,8 +236,10 @@ }, "gatekeeperStatus" : "pass", "gatekeeperRuleSetVersion" : null, + "promptAuditVersion" : null, "composerStatus" : "composer_malformed", "claimCheckCount" : 2, + "gatekeeperRuleCount" : null, "toolCallCount" : 1, "durationMs" : 47000 } ] diff --git a/mvp/eval/reports/baseline-report.md b/mvp/eval/reports/baseline-report.md index 5be56a9..49ac6e2 100644 --- a/mvp/eval/reports/baseline-report.md +++ b/mvp/eval/reports/baseline-report.md @@ -1,28 +1,30 @@ # Diagnosis Eval Report -- Total cases: 10 -- Passed cases: 10 +- Total cases: 12 +- Passed cases: 12 - Pass rate: 100.00% -- Average tool calls: 1.50 -- Average duration ms: 39800.00 +- Average tool calls: 1.42 +- Average duration ms: 38500.00 ## Verdict Distribution -- PASS: 4 -- LOW_CONFID: 5 +- PASS: 5 +- LOW_CONFID: 6 - REJECT: 1 ## Cases -| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks | -| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- | -| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 18000 | - | -| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 21000 | - | -| payment-timeout | PASS | PASS | - | - | - | - | 3/3 | 3 | 42000 | - | -| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 51000 | - | -| redis-timeout | PASS | LOW_CONFID | - | - | - | - | 2/2 | 1 | 36000 | - | -| slow-response | PASS | PASS | - | - | - | - | 2/2 | 2 | 47000 | - | -| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 53000 | - | -| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | valid | 1 | 3/3 | 1 | 39000 | - | -| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | valid | 2 | 2/2 | 1 | 44000 | - | -| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | composer_malformed | 2 | 2/2 | 1 | 47000 | - | +| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks | +| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- | +| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 18000 | - | +| prompt-gatekeeper-audit-closure | PASS | PASS | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 1 | 2 | 3/3 | 1 | 19000 | - | +| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 21000 | - | +| payment-timeout | PASS | PASS | - | - | - | - | - | - | 3/3 | 3 | 42000 | - | +| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 51000 | - | +| redis-timeout | PASS | LOW_CONFID | - | - | - | - | - | - | 2/2 | 1 | 36000 | - | +| slow-response | PASS | PASS | - | - | - | - | - | - | 2/2 | 2 | 47000 | - | +| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 53000 | - | +| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | - | valid | 1 | - | 3/3 | 1 | 39000 | - | +| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | - | valid | 2 | - | 2/2 | 1 | 44000 | - | +| audit-metadata-low-confid | PASS | LOW_CONFID | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 2 | 2 | 2/2 | 1 | 45000 | - | +| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | - | composer_malformed | 2 | - | 2/2 | 1 | 47000 | - | diff --git a/mvp/eval/schema.md b/mvp/eval/schema.md index 36139fe..2f028e8 100644 --- a/mvp/eval/schema.md +++ b/mvp/eval/schema.md @@ -34,7 +34,16 @@ baseline report:整套固定集当前认可的结果 "expectedGatekeeperStatuses": ["pass"], "expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1", "expectedComposerStatuses": ["valid"], - "forbiddenConfirmedClaimKeywords": ["主库故障"] + "forbiddenConfirmedClaimKeywords": ["主库故障"], + "requirePromptAudit": true, + "expectedPromptAuditVersion": "chat-prompts-v1", + "expectedPromptVersions": { + "chat_planner": "chat-planner-v1", + "chat_executor": "chat-executor-v2", + "chat_verifier": "chat-verifier-v2", + "chat_composer": "chat-composer-v1" + }, + "requireGatekeeperRules": true } ``` @@ -58,6 +67,10 @@ baseline report:整套固定集当前认可的结果 | `expectedGatekeeperRuleSetVersion` | 期望的 Gatekeeper 规则集版本 | 配置后校验 `gatekeeper_result.rule_set_version` | | `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 | | `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 | +| `requirePromptAudit` | 是否要求 Prompt 审计元数据 | 要求 `prompt_audit.version` 存在 | +| `expectedPromptAuditVersion` | 期望的 Prompt 审计目录版本 | 配置后校验 `prompt_audit.version` | +| `expectedPromptVersions` | 期望的各角色 Prompt 版本 | 校验 `prompt_audit.prompts[*].name/version` | +| `requireGatekeeperRules` | 是否要求 Gatekeeper 规则元数据 | 要求 `gatekeeper_result.rules` 非空,且每条规则有 `id`、`enabled`、`default_severity` | ## 2. Trace Fixture @@ -72,6 +85,9 @@ fixture 是一次 Agent 运行后的 trace 快照。评测器只读取当前规 | `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` | | `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` | | `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | Gatekeeper 规则集版本 | 新矩阵 case 可显式断言该版本 | +| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rules` | Gatekeeper 规则元数据摘要 | 审计 case 可要求规则列表非空且字段完整 | +| `session.selfEvaluation.verifier_evaluation.prompt_audit.version` | Chat Prompt 审计目录版本 | 审计 case 可显式断言该版本 | +| `session.selfEvaluation.verifier_evaluation.prompt_audit.prompts` | 各 Chat Prompt 名称、版本和资源路径 | 审计 case 可断言 planner、executor、verifier、composer 版本 | | `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` | | `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 | | `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 | @@ -95,8 +111,10 @@ Java 类型:`DiagnosisEvalResult` | `evidenceCoverage` | 每个必需工具是否出现 | | `gatekeeperStatus` | 读到的 `gatekeeper_result.status` | | `gatekeeperRuleSetVersion` | 读到的 `gatekeeper_result.rule_set_version` | +| `promptAuditVersion` | 读到的 `prompt_audit.version` | | `composerStatus` | 读到的 `composer_output.status` | | `claimCheckCount` | `claim_checks` 数量 | +| `gatekeeperRuleCount` | `gatekeeper_result.rules` 数量 | | `toolCallCount` | trace 中工具调用总数 | | `durationMs` | trace 总耗时 | @@ -130,6 +148,10 @@ Executor structured output - V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。 - `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。 - 配置 `expectedGatekeeperRuleSetVersion` 的 case 必须匹配 `gatekeeper_result.rule_set_version`。 +- 配置 `requirePromptAudit` 的 case 必须包含 `prompt_audit.version`。 +- 配置 `expectedPromptAuditVersion` 的 case 必须匹配 `prompt_audit.version`。 +- 配置 `expectedPromptVersions` 的 case 必须能在 `prompt_audit.prompts` 中找到对应角色和版本。 +- 配置 `requireGatekeeperRules` 的 case 必须包含非空 `gatekeeper_result.rules`,且每条规则有 `id`、`enabled`、`default_severity`。 - `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。 - Composer 输出必须记录 `status`。 - 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。 diff --git a/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/.archive-ready b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/.archive-ready new file mode 100644 index 0000000..395527d --- /dev/null +++ b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/.archive-ready @@ -0,0 +1 @@ +ready diff --git a/openspec/changes/interview-demo-quality-audit/.committed b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/.committed similarity index 100% rename from openspec/changes/interview-demo-quality-audit/.committed rename to openspec/changes/archive/2026-07-09-interview-demo-quality-audit/.committed diff --git a/openspec/changes/interview-demo-quality-audit/design.md b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/design.md similarity index 100% rename from openspec/changes/interview-demo-quality-audit/design.md rename to openspec/changes/archive/2026-07-09-interview-demo-quality-audit/design.md diff --git a/openspec/changes/interview-demo-quality-audit/proposal.md b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/proposal.md similarity index 100% rename from openspec/changes/interview-demo-quality-audit/proposal.md rename to openspec/changes/archive/2026-07-09-interview-demo-quality-audit/proposal.md diff --git a/openspec/changes/interview-demo-quality-audit/specs/chat-verifier-agent/spec.md b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/specs/chat-verifier-agent/spec.md similarity index 100% rename from openspec/changes/interview-demo-quality-audit/specs/chat-verifier-agent/spec.md rename to openspec/changes/archive/2026-07-09-interview-demo-quality-audit/specs/chat-verifier-agent/spec.md diff --git a/openspec/changes/interview-demo-quality-audit/specs/diagnosis-eval-harness/spec.md b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/specs/diagnosis-eval-harness/spec.md similarity index 100% rename from openspec/changes/interview-demo-quality-audit/specs/diagnosis-eval-harness/spec.md rename to openspec/changes/archive/2026-07-09-interview-demo-quality-audit/specs/diagnosis-eval-harness/spec.md diff --git a/openspec/changes/interview-demo-quality-audit/specs/mvp-demo-trace-acceptance/spec.md b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/specs/mvp-demo-trace-acceptance/spec.md similarity index 100% rename from openspec/changes/interview-demo-quality-audit/specs/mvp-demo-trace-acceptance/spec.md rename to openspec/changes/archive/2026-07-09-interview-demo-quality-audit/specs/mvp-demo-trace-acceptance/spec.md diff --git a/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/tasks.md b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/tasks.md new file mode 100644 index 0000000..bd8d13b --- /dev/null +++ b/openspec/changes/archive/2026-07-09-interview-demo-quality-audit/tasks.md @@ -0,0 +1,32 @@ +## 1. OpenSpec And devflow + +- [x] 1.1 Create OpenSpec proposal/design/spec/tasks for `interview-demo-quality-audit`. +- [x] 1.2 Record context, question pool, interface impact, audit, and verification plan in devflow decisions. +- [x] 1.3 Pass OpenSpec validation and create `.committed`. + +## 2. Prompt/Gatekeeper Version Audit + +- [x] 2.1 Add compact Chat prompt audit metadata for planner, executor, verifier, and composer prompts. +- [x] 2.2 Persist `prompt_audit` under `verifier_evaluation` for Chat verifier/composer outcomes. +- [x] 2.3 Add focused tests proving prompt audit appears in persisted verifier evaluation. +- [x] 2.4 Extend eval checks for prompt audit and Gatekeeper rule metadata. + +## 3. Eval Expansion + +- [x] 3.1 Extend diagnosis eval case/result schema for prompt audit fields. +- [x] 3.2 Add fixture-backed cases for audit closure coverage. +- [x] 3.3 Regenerate baseline JSON and Markdown reports. +- [x] 3.4 Update eval docs/schema. + +## 4. Interview Demo Stabilization + +- [x] 4.1 Add `run-interview-demo-check.ps1` with service preflight, chat, trace, feedback, and summary output. +- [x] 4.2 Update demo README and 10-minute script to use the preflight path. +- [x] 4.3 Add interview Q&A documentation focused on Agent engineering tradeoffs. + +## 5. Verification And Archive + +- [x] 5.1 Run targeted tests for ChatService/prompt audit and diagnosis eval. +- [x] 5.2 Run relevant broader regression tests. +- [x] 5.3 Run E2E demo check with `mvp-demo` profile if dependencies are available; otherwise record the blocker. +- [x] 5.4 Archive the OpenSpec change, update devflow artifacts, and commit implementation + archive. diff --git a/openspec/changes/interview-demo-quality-audit/tasks.md b/openspec/changes/interview-demo-quality-audit/tasks.md deleted file mode 100644 index 3e32d1c..0000000 --- a/openspec/changes/interview-demo-quality-audit/tasks.md +++ /dev/null @@ -1,32 +0,0 @@ -## 1. OpenSpec And devflow - -- [x] 1.1 Create OpenSpec proposal/design/spec/tasks for `interview-demo-quality-audit`. -- [x] 1.2 Record context, question pool, interface impact, audit, and verification plan in devflow decisions. -- [x] 1.3 Pass OpenSpec validation and create `.committed`. - -## 2. Prompt/Gatekeeper Version Audit - -- [ ] 2.1 Add compact Chat prompt audit metadata for planner, executor, verifier, and composer prompts. -- [ ] 2.2 Persist `prompt_audit` under `verifier_evaluation` for Chat verifier/composer outcomes. -- [ ] 2.3 Add focused tests proving prompt audit appears in persisted verifier evaluation. -- [ ] 2.4 Extend eval checks for prompt audit and Gatekeeper rule metadata. - -## 3. Eval Expansion - -- [ ] 3.1 Extend diagnosis eval case/result schema for prompt audit fields. -- [ ] 3.2 Add fixture-backed cases for audit closure coverage. -- [ ] 3.3 Regenerate baseline JSON and Markdown reports. -- [ ] 3.4 Update eval docs/schema. - -## 4. Interview Demo Stabilization - -- [ ] 4.1 Add `run-interview-demo-check.ps1` with service preflight, chat, trace, feedback, and summary output. -- [ ] 4.2 Update demo README and 10-minute script to use the preflight path. -- [ ] 4.3 Add interview Q&A documentation focused on Agent engineering tradeoffs. - -## 5. Verification And Archive - -- [ ] 5.1 Run targeted tests for ChatService/prompt audit and diagnosis eval. -- [ ] 5.2 Run relevant broader regression tests. -- [ ] 5.3 Run E2E demo check with `mvp-demo` profile if dependencies are available; otherwise record the blocker. -- [ ] 5.4 Archive the OpenSpec change, update devflow artifacts, and commit implementation + archive. diff --git a/openspec/specs/chat-verifier-agent/spec.md b/openspec/specs/chat-verifier-agent/spec.md index 24df96a..af656be 100644 --- a/openspec/specs/chat-verifier-agent/spec.md +++ b/openspec/specs/chat-verifier-agent/spec.md @@ -145,6 +145,17 @@ The Verifier's verdict and downstream final-answer composition SHALL be persiste - **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `gatekeeper_result` - **AND** existing verifier fields such as `verdict`, `facts_checked`, `executor_output_parse_status`, and `tool_trace_summary` SHALL be preserved +#### Scenario: prompt audit written to verifier evaluation +- **WHEN** the Chat verifier evaluation is persisted +- **THEN** the system SHALL include a `prompt_audit` object under `diagnosis_session.self_evaluation.verifier_evaluation` +- **AND** `prompt_audit.version` SHALL identify the Chat prompt audit catalog version +- **AND** `prompt_audit.prompts` SHALL include the planner, executor, verifier, and composer prompt names and versions +- **AND** full prompt text SHALL NOT be persisted in `prompt_audit` + +#### Scenario: prompt audit available on fallback paths +- **WHEN** Chat verifier parsing fails, Composer parsing fails, or Chat produces a degraded answer +- **THEN** the persisted verifier evaluation SHALL still include `prompt_audit` + ### Requirement: self_evaluation SHALL be a container object The `diagnosis_session.self_evaluation` field SHALL store multiple evaluation channels in one JSON object. diff --git a/openspec/specs/diagnosis-eval-harness/spec.md b/openspec/specs/diagnosis-eval-harness/spec.md index a683d46..6cbf4b6 100644 --- a/openspec/specs/diagnosis-eval-harness/spec.md +++ b/openspec/specs/diagnosis-eval-harness/spec.md @@ -195,3 +195,22 @@ The evaluation harness SHALL be able to assert the Gatekeeper rule set version r - **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion` - **AND** the fixture has a different or missing rule set version - **THEN** the case SHALL fail with a clear failed check + +### Requirement: Diagnosis eval SHALL validate trace fixtures deterministically +The diagnosis eval harness SHALL evaluate saved trace fixtures without invoking an LLM judge. + +#### Scenario: prompt audit assertions are enforced +- **WHEN** an eval case sets `requirePromptAudit=true` +- **THEN** the evaluator SHALL require `verifier_evaluation.prompt_audit.version` +- **AND** when `expectedPromptAuditVersion` is configured, it SHALL match exactly +- **AND** when `expectedPromptVersions` is configured, each configured prompt name SHALL appear with the expected version + +#### Scenario: Gatekeeper rule metadata assertions are enforced +- **WHEN** an eval case sets `requireGatekeeperRules=true` +- **THEN** the evaluator SHALL require `verifier_evaluation.gatekeeper_result.rules` to be non-empty +- **AND** each rule item SHALL include `id`, `enabled`, and `default_severity` + +#### Scenario: expanded baseline remains passing +- **WHEN** the committed fixture set is evaluated +- **THEN** every case SHALL pass +- **AND** baseline JSON and Markdown reports SHALL reflect the expanded case count and verdict distribution diff --git a/openspec/specs/mvp-demo-trace-acceptance/spec.md b/openspec/specs/mvp-demo-trace-acceptance/spec.md index 0eb3199..3d93d30 100644 --- a/openspec/specs/mvp-demo-trace-acceptance/spec.md +++ b/openspec/specs/mvp-demo-trace-acceptance/spec.md @@ -56,6 +56,26 @@ The MVP demo SHALL provide scripts and request payloads for running the payment- - **WHEN** the demo script finishes successfully - **THEN** it SHALL write chat, trace, and feedback responses under a demo output directory +### Requirement: MVP demo SHALL be reproducible for interviews +The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, verifier evaluation, and feedback. + +#### Scenario: interview demo check script records an evidence bundle +- **WHEN** the user runs the interview demo check script against a running `mvp-demo` service +- **THEN** the script SHALL submit a fixed Chat diagnosis request +- **AND** it SHALL fetch the trace for the same session id +- **AND** it SHALL submit useful feedback for that session +- **AND** it SHALL write chat, trace, feedback, and summary outputs under `mvp/demo/output/` + +#### Scenario: interview demo check fails with actionable readiness output +- **WHEN** the target service is not reachable +- **THEN** the script SHALL fail before issuing diagnosis requests +- **AND** the failure message SHALL name the base URL and the expected startup profile + +#### Scenario: interview documentation explains audit fields +- **WHEN** an interviewer asks how prompt or Gatekeeper changes are audited +- **THEN** the demo documentation SHALL point to `prompt_audit.version` and `gatekeeper_result.rule_set_version` +- **AND** it SHALL explain that deterministic eval fixtures are the regression source of truth + ### Requirement: MVP demo SHALL provide a trace inspection checklist The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and session-level auditability. diff --git a/src/main/java/com/superbiz/agent/eval/DiagnosisEvalCase.java b/src/main/java/com/superbiz/agent/eval/DiagnosisEvalCase.java index 63bee36..949511a 100644 --- a/src/main/java/com/superbiz/agent/eval/DiagnosisEvalCase.java +++ b/src/main/java/com/superbiz/agent/eval/DiagnosisEvalCase.java @@ -6,6 +6,7 @@ import lombok.Data; import lombok.NoArgsConstructor; import java.util.List; +import java.util.Map; @Data @Builder @@ -29,4 +30,8 @@ public class DiagnosisEvalCase { private String expectedGatekeeperRuleSetVersion; private List expectedComposerStatuses; private List forbiddenConfirmedClaimKeywords; + private Boolean requirePromptAudit; + private String expectedPromptAuditVersion; + private Map expectedPromptVersions; + private Boolean requireGatekeeperRules; } diff --git a/src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java b/src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java index 7fc820e..43430d4 100644 --- a/src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java +++ b/src/main/java/com/superbiz/agent/eval/DiagnosisEvalReportWriter.java @@ -46,8 +46,8 @@ public class DiagnosisEvalReportWriter { } builder.append("## Cases\n\n"); - builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |\n"); - builder.append("| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |\n"); + builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks |\n"); + builder.append("| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- |\n"); for (DiagnosisEvalResult result : report.getResults()) { builder.append("| ") .append(result.getCaseId()) @@ -60,10 +60,14 @@ public class DiagnosisEvalReportWriter { .append(" | ") .append(valueOrDash(result.getGatekeeperRuleSetVersion())) .append(" | ") + .append(valueOrDash(result.getPromptAuditVersion())) + .append(" | ") .append(valueOrDash(result.getComposerStatus())) .append(" | ") .append(result.getClaimCheckCount() == null ? "-" : result.getClaimCheckCount()) .append(" | ") + .append(result.getGatekeeperRuleCount() == null ? "-" : result.getGatekeeperRuleCount()) + .append(" | ") .append(result.getMatchedKeywordCount()).append("/").append(result.getRequiredKeywordCount()) .append(" | ") .append(result.getToolCallCount() == null ? "-" : result.getToolCallCount()) diff --git a/src/main/java/com/superbiz/agent/eval/DiagnosisEvalResult.java b/src/main/java/com/superbiz/agent/eval/DiagnosisEvalResult.java index 4053130..75165d1 100644 --- a/src/main/java/com/superbiz/agent/eval/DiagnosisEvalResult.java +++ b/src/main/java/com/superbiz/agent/eval/DiagnosisEvalResult.java @@ -24,8 +24,10 @@ public class DiagnosisEvalResult { private Map evidenceCoverage; private String gatekeeperStatus; private String gatekeeperRuleSetVersion; + private String promptAuditVersion; private String composerStatus; private Integer claimCheckCount; + private Integer gatekeeperRuleCount; private Integer toolCallCount; private Integer durationMs; } diff --git a/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java b/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java index 7648b14..c4f4c7e 100644 --- a/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java +++ b/src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java @@ -67,8 +67,10 @@ public class DiagnosisTraceEvaluator { .evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools())) .gatekeeperStatus(null) .gatekeeperRuleSetVersion(null) + .promptAuditVersion(null) .composerStatus(null) .claimCheckCount(null) + .gatekeeperRuleCount(null) .toolCallCount(null) .durationMs(null) .build()); @@ -123,10 +125,13 @@ public class DiagnosisTraceEvaluator { String gatekeeperStatus = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "status"); String gatekeeperRuleSetVersion = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "rule_set_version"); + String promptAuditVersion = extractNestedString(trace, "verifier_evaluation", + "prompt_audit", "version"); String composerStatus = extractNestedString(trace, "verifier_evaluation", "composer_output", "status"); Integer claimCheckCount = countList(trace, "verifier_evaluation", "claim_checks"); + Integer gatekeeperRuleCount = countNestedList(trace, "verifier_evaluation", "gatekeeper_result", "rules"); failedChecks.addAll(validateV2AuditClosure(evalCase, trace, normalizedAnswer, verdict, - gatekeeperStatus, gatekeeperRuleSetVersion, composerStatus)); + gatekeeperStatus, gatekeeperRuleSetVersion, promptAuditVersion, composerStatus)); Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size(); Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs(); @@ -142,8 +147,10 @@ public class DiagnosisTraceEvaluator { .evidenceCoverage(evidenceCoverage) .gatekeeperStatus(gatekeeperStatus) .gatekeeperRuleSetVersion(gatekeeperRuleSetVersion) + .promptAuditVersion(promptAuditVersion) .composerStatus(composerStatus) .claimCheckCount(claimCheckCount) + .gatekeeperRuleCount(gatekeeperRuleCount) .toolCallCount(toolCallCount) .durationMs(durationMs) .build(); @@ -237,6 +244,7 @@ public class DiagnosisTraceEvaluator { String verdict, String gatekeeperStatus, String gatekeeperRuleSetVersion, + String promptAuditVersion, String composerStatus) { List failedChecks = new ArrayList<>(); boolean requireV2AuditClosure = Boolean.TRUE.equals(evalCase.getRequireV2AuditClosure()); @@ -259,6 +267,10 @@ public class DiagnosisTraceEvaluator { failedChecks.add("gatekeeper rule set version not expected: " + valueOrMissing(gatekeeperRuleSetVersion)); } + if (Boolean.TRUE.equals(evalCase.getRequireGatekeeperRules())) { + failedChecks.addAll(validateGatekeeperRules(trace)); + } + failedChecks.addAll(validatePromptAudit(evalCase, trace, promptAuditVersion)); failedChecks.addAll(validateClaimChecks(trace, requireClaimChecks)); @@ -290,6 +302,78 @@ public class DiagnosisTraceEvaluator { return failedChecks; } + private List validatePromptAudit(DiagnosisEvalCase evalCase, + DiagnosisTraceResponse trace, + String promptAuditVersion) { + List failedChecks = new ArrayList<>(); + Object promptAudit = nestedValue(trace, "verifier_evaluation", "prompt_audit"); + if (Boolean.TRUE.equals(evalCase.getRequirePromptAudit()) && !(promptAudit instanceof Map)) { + failedChecks.add("missing prompt_audit"); + return failedChecks; + } + if (!isBlank(evalCase.getExpectedPromptAuditVersion()) + && !evalCase.getExpectedPromptAuditVersion().equals(promptAuditVersion)) { + failedChecks.add("prompt audit version not expected: " + valueOrMissing(promptAuditVersion)); + } + if (evalCase.getExpectedPromptVersions() == null || evalCase.getExpectedPromptVersions().isEmpty()) { + return failedChecks; + } + if (!(promptAudit instanceof Map audit)) { + failedChecks.add("missing prompt_audit"); + return failedChecks; + } + Object promptsValue = audit.get("prompts"); + if (!(promptsValue instanceof List prompts)) { + failedChecks.add("prompt_audit missing prompts"); + return failedChecks; + } + Map actualVersions = new LinkedHashMap<>(); + for (Object promptValue : prompts) { + if (promptValue instanceof Map prompt) { + String name = stringValue(prompt.get("name")); + String version = stringValue(prompt.get("version")); + if (!isBlank(name)) { + actualVersions.put(name, version); + } + } + } + for (Map.Entry expected : evalCase.getExpectedPromptVersions().entrySet()) { + String actual = actualVersions.get(expected.getKey()); + if (!expected.getValue().equals(actual)) { + failedChecks.add("prompt version not expected: " + + expected.getKey() + "=" + valueOrMissing(actual)); + } + } + return failedChecks; + } + + private List validateGatekeeperRules(DiagnosisTraceResponse trace) { + Object rulesValue = nestedNestedValue(trace, "verifier_evaluation", "gatekeeper_result", "rules"); + if (!(rulesValue instanceof List rules) || rules.isEmpty()) { + return List.of("gatekeeper_result missing rules"); + } + List failedChecks = new ArrayList<>(); + for (Object item : rules) { + if (!(item instanceof Map rule)) { + failedChecks.add("gatekeeper rule metadata is not an object"); + continue; + } + String id = stringValue(rule.get("id")); + Object enabled = rule.get("enabled"); + String severity = stringValue(rule.get("default_severity")); + if (isBlank(id)) { + failedChecks.add("gatekeeper rule metadata missing id"); + } + if (!(enabled instanceof Boolean)) { + failedChecks.add("gatekeeper rule metadata missing enabled: " + valueOrMissing(id)); + } + if (isBlank(severity)) { + failedChecks.add("gatekeeper rule metadata missing default_severity: " + valueOrMissing(id)); + } + } + return failedChecks; + } + private List validateClaimChecks(DiagnosisTraceResponse trace, boolean required) { Object claimChecks = nestedValue(trace, "verifier_evaluation", "claim_checks"); if (!(claimChecks instanceof List claimCheckList)) { @@ -347,6 +431,19 @@ public class DiagnosisTraceEvaluator { return value instanceof List list ? list.size() : null; } + private Integer countNestedList(DiagnosisTraceResponse trace, String firstKey, String secondKey, String thirdKey) { + Object value = nestedNestedValue(trace, firstKey, secondKey, thirdKey); + return value instanceof List list ? list.size() : null; + } + + private Object nestedNestedValue(DiagnosisTraceResponse trace, String firstKey, String secondKey, String thirdKey) { + Object value = nestedValue(trace, firstKey, secondKey); + if (!(value instanceof Map map)) { + return null; + } + return map.get(thirdKey); + } + private int countMatches(String normalizedAnswer, List keywords) { int count = 0; for (String keyword : safeList(keywords)) { diff --git a/src/main/java/com/superbiz/agent/service/ChatService.java b/src/main/java/com/superbiz/agent/service/ChatService.java index 162d436..f07645f 100644 --- a/src/main/java/com/superbiz/agent/service/ChatService.java +++ b/src/main/java/com/superbiz/agent/service/ChatService.java @@ -60,6 +60,7 @@ public class ChatService { private static final Logger logger = LoggerFactory.getLogger(ChatService.class); private static final String LOW_CONFID_DISCLAIMER = "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。"; private static final String DEGRADED_PREFIX = "当前无法基于已获取证据生成可靠结论,建议人工介入。"; + private static final String CHAT_PROMPT_AUDIT_VERSION = "chat-prompts-v1"; /** 封装 answer + 后端生成的 sessionId,用于 feedback 关联 */ public record ChatResult(String answer, String sessionId) {} @@ -884,6 +885,7 @@ public class ChatService { verifierEvaluation.put("rationale", decision.rationale()); verifierEvaluation.put("round", round); verifierEvaluation.put("traceability_version", "v1"); + verifierEvaluation.put("prompt_audit", promptAuditSnapshot()); verifierEvaluation.put("executor_output_parse_status", Optional.ofNullable(VerifierContextHolder.getExecutorOutputParseStatus()) .orElse(Map.of("status", "missing", "detail", "executor parse status unavailable"))); @@ -902,6 +904,26 @@ public class ChatService { diagnosisSessionRepository.save(session); } + private Map promptAuditSnapshot() { + Map audit = new LinkedHashMap<>(); + audit.put("version", CHAT_PROMPT_AUDIT_VERSION); + audit.put("prompts", List.of( + promptAuditItem("chat_planner", "chat-planner-v1", "prompts/chat-planner-prompt.md"), + promptAuditItem("chat_executor", "chat-executor-v2", "prompts/chat-executor-prompt.md"), + promptAuditItem("chat_verifier", "chat-verifier-v2", "prompts/chat-verifier-prompt.md"), + promptAuditItem("chat_composer", "chat-composer-v1", "prompts/chat-composer-prompt.md") + )); + return audit; + } + + private Map promptAuditItem(String name, String version, String resource) { + Map item = new LinkedHashMap<>(); + item.put("name", name); + item.put("version", version); + item.put("resource", resource); + return item; + } + private Map defaultGatekeeperPass() { GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback(); return Map.of( diff --git a/src/test/java/com/superbiz/agent/eval/DiagnosisEvalBaselineDiffTest.java b/src/test/java/com/superbiz/agent/eval/DiagnosisEvalBaselineDiffTest.java index 7d4b07e..55d1c9f 100644 --- a/src/test/java/com/superbiz/agent/eval/DiagnosisEvalBaselineDiffTest.java +++ b/src/test/java/com/superbiz/agent/eval/DiagnosisEvalBaselineDiffTest.java @@ -100,13 +100,13 @@ class DiagnosisEvalBaselineDiffTest { } private void degradeRedisCase(DiagnosisEvalReport report) { - report.setPassedCases(9); - report.setPassRate(0.9); + report.setPassedCases(11); + report.setPassRate(11.0 / 12.0); report.setAverageToolCallCount(3.0); - report.setAverageDurationMs(39800.0); + report.setAverageDurationMs(38500.0); report.setVerdictDistribution(new LinkedHashMap<>()); - report.getVerdictDistribution().put("PASS", 4L); - report.getVerdictDistribution().put("LOW_CONFID", 4L); + report.getVerdictDistribution().put("PASS", 5L); + report.getVerdictDistribution().put("LOW_CONFID", 5L); report.getVerdictDistribution().put("REJECT", 2L); DiagnosisEvalResult redis = result(report, "redis-timeout"); diff --git a/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java b/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java index 03c58ec..2ea6880 100644 --- a/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java +++ b/src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java @@ -24,11 +24,11 @@ class DiagnosisTraceEvaluatorTest { DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures")); - assertEquals(10, report.getTotalCases()); - assertEquals(10, report.getPassedCases()); + assertEquals(12, report.getTotalCases()); + assertEquals(12, report.getPassedCases()); assertEquals(1.0, report.getPassRate(), 0.001); - assertEquals(4L, report.getVerdictDistribution().get("PASS")); - assertEquals(5L, report.getVerdictDistribution().get("LOW_CONFID")); + assertEquals(5L, report.getVerdictDistribution().get("PASS")); + assertEquals(6L, report.getVerdictDistribution().get("LOW_CONFID")); assertEquals(1L, report.getVerdictDistribution().get("REJECT")); DiagnosisEvalResult narrowHighCpu = result(report, "narrow-highcpu-observation"); @@ -36,6 +36,12 @@ class DiagnosisTraceEvaluatorTest { assertEquals("gatekeeper-rules-v1", narrowHighCpu.getGatekeeperRuleSetVersion()); assertEquals("pass", narrowHighCpu.getGatekeeperStatus()); + DiagnosisEvalResult promptGatekeeperAudit = result(report, "prompt-gatekeeper-audit-closure"); + assertTrue(promptGatekeeperAudit.isPassed()); + assertEquals("chat-prompts-v1", promptGatekeeperAudit.getPromptAuditVersion()); + assertEquals("gatekeeper-rules-v1", promptGatekeeperAudit.getGatekeeperRuleSetVersion()); + assertEquals(2, promptGatekeeperAudit.getGatekeeperRuleCount()); + DiagnosisEvalResult hikariNoEvidence = result(report, "hikari-no-evidence-negative-observation"); assertTrue(hikariNoEvidence.isPassed()); assertEquals("gatekeeper-rules-v1", hikariNoEvidence.getGatekeeperRuleSetVersion()); @@ -60,6 +66,12 @@ class DiagnosisTraceEvaluatorTest { DiagnosisEvalResult composerFallback = result(report, "composer-fallback-no-raw-json"); assertTrue(composerFallback.isPassed()); assertEquals("composer_malformed", composerFallback.getComposerStatus()); + + DiagnosisEvalResult auditMetadataLowConfid = result(report, "audit-metadata-low-confid"); + assertTrue(auditMetadataLowConfid.isPassed()); + assertEquals("LOW_CONFID", auditMetadataLowConfid.getVerdict()); + assertEquals("chat-prompts-v1", auditMetadataLowConfid.getPromptAuditVersion()); + assertEquals(2, auditMetadataLowConfid.getGatekeeperRuleCount()); } @Test @@ -195,6 +207,76 @@ class DiagnosisTraceEvaluatorTest { "gatekeeper rule set version not expected: old-rules")); } + @Test + void evaluateFailsWhenPromptAuditMissingOrMismatches() { + DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder() + .id("prompt-audit") + .title("Prompt audit") + .expectedRootCauseKeywords(List.of()) + .requiredEvidenceTools(List.of()) + .allowedVerdicts(List.of("PASS")) + .requirePromptAudit(true) + .expectedPromptAuditVersion("chat-prompts-v1") + .expectedPromptVersions(java.util.Map.of("chat_executor", "chat-executor-v2")) + .build(); + DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder() + .session(DiagnosisTraceResponse.SessionTrace.builder() + .answer("安全回答") + .selfEvaluation(java.util.Map.of( + "verifier_evaluation", java.util.Map.of( + "verdict", "PASS", + "prompt_audit", java.util.Map.of( + "version", "old-prompts", + "prompts", java.util.List.of(java.util.Map.of( + "name", "chat_executor", + "version", "chat-executor-v1" + )) + ) + ))) + .build()) + .toolInvocations(List.of()) + .build(); + + DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace); + + assertFalse(result.isPassed()); + assertTrue(result.getFailedChecks().contains( + "prompt audit version not expected: old-prompts")); + assertTrue(result.getFailedChecks().contains( + "prompt version not expected: chat_executor=chat-executor-v1")); + } + + @Test + void evaluateFailsWhenGatekeeperRulesAreMissing() { + DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder() + .id("gatekeeper-rules") + .title("Gatekeeper rules") + .expectedRootCauseKeywords(List.of()) + .requiredEvidenceTools(List.of()) + .allowedVerdicts(List.of("PASS")) + .requireGatekeeperRules(true) + .build(); + DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder() + .session(DiagnosisTraceResponse.SessionTrace.builder() + .answer("安全回答") + .selfEvaluation(java.util.Map.of( + "verifier_evaluation", java.util.Map.of( + "verdict", "PASS", + "gatekeeper_result", java.util.Map.of( + "status", "pass", + "rule_set_version", "gatekeeper-rules-v1" + ) + ))) + .build()) + .toolInvocations(List.of()) + .build(); + + DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace); + + assertFalse(result.isPassed()); + assertTrue(result.getFailedChecks().contains("gatekeeper_result missing rules")); + } + @Test void evaluateFailsWhenUnsupportedClaimLeaksIntoFinalAnswer() { diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index f02f32b..7d43c76 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -394,10 +394,20 @@ class ChatServiceSequentialAgentTest { verify(mergeService).mergeVerifierEvaluation(isNull(), captor.capture()); Map verifierEvaluation = captor.getValue(); assertTrue(verifierEvaluation.containsKey("gatekeeper_result")); + assertTrue(verifierEvaluation.containsKey("prompt_audit")); @SuppressWarnings("unchecked") Map gatekeeperResult = (Map) verifierEvaluation.get("gatekeeper_result"); assertEquals("pass", gatekeeperResult.get("status")); assertEquals("none", gatekeeperResult.get("severity")); + @SuppressWarnings("unchecked") + Map promptAudit = (Map) verifierEvaluation.get("prompt_audit"); + assertEquals("chat-prompts-v1", promptAudit.get("version")); + @SuppressWarnings("unchecked") + List> prompts = (List>) promptAudit.get("prompts"); + assertEquals(4, prompts.size()); + assertTrue(prompts.stream().anyMatch(prompt -> + "chat_executor".equals(prompt.get("name")) + && "chat-executor-v2".equals(prompt.get("version")))); } @Test