feat(demo): add interview quality audit
This commit is contained in:
+8
-4
@@ -29,10 +29,10 @@ The baseline evaluates saved trace fixtures. It does not start the application a
|
||||
The committed baseline currently contains:
|
||||
|
||||
```text
|
||||
10 fixed cases
|
||||
10 passing fixture evaluations
|
||||
4 PASS verdicts
|
||||
5 LOW_CONFID verdicts
|
||||
12 fixed cases
|
||||
12 passing fixture evaluations
|
||||
5 PASS verdicts
|
||||
6 LOW_CONFID verdicts
|
||||
1 REJECT verdict
|
||||
```
|
||||
|
||||
@@ -44,6 +44,8 @@ The V2 evidence-pipeline matrix covers:
|
||||
- Unsupported claim filtering before the final answer.
|
||||
- Composer fallback rendering without raw Executor JSON leakage.
|
||||
- Gatekeeper rule set version audit for new matrix fixtures.
|
||||
- Prompt audit version checks for planner, executor, verifier, and composer prompts.
|
||||
- Gatekeeper rule metadata checks for enabled rule id and default severity.
|
||||
|
||||
## Verification
|
||||
|
||||
@@ -80,6 +82,8 @@ Stage 5 adds these V2 checks:
|
||||
- `claim_checks` must be structurally auditable.
|
||||
- Composer output must record whether normal parsing or fallback rendering was used.
|
||||
- Gatekeeper rule set version can be asserted per fixture.
|
||||
- Prompt audit version and per-prompt versions can be asserted per fixture.
|
||||
- Gatekeeper rule metadata can be required per fixture.
|
||||
- Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`.
|
||||
- Configured unsupported claim keywords must not appear as confirmed final-answer content.
|
||||
|
||||
|
||||
@@ -17,6 +17,33 @@
|
||||
"expectedComposerStatuses": ["valid"],
|
||||
"forbiddenConfirmedClaimKeywords": ["数据库连接池"]
|
||||
},
|
||||
{
|
||||
"id": "prompt-gatekeeper-audit-closure",
|
||||
"title": "Prompt and Gatekeeper audit closure",
|
||||
"question": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。",
|
||||
"traceFixture": "prompt-gatekeeper-audit-closure-pass.json",
|
||||
"expectedRootCauseKeywords": ["payment-service", "HighCPUUsage", "92%"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["query_metrics"],
|
||||
"allowedVerdicts": ["PASS"],
|
||||
"forbiddenAnswerKeywords": ["根因", "修复建议", "通常情况下"],
|
||||
"requireV2AuditClosure": true,
|
||||
"requireClaimChecks": true,
|
||||
"requireComposerOutput": true,
|
||||
"expectedGatekeeperStatuses": ["pass"],
|
||||
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||
"expectedComposerStatuses": ["valid"],
|
||||
"forbiddenConfirmedClaimKeywords": ["数据库连接池"],
|
||||
"requirePromptAudit": true,
|
||||
"expectedPromptAuditVersion": "chat-prompts-v1",
|
||||
"expectedPromptVersions": {
|
||||
"chat_planner": "chat-planner-v1",
|
||||
"chat_executor": "chat-executor-v2",
|
||||
"chat_verifier": "chat-verifier-v2",
|
||||
"chat_composer": "chat-composer-v1"
|
||||
},
|
||||
"requireGatekeeperRules": true
|
||||
},
|
||||
{
|
||||
"id": "hikari-no-evidence-negative-observation",
|
||||
"title": "Hikari no-evidence negative observation",
|
||||
@@ -124,6 +151,33 @@
|
||||
"expectedComposerStatuses": ["valid"],
|
||||
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
||||
},
|
||||
{
|
||||
"id": "audit-metadata-low-confid",
|
||||
"title": "Audit metadata low confidence",
|
||||
"question": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?",
|
||||
"traceFixture": "audit-metadata-low-confid.json",
|
||||
"expectedRootCauseKeywords": ["超时", "证据"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["query_logs"],
|
||||
"allowedVerdicts": ["LOW_CONFID"],
|
||||
"forbiddenAnswerKeywords": ["已经确认"],
|
||||
"requireV2AuditClosure": true,
|
||||
"requireClaimChecks": true,
|
||||
"requireComposerOutput": true,
|
||||
"expectedGatekeeperStatuses": ["pass"],
|
||||
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||
"expectedComposerStatuses": ["valid"],
|
||||
"forbiddenConfirmedClaimKeywords": ["主库故障"],
|
||||
"requirePromptAudit": true,
|
||||
"expectedPromptAuditVersion": "chat-prompts-v1",
|
||||
"expectedPromptVersions": {
|
||||
"chat_planner": "chat-planner-v1",
|
||||
"chat_executor": "chat-executor-v2",
|
||||
"chat_verifier": "chat-verifier-v2",
|
||||
"chat_composer": "chat-composer-v1"
|
||||
},
|
||||
"requireGatekeeperRules": true
|
||||
},
|
||||
{
|
||||
"id": "composer-fallback-no-raw-json",
|
||||
"title": "Composer fallback no raw JSON",
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
{
|
||||
"session": {
|
||||
"sessionId": "eval-audit-metadata-low-confid",
|
||||
"query": "订单超时是否可以确认由数据库主库故障导致,并检查审计元数据是否完整?",
|
||||
"status": "SUCCESS",
|
||||
"agentFlow": "CHAT",
|
||||
"totalDurationMs": 45000,
|
||||
"toolCallCount": 1,
|
||||
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。",
|
||||
"selfEvaluation": {
|
||||
"verifier_evaluation": {
|
||||
"verdict": "LOW_CONFID",
|
||||
"groundedness_score": 0.42,
|
||||
"critical_fact_count": 2,
|
||||
"prompt_audit": {
|
||||
"version": "chat-prompts-v1",
|
||||
"prompts": [
|
||||
{
|
||||
"name": "chat_planner",
|
||||
"version": "chat-planner-v1",
|
||||
"resource": "prompts/chat-planner-prompt.md"
|
||||
},
|
||||
{
|
||||
"name": "chat_executor",
|
||||
"version": "chat-executor-v2",
|
||||
"resource": "prompts/chat-executor-prompt.md"
|
||||
},
|
||||
{
|
||||
"name": "chat_verifier",
|
||||
"version": "chat-verifier-v2",
|
||||
"resource": "prompts/chat-verifier-prompt.md"
|
||||
},
|
||||
{
|
||||
"name": "chat_composer",
|
||||
"version": "chat-composer-v1",
|
||||
"resource": "prompts/chat-composer-prompt.md"
|
||||
}
|
||||
]
|
||||
},
|
||||
"gatekeeper_result": {
|
||||
"status": "pass",
|
||||
"severity": "none",
|
||||
"rule_set_version": "gatekeeper-rules-v1",
|
||||
"rules": [
|
||||
{
|
||||
"id": "evidence.invocation",
|
||||
"description": "source_invocation_id must reference an existing tool invocation",
|
||||
"enabled": true,
|
||||
"default_severity": "reject"
|
||||
},
|
||||
{
|
||||
"id": "evidence.excerpt",
|
||||
"description": "evidence_excerpt must be supported by recorded evidence text",
|
||||
"enabled": true,
|
||||
"default_severity": "reject"
|
||||
}
|
||||
],
|
||||
"checked_bindings": [
|
||||
{
|
||||
"claim_id": "claim-timeout",
|
||||
"tool_name": "query_logs",
|
||||
"source_invocation_id": 22,
|
||||
"raw_path": "$.logs[0]",
|
||||
"matched_text": "order api timeout",
|
||||
"status": "pass"
|
||||
}
|
||||
],
|
||||
"failed_rules": [],
|
||||
"warnings": [],
|
||||
"errors": []
|
||||
},
|
||||
"executor_structured_output": {
|
||||
"answer_version": "executor_evidence_v2",
|
||||
"claims": [
|
||||
{
|
||||
"claim_id": "claim-timeout",
|
||||
"claim_type": "symptom",
|
||||
"claim_text": "订单接口出现超时",
|
||||
"support_level": "direct",
|
||||
"evidence_bindings": [
|
||||
{
|
||||
"source_type": "tool_trace",
|
||||
"tool_name": "query_logs",
|
||||
"source_invocation_id": 22,
|
||||
"raw_path": "$.logs[0]",
|
||||
"evidence_excerpt": "order api timeout"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"claim_id": "claim-db-primary",
|
||||
"claim_type": "root_cause",
|
||||
"claim_text": "数据库主库故障导致订单超时",
|
||||
"support_level": "weak",
|
||||
"evidence_bindings": [
|
||||
{
|
||||
"source_type": "tool_trace",
|
||||
"tool_name": "query_logs",
|
||||
"source_invocation_id": 22,
|
||||
"raw_path": "$.logs[0]",
|
||||
"evidence_excerpt": "order api timeout"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"hypotheses": [],
|
||||
"recommended_actions": [
|
||||
{
|
||||
"action_text": "补充查询数据库主库状态和错误日志",
|
||||
"reason": "当前只有订单接口超时日志"
|
||||
}
|
||||
],
|
||||
"missing_info": ["数据库主库状态", "数据库错误日志"]
|
||||
},
|
||||
"claim_checks": [
|
||||
{
|
||||
"claim_id": "claim-timeout",
|
||||
"claim_text": "订单接口出现超时",
|
||||
"claim_type": "symptom",
|
||||
"verification": "direct_observation",
|
||||
"detail": "日志直接记录 order api timeout",
|
||||
"evidence_refs": [
|
||||
{
|
||||
"source_invocation_id": 22,
|
||||
"raw_path": "$.logs[0]"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"claim_id": "claim-db-primary",
|
||||
"claim_text": "数据库主库故障导致订单超时",
|
||||
"claim_type": "root_cause",
|
||||
"verification": "unsupported",
|
||||
"detail": "日志只能证明订单接口超时,不能证明数据库主库故障",
|
||||
"evidence_refs": [
|
||||
{
|
||||
"source_invocation_id": 22,
|
||||
"raw_path": "$.logs[0]"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"facts_checked": [],
|
||||
"composer_output": {
|
||||
"status": "valid",
|
||||
"answer_summary": "日志显示订单接口超时,但数据库方向证据不足。",
|
||||
"recommended_actions": [
|
||||
{
|
||||
"action_text": "补充查询数据库主库状态和错误日志",
|
||||
"reason": "当前只有订单接口超时日志"
|
||||
}
|
||||
],
|
||||
"user_facing_answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n已确认信息:日志显示订单接口出现超时。\n\n仍需补充信息:目前没有数据库故障日志或主库状态证据,不能把该方向写成确认根因。"
|
||||
},
|
||||
"tool_trace_summary": [
|
||||
{
|
||||
"tool_name": "query_logs",
|
||||
"success": true,
|
||||
"evidence_level": "direct"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"steps": [],
|
||||
"toolInvocations": [
|
||||
{
|
||||
"id": 22,
|
||||
"sessionId": "eval-audit-metadata-low-confid",
|
||||
"toolName": "query_logs",
|
||||
"outputPreview": "order api timeout",
|
||||
"retrievalDetails": {
|
||||
"evidence_status": "supported",
|
||||
"evidence_refs": [
|
||||
{
|
||||
"raw_path": "$.logs[0]",
|
||||
"text": "order api timeout"
|
||||
}
|
||||
]
|
||||
},
|
||||
"success": true
|
||||
}
|
||||
],
|
||||
"summary": {
|
||||
"persistedStepCount": 3,
|
||||
"returnedStepCount": 3,
|
||||
"persistedToolCallCount": 1,
|
||||
"returnedToolCallCount": 1,
|
||||
"hasVerifierEvaluation": true,
|
||||
"hasFeedback": false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
{
|
||||
"session": {
|
||||
"sessionId": "eval-prompt-gatekeeper-audit-closure",
|
||||
"query": "确认 payment-service 当前是否存在 HighCPUUsage 告警,并检查审计元数据是否完整。",
|
||||
"status": "SUCCESS",
|
||||
"agentFlow": "CHAT",
|
||||
"totalDurationMs": 19000,
|
||||
"toolCallCount": 1,
|
||||
"answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||
"selfEvaluation": {
|
||||
"verifier_evaluation": {
|
||||
"verdict": "PASS",
|
||||
"groundedness_score": 1.0,
|
||||
"critical_fact_count": 1,
|
||||
"prompt_audit": {
|
||||
"version": "chat-prompts-v1",
|
||||
"prompts": [
|
||||
{
|
||||
"name": "chat_planner",
|
||||
"version": "chat-planner-v1",
|
||||
"resource": "prompts/chat-planner-prompt.md"
|
||||
},
|
||||
{
|
||||
"name": "chat_executor",
|
||||
"version": "chat-executor-v2",
|
||||
"resource": "prompts/chat-executor-prompt.md"
|
||||
},
|
||||
{
|
||||
"name": "chat_verifier",
|
||||
"version": "chat-verifier-v2",
|
||||
"resource": "prompts/chat-verifier-prompt.md"
|
||||
},
|
||||
{
|
||||
"name": "chat_composer",
|
||||
"version": "chat-composer-v1",
|
||||
"resource": "prompts/chat-composer-prompt.md"
|
||||
}
|
||||
]
|
||||
},
|
||||
"gatekeeper_result": {
|
||||
"status": "pass",
|
||||
"severity": "none",
|
||||
"rule_set_version": "gatekeeper-rules-v1",
|
||||
"rules": [
|
||||
{
|
||||
"id": "evidence.invocation",
|
||||
"description": "source_invocation_id must reference an existing tool invocation",
|
||||
"enabled": true,
|
||||
"default_severity": "reject"
|
||||
},
|
||||
{
|
||||
"id": "evidence.raw_path",
|
||||
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||
"enabled": true,
|
||||
"default_severity": "reject"
|
||||
}
|
||||
],
|
||||
"checked_bindings": [
|
||||
{
|
||||
"claim_id": "claim-1",
|
||||
"tool_name": "query_metrics",
|
||||
"source_invocation_id": 21,
|
||||
"raw_path": "$.alerts[0]",
|
||||
"matched_text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
|
||||
"status": "pass"
|
||||
}
|
||||
],
|
||||
"failed_rules": [],
|
||||
"warnings": [],
|
||||
"errors": []
|
||||
},
|
||||
"executor_structured_output": {
|
||||
"answer_version": "executor_evidence_v2",
|
||||
"claims": [
|
||||
{
|
||||
"claim_id": "claim-1",
|
||||
"claim_type": "observation",
|
||||
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||
"support_level": "direct",
|
||||
"evidence_bindings": [
|
||||
{
|
||||
"source_type": "tool_trace",
|
||||
"tool_name": "query_metrics",
|
||||
"source_invocation_id": 21,
|
||||
"raw_path": "$.alerts[0]",
|
||||
"evidence_excerpt": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"hypotheses": [],
|
||||
"recommended_actions": [],
|
||||
"missing_info": []
|
||||
},
|
||||
"claim_checks": [
|
||||
{
|
||||
"claim_id": "claim-1",
|
||||
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||
"claim_type": "observation",
|
||||
"verification": "direct_observation",
|
||||
"detail": "已核验的指标证据直接包含服务名、告警名和 CPU 当前值。",
|
||||
"evidence_refs": [
|
||||
{
|
||||
"source_invocation_id": 21,
|
||||
"raw_path": "$.alerts[0]"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"facts_checked": [],
|
||||
"composer_output": {
|
||||
"status": "valid",
|
||||
"answer_summary": "payment-service 当前存在 HighCPUUsage 告警。",
|
||||
"recommended_actions": [],
|
||||
"user_facing_answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。"
|
||||
},
|
||||
"tool_trace_summary": [
|
||||
{
|
||||
"tool_name": "query_metrics",
|
||||
"success": true,
|
||||
"source_invocation_ids": [21],
|
||||
"evidence_level": "direct"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"steps": [],
|
||||
"toolInvocations": [
|
||||
{
|
||||
"id": 21,
|
||||
"sessionId": "eval-prompt-gatekeeper-audit-closure",
|
||||
"toolName": "query_metrics",
|
||||
"outputPreview": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
|
||||
"retrievalDetails": {
|
||||
"evidence_status": "supported",
|
||||
"evidence_refs": [
|
||||
{
|
||||
"raw_path": "$.alerts[0]",
|
||||
"text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
|
||||
}
|
||||
]
|
||||
},
|
||||
"success": true
|
||||
}
|
||||
],
|
||||
"summary": {
|
||||
"persistedStepCount": 3,
|
||||
"returnedStepCount": 3,
|
||||
"persistedToolCallCount": 1,
|
||||
"returnedToolCallCount": 1,
|
||||
"hasVerifierEvaluation": true,
|
||||
"hasFeedback": false
|
||||
}
|
||||
}
|
||||
@@ -1,14 +1,14 @@
|
||||
{
|
||||
"totalCases" : 10,
|
||||
"passedCases" : 10,
|
||||
"totalCases" : 12,
|
||||
"passedCases" : 12,
|
||||
"passRate" : 1.0,
|
||||
"verdictDistribution" : {
|
||||
"PASS" : 4,
|
||||
"LOW_CONFID" : 5,
|
||||
"PASS" : 5,
|
||||
"LOW_CONFID" : 6,
|
||||
"REJECT" : 1
|
||||
},
|
||||
"averageToolCallCount" : 1.5,
|
||||
"averageDurationMs" : 39800.0,
|
||||
"averageToolCallCount" : 1.4166666666666667,
|
||||
"averageDurationMs" : 38500.0,
|
||||
"results" : [ {
|
||||
"caseId" : "narrow-highcpu-observation",
|
||||
"title" : "Narrow HighCPU observation",
|
||||
@@ -22,10 +22,31 @@
|
||||
},
|
||||
"gatekeeperStatus" : "pass",
|
||||
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : "valid",
|
||||
"claimCheckCount" : 1,
|
||||
"gatekeeperRuleCount" : 1,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 18000
|
||||
}, {
|
||||
"caseId" : "prompt-gatekeeper-audit-closure",
|
||||
"title" : "Prompt and Gatekeeper audit closure",
|
||||
"passed" : true,
|
||||
"failedChecks" : [ ],
|
||||
"verdict" : "PASS",
|
||||
"matchedKeywordCount" : 3,
|
||||
"requiredKeywordCount" : 3,
|
||||
"evidenceCoverage" : {
|
||||
"query_metrics" : true
|
||||
},
|
||||
"gatekeeperStatus" : "pass",
|
||||
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||
"promptAuditVersion" : "chat-prompts-v1",
|
||||
"composerStatus" : "valid",
|
||||
"claimCheckCount" : 1,
|
||||
"gatekeeperRuleCount" : 2,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 19000
|
||||
}, {
|
||||
"caseId" : "hikari-no-evidence-negative-observation",
|
||||
"title" : "Hikari no-evidence negative observation",
|
||||
@@ -39,8 +60,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : "pass",
|
||||
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : "valid",
|
||||
"claimCheckCount" : 1,
|
||||
"gatekeeperRuleCount" : 1,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 21000
|
||||
}, {
|
||||
@@ -58,8 +81,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : null,
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : null,
|
||||
"claimCheckCount" : null,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 3,
|
||||
"durationMs" : 42000
|
||||
}, {
|
||||
@@ -76,8 +101,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : null,
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : null,
|
||||
"claimCheckCount" : null,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 2,
|
||||
"durationMs" : 51000
|
||||
}, {
|
||||
@@ -93,8 +120,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : null,
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : null,
|
||||
"claimCheckCount" : null,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 36000
|
||||
}, {
|
||||
@@ -111,8 +140,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : null,
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : null,
|
||||
"claimCheckCount" : null,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 2,
|
||||
"durationMs" : 47000
|
||||
}, {
|
||||
@@ -129,8 +160,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : null,
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : null,
|
||||
"claimCheckCount" : null,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 2,
|
||||
"durationMs" : 53000
|
||||
}, {
|
||||
@@ -146,8 +179,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : "fail",
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : "valid",
|
||||
"claimCheckCount" : 1,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 39000
|
||||
}, {
|
||||
@@ -163,10 +198,31 @@
|
||||
},
|
||||
"gatekeeperStatus" : "pass",
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : "valid",
|
||||
"claimCheckCount" : 2,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 44000
|
||||
}, {
|
||||
"caseId" : "audit-metadata-low-confid",
|
||||
"title" : "Audit metadata low confidence",
|
||||
"passed" : true,
|
||||
"failedChecks" : [ ],
|
||||
"verdict" : "LOW_CONFID",
|
||||
"matchedKeywordCount" : 2,
|
||||
"requiredKeywordCount" : 2,
|
||||
"evidenceCoverage" : {
|
||||
"query_logs" : true
|
||||
},
|
||||
"gatekeeperStatus" : "pass",
|
||||
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||
"promptAuditVersion" : "chat-prompts-v1",
|
||||
"composerStatus" : "valid",
|
||||
"claimCheckCount" : 2,
|
||||
"gatekeeperRuleCount" : 2,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 45000
|
||||
}, {
|
||||
"caseId" : "composer-fallback-no-raw-json",
|
||||
"title" : "Composer fallback no raw JSON",
|
||||
@@ -180,8 +236,10 @@
|
||||
},
|
||||
"gatekeeperStatus" : "pass",
|
||||
"gatekeeperRuleSetVersion" : null,
|
||||
"promptAuditVersion" : null,
|
||||
"composerStatus" : "composer_malformed",
|
||||
"claimCheckCount" : 2,
|
||||
"gatekeeperRuleCount" : null,
|
||||
"toolCallCount" : 1,
|
||||
"durationMs" : 47000
|
||||
} ]
|
||||
|
||||
@@ -1,28 +1,30 @@
|
||||
# Diagnosis Eval Report
|
||||
|
||||
- Total cases: 10
|
||||
- Passed cases: 10
|
||||
- Total cases: 12
|
||||
- Passed cases: 12
|
||||
- Pass rate: 100.00%
|
||||
- Average tool calls: 1.50
|
||||
- Average duration ms: 39800.00
|
||||
- Average tool calls: 1.42
|
||||
- Average duration ms: 38500.00
|
||||
|
||||
## Verdict Distribution
|
||||
|
||||
- PASS: 4
|
||||
- LOW_CONFID: 5
|
||||
- PASS: 5
|
||||
- LOW_CONFID: 6
|
||||
- REJECT: 1
|
||||
|
||||
## Cases
|
||||
|
||||
| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |
|
||||
| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |
|
||||
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 18000 | - |
|
||||
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 21000 | - |
|
||||
| payment-timeout | PASS | PASS | - | - | - | - | 3/3 | 3 | 42000 | - |
|
||||
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 51000 | - |
|
||||
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | 2/2 | 1 | 36000 | - |
|
||||
| slow-response | PASS | PASS | - | - | - | - | 2/2 | 2 | 47000 | - |
|
||||
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 53000 | - |
|
||||
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | valid | 1 | 3/3 | 1 | 39000 | - |
|
||||
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | valid | 2 | 2/2 | 1 | 44000 | - |
|
||||
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | composer_malformed | 2 | 2/2 | 1 | 47000 | - |
|
||||
| Case | Result | Verdict | Gatekeeper | Rule Set | Prompt Audit | Composer | Claim Checks | Rules | Keywords | Tool Calls | Duration ms | Failed Checks |
|
||||
| --- | --- | --- | --- | --- | --- | --- | ---: | ---: | --- | ---: | ---: | --- |
|
||||
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 18000 | - |
|
||||
| prompt-gatekeeper-audit-closure | PASS | PASS | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 1 | 2 | 3/3 | 1 | 19000 | - |
|
||||
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | - | valid | 1 | 1 | 3/3 | 1 | 21000 | - |
|
||||
| payment-timeout | PASS | PASS | - | - | - | - | - | - | 3/3 | 3 | 42000 | - |
|
||||
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 51000 | - |
|
||||
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | - | - | 2/2 | 1 | 36000 | - |
|
||||
| slow-response | PASS | PASS | - | - | - | - | - | - | 2/2 | 2 | 47000 | - |
|
||||
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | - | - | 3/3 | 2 | 53000 | - |
|
||||
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | - | valid | 1 | - | 3/3 | 1 | 39000 | - |
|
||||
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | - | valid | 2 | - | 2/2 | 1 | 44000 | - |
|
||||
| audit-metadata-low-confid | PASS | LOW_CONFID | pass | gatekeeper-rules-v1 | chat-prompts-v1 | valid | 2 | 2 | 2/2 | 1 | 45000 | - |
|
||||
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | - | composer_malformed | 2 | - | 2/2 | 1 | 47000 | - |
|
||||
|
||||
+23
-1
@@ -34,7 +34,16 @@ baseline report:整套固定集当前认可的结果
|
||||
"expectedGatekeeperStatuses": ["pass"],
|
||||
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||
"expectedComposerStatuses": ["valid"],
|
||||
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
||||
"forbiddenConfirmedClaimKeywords": ["主库故障"],
|
||||
"requirePromptAudit": true,
|
||||
"expectedPromptAuditVersion": "chat-prompts-v1",
|
||||
"expectedPromptVersions": {
|
||||
"chat_planner": "chat-planner-v1",
|
||||
"chat_executor": "chat-executor-v2",
|
||||
"chat_verifier": "chat-verifier-v2",
|
||||
"chat_composer": "chat-composer-v1"
|
||||
},
|
||||
"requireGatekeeperRules": true
|
||||
}
|
||||
```
|
||||
|
||||
@@ -58,6 +67,10 @@ baseline report:整套固定集当前认可的结果
|
||||
| `expectedGatekeeperRuleSetVersion` | 期望的 Gatekeeper 规则集版本 | 配置后校验 `gatekeeper_result.rule_set_version` |
|
||||
| `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 |
|
||||
| `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 |
|
||||
| `requirePromptAudit` | 是否要求 Prompt 审计元数据 | 要求 `prompt_audit.version` 存在 |
|
||||
| `expectedPromptAuditVersion` | 期望的 Prompt 审计目录版本 | 配置后校验 `prompt_audit.version` |
|
||||
| `expectedPromptVersions` | 期望的各角色 Prompt 版本 | 校验 `prompt_audit.prompts[*].name/version` |
|
||||
| `requireGatekeeperRules` | 是否要求 Gatekeeper 规则元数据 | 要求 `gatekeeper_result.rules` 非空,且每条规则有 `id`、`enabled`、`default_severity` |
|
||||
|
||||
## 2. Trace Fixture
|
||||
|
||||
@@ -72,6 +85,9 @@ fixture 是一次 Agent 运行后的 trace 快照。评测器只读取当前规
|
||||
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` |
|
||||
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` |
|
||||
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | Gatekeeper 规则集版本 | 新矩阵 case 可显式断言该版本 |
|
||||
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rules` | Gatekeeper 规则元数据摘要 | 审计 case 可要求规则列表非空且字段完整 |
|
||||
| `session.selfEvaluation.verifier_evaluation.prompt_audit.version` | Chat Prompt 审计目录版本 | 审计 case 可显式断言该版本 |
|
||||
| `session.selfEvaluation.verifier_evaluation.prompt_audit.prompts` | 各 Chat Prompt 名称、版本和资源路径 | 审计 case 可断言 planner、executor、verifier、composer 版本 |
|
||||
| `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` |
|
||||
| `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 |
|
||||
| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 |
|
||||
@@ -95,8 +111,10 @@ Java 类型:`DiagnosisEvalResult`
|
||||
| `evidenceCoverage` | 每个必需工具是否出现 |
|
||||
| `gatekeeperStatus` | 读到的 `gatekeeper_result.status` |
|
||||
| `gatekeeperRuleSetVersion` | 读到的 `gatekeeper_result.rule_set_version` |
|
||||
| `promptAuditVersion` | 读到的 `prompt_audit.version` |
|
||||
| `composerStatus` | 读到的 `composer_output.status` |
|
||||
| `claimCheckCount` | `claim_checks` 数量 |
|
||||
| `gatekeeperRuleCount` | `gatekeeper_result.rules` 数量 |
|
||||
| `toolCallCount` | trace 中工具调用总数 |
|
||||
| `durationMs` | trace 总耗时 |
|
||||
|
||||
@@ -130,6 +148,10 @@ Executor structured output
|
||||
- V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。
|
||||
- `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。
|
||||
- 配置 `expectedGatekeeperRuleSetVersion` 的 case 必须匹配 `gatekeeper_result.rule_set_version`。
|
||||
- 配置 `requirePromptAudit` 的 case 必须包含 `prompt_audit.version`。
|
||||
- 配置 `expectedPromptAuditVersion` 的 case 必须匹配 `prompt_audit.version`。
|
||||
- 配置 `expectedPromptVersions` 的 case 必须能在 `prompt_audit.prompts` 中找到对应角色和版本。
|
||||
- 配置 `requireGatekeeperRules` 的 case 必须包含非空 `gatekeeper_result.rules`,且每条规则有 `id`、`enabled`、`default_severity`。
|
||||
- `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。
|
||||
- Composer 输出必须记录 `status`。
|
||||
- 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。
|
||||
|
||||
Reference in New Issue
Block a user