feat(eval): add evidence pipeline acceptance closure
This commit is contained in:
@@ -10,6 +10,7 @@
|
|||||||
| 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived |
|
| 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived |
|
||||||
| 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived |
|
| 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived |
|
||||||
| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived |
|
| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived |
|
||||||
|
| 2026-07-08 | diagnosis-eval-demo-gatekeeper-closure | Agent eval/demo/Gatekeeper | diagnosis eval matrix, stable demo scenarios, Gatekeeper rule set version, audit metadata | openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure | archived |
|
||||||
| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived |
|
| 2026-07-08 | verifier-evidence-reference-fidelity | Chat质量门禁/证据归因 | evidence_refs, raw_path, Gatekeeper severity, verifier evidence excerpt, HikariCP mock, no_evidence | openspec/changes/archive/2026-07-08-verifier-evidence-reference-fidelity | archived |
|
||||||
| 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived |
|
| 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived |
|
||||||
| 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived |
|
| 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived |
|
||||||
|
|||||||
@@ -0,0 +1,48 @@
|
|||||||
|
# diagnosis-eval-demo-gatekeeper-closure Acceptance
|
||||||
|
|
||||||
|
## Static / Structure Verification
|
||||||
|
|
||||||
|
- `cmd /c openspec validate diagnosis-eval-demo-gatekeeper-closure --strict`
|
||||||
|
- Result: passed.
|
||||||
|
- `cmd /c openspec validate --specs`
|
||||||
|
- Result: passed, 10 specs passed.
|
||||||
|
|
||||||
|
## Script Verification
|
||||||
|
|
||||||
|
- `mvn "-Dtest=ExecutorGatekeeperServiceTest,DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,VerifierInputHookTest" test`
|
||||||
|
- Result: 36 tests, 0 failures, 0 errors.
|
||||||
|
- `mvn "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest,ToolInvocationRecorderTest,QueryLogsToolsTest" test`
|
||||||
|
- Result: 61 tests, 0 failures, 0 errors.
|
||||||
|
- `mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`
|
||||||
|
- Result after E2E startup fix: 23 tests, 0 failures, 0 errors.
|
||||||
|
|
||||||
|
## Live E2E Verification
|
||||||
|
|
||||||
|
- Start command: `mvn spring-boot:run "-Dspring-boot.run.profiles=mvp-demo"`.
|
||||||
|
- Demo command: `powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1`.
|
||||||
|
- Result: chat, trace, and feedback requests completed successfully.
|
||||||
|
- Output files:
|
||||||
|
- `mvp/demo/output/chat-response.json`
|
||||||
|
- `mvp/demo/output/trace-response.json`
|
||||||
|
- `mvp/demo/output/feedback-response.json`
|
||||||
|
- Trace observations:
|
||||||
|
- `hasVerifierEvaluation=true`
|
||||||
|
- `gatekeeper_result.rule_set_version=gatekeeper-rules-v1`
|
||||||
|
|
||||||
|
## Fixed During Verification
|
||||||
|
|
||||||
|
- E2E startup initially failed because Spring could not instantiate `ExecutorGatekeeperService`.
|
||||||
|
- Root cause: two public constructors and no explicit `@Autowired` constructor.
|
||||||
|
- Fix: annotate the production constructor with `@Autowired`.
|
||||||
|
|
||||||
|
## Residual Risk
|
||||||
|
|
||||||
|
- The live payment-timeout path can still produce `LOW_CONFID` because model-generated evidence bindings may omit some explicit `source_invocation_id` values.
|
||||||
|
- This is not a blocker for this change because deterministic matrix behavior is covered by saved fixtures and baseline evaluation.
|
||||||
|
- Existing Maven warnings remain: duplicate `spring-boot-starter-test` declaration and Lombok `@Builder` default warnings.
|
||||||
|
|
||||||
|
## Archive Status
|
||||||
|
|
||||||
|
- Devflow archive artifacts created.
|
||||||
|
- OpenSpec change archived to `openspec/changes/archive/2026-07-08-diagnosis-eval-demo-gatekeeper-closure`.
|
||||||
|
- Main specs synced by `cmd /c openspec archive diagnosis-eval-demo-gatekeeper-closure --yes`.
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
# diagnosis-eval-demo-gatekeeper-closure Brief
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The Chat evidence pipeline already had Executor V2 structured output, deterministic Gatekeeper validation, Verifier claim checks, and Composer final rendering. The missing piece was an interview-ready acceptance story that made the anti-hallucination behavior easy to demonstrate and regress.
|
||||||
|
|
||||||
|
## Goal
|
||||||
|
|
||||||
|
Close the next three interview-readiness gaps together:
|
||||||
|
|
||||||
|
- diagnosis eval fixture matrix
|
||||||
|
- stable demo data set
|
||||||
|
- Gatekeeper rule configuration and audit version
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Expand `mvp/eval` with matrix-oriented cases, fixtures, and baseline reports.
|
||||||
|
- Add stable demo request payloads and scenario documentation.
|
||||||
|
- Add a lightweight local Gatekeeper rule catalog with `rule_set_version` and rule metadata in `gatekeeper_result`.
|
||||||
|
- Update architecture, demo, and eval docs to describe the current implementation.
|
||||||
|
|
||||||
|
## Non-goals
|
||||||
|
|
||||||
|
- No new public HTTP endpoint.
|
||||||
|
- No new database table.
|
||||||
|
- No Planner `scope_contract`.
|
||||||
|
- No Gatekeeper retry loop.
|
||||||
|
- No remote or dynamic rule execution engine.
|
||||||
|
|
||||||
|
## OpenSpec
|
||||||
|
|
||||||
|
- Change: `openspec/changes/diagnosis-eval-demo-gatekeeper-closure`
|
||||||
|
- Interface impact: L2 internal contract change.
|
||||||
@@ -0,0 +1,144 @@
|
|||||||
|
# diagnosis-eval-demo-gatekeeper-closure Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: implement the next three interview-readiness items together: diagnosis eval fixture matrix, stable demo data set, and Gatekeeper rule configuration/audit version.
|
||||||
|
- Slug: `diagnosis-eval-demo-gatekeeper-closure`
|
||||||
|
- Devflow scale: `standard`
|
||||||
|
- Interface impact: expected L2 internal contract change because `gatekeeper_result` audit JSON will gain rule metadata/version fields.
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- `devflow/index.md` used: related entries found for diagnosis eval harness, fixture expansion, MVP demo runbook, Gatekeeper hook, and verifier evidence reference fidelity.
|
||||||
|
- Relevant glossary:
|
||||||
|
- Evidence Tools produce incident facts and must be recorded in `tool_invocation`.
|
||||||
|
- Verifier should not use skills/runbooks as incident evidence.
|
||||||
|
- `tool_invocation.retrieval_details` is the structured evidence/audit home for tool-specific details.
|
||||||
|
- Historical constraints that must enter OpenSpec:
|
||||||
|
- Diagnosis eval is offline and deterministic; no LLM-as-judge.
|
||||||
|
- Demo assets should be runnable, but fixed regression should use saved fixtures.
|
||||||
|
- Gatekeeper remains in the Verifier hook path.
|
||||||
|
- No new database table for Gatekeeper audit; use `self_evaluation.verifier_evaluation.gatekeeper_result`.
|
||||||
|
- `$.no_evidence` is a query no-hit signal, not proof that a problem is impossible.
|
||||||
|
|
||||||
|
## Question Pool
|
||||||
|
|
||||||
|
| ID | Dimension | Mode | Question | Status |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Q1 | Terminology | evidence-driven | What names should this change use for the matrix, demo set, and Gatekeeper rule metadata? | Resolved |
|
||||||
|
| Q2 | Boundary | evidence-driven | Should this change alter public APIs, database schema, Planner output, or retry behavior? | Resolved |
|
||||||
|
| Q3 | Acceptance | evidence-driven | Which existing tests and baseline assets define the current acceptance style? | Resolved |
|
||||||
|
| Q4 | Technical | evidence-driven | Where should Gatekeeper rule metadata live with minimal implementation risk? | Pending code research |
|
||||||
|
| Q5 | Scope | user-interview | Should the stable demo set be documentation/payloads only, or should it include live E2E scripts for all scenarios? | Confirmed |
|
||||||
|
|
||||||
|
## Evidence-driven Conclusions
|
||||||
|
|
||||||
|
- Q1 conclusion: use `diagnosis eval matrix`, `stable demo scenarios`, and `Gatekeeper rule set version` as terms.
|
||||||
|
- Q2 conclusion: keep this as an internal contract change. Do not add public endpoints, tables, Planner `scope_contract`, or Gatekeeper retry.
|
||||||
|
- Q3 conclusion: existing `DiagnosisTraceEvaluatorTest`, `ExecutorGatekeeperServiceTest`, `VerifierInputHookTest`, `ToolInvocationRecorderTest`, and `mvp/eval/reports` define the current acceptance style.
|
||||||
|
- Q4 conclusion: Gatekeeper metadata should live behind a small rule catalog loaded by `ExecutorGatekeeperService`; the audit output should include a rule set version and enabled rule metadata summary, without adding tables or remote registry.
|
||||||
|
|
||||||
|
## User-interview Confirmations
|
||||||
|
|
||||||
|
- Q5 confirmed by resumed objective: complete items 1/2/3 with sm-flow, archive, submit, and run end-to-end if necessary.
|
||||||
|
- Implementation interpretation: stable demo scenarios will be fixed request payloads and runbook docs plus deterministic fixture-backed eval. Live E2E remains necessary only for at least one main path or where unit/fixture evidence is insufficient.
|
||||||
|
|
||||||
|
## OpenSpec Backfill
|
||||||
|
|
||||||
|
- Created Draft proposal at `openspec/changes/diagnosis-eval-demo-gatekeeper-closure/proposal.md`.
|
||||||
|
- Context constraints from historical devflow entries were written into the proposal.
|
||||||
|
- Scope confirmation and Gatekeeper catalog placement were written into the proposal/design.
|
||||||
|
|
||||||
|
## Current Checkpoint
|
||||||
|
|
||||||
|
- Discover completed.
|
||||||
|
- No implementation files changed yet.
|
||||||
|
|
||||||
|
## Specify / Alignment
|
||||||
|
|
||||||
|
### Cross-artifact Alignment
|
||||||
|
|
||||||
|
| Check | Status | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| brief/proposal goals -> proposal | Aligned | Proposal covers eval matrix, stable demo scenarios, and Gatekeeper rule catalog/audit version. |
|
||||||
|
| proposal scope/constraints -> design | Aligned | Design records offline deterministic eval, fixture-backed demo distinction, local rule catalog, and no new table/API. |
|
||||||
|
| design decisions -> specs/tasks | Aligned | Specs cover eval matrix, rule set version validation, demo scenarios, and Gatekeeper rule metadata; tasks cover matching implementation slices. |
|
||||||
|
| specs observable behavior -> tasks | Aligned | Each requirement has an executable task and acceptance check. |
|
||||||
|
|
||||||
|
### Interface Impact
|
||||||
|
|
||||||
|
- Level: L2 internal contract change.
|
||||||
|
- Reason: `gatekeeper_result` internal audit JSON gains `rule_set_version` and rule metadata summary. Eval case/result fields may gain optional rule set checks. No public HTTP API, database schema, or external DTO contract changes.
|
||||||
|
|
||||||
|
## Audit
|
||||||
|
|
||||||
|
Input -> processing -> output chain:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mvp/demo request docs + mvp/eval fixtures
|
||||||
|
-> DiagnosisTraceEvaluator
|
||||||
|
-> baseline reports
|
||||||
|
-> interview/demo evidence
|
||||||
|
|
||||||
|
Gatekeeper rule catalog
|
||||||
|
-> ExecutorGatekeeperService
|
||||||
|
-> VerifierInputHook / ChatService persisted self_evaluation
|
||||||
|
-> Trace and eval audit
|
||||||
|
```
|
||||||
|
|
||||||
|
Architecture risk assessment:
|
||||||
|
|
||||||
|
1. The change is intentionally internal and should not add new public consumers.
|
||||||
|
2. Gatekeeper catalog must stay metadata-only; dynamic rule execution would be a different, riskier architecture.
|
||||||
|
3. Fixture-backed demo scenarios should be documented as deterministic regression artifacts, not live LLM guarantees.
|
||||||
|
4. Baseline report churn is expected and must be committed with case/fixture changes.
|
||||||
|
5. No devflow/OpenSpec conflict found.
|
||||||
|
|
||||||
|
## Commit Gate
|
||||||
|
|
||||||
|
- `cmd /c openspec validate diagnosis-eval-demo-gatekeeper-closure --strict`: passed.
|
||||||
|
- `cmd /c openspec validate --specs`: passed, 10 specs passed.
|
||||||
|
- File completeness:
|
||||||
|
- proposal.md: present.
|
||||||
|
- design.md: present.
|
||||||
|
- specs: present for `diagnosis-eval-harness`, `mvp-demo-trace-acceptance`, `chat-verifier-agent`.
|
||||||
|
- tasks.md: present.
|
||||||
|
- Consistency:
|
||||||
|
- Proposal concepts have corresponding design sections.
|
||||||
|
- Design decisions are reflected in specs/tasks.
|
||||||
|
- Task acceptance checks are verifiable.
|
||||||
|
|
||||||
|
## Current Checkpoint
|
||||||
|
|
||||||
|
- Commit completed.
|
||||||
|
- `.committed` marker created.
|
||||||
|
|
||||||
|
## Apply Verification
|
||||||
|
|
||||||
|
- Focused verification passed:
|
||||||
|
- `mvn "-Dtest=ExecutorGatekeeperServiceTest,DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,VerifierInputHookTest" test`
|
||||||
|
- Result: 36 tests, 0 failures, 0 errors.
|
||||||
|
- Broader relevant regression passed:
|
||||||
|
- `mvn "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest,ToolInvocationRecorderTest,QueryLogsToolsTest" test`
|
||||||
|
- Result: 61 tests, 0 failures, 0 errors.
|
||||||
|
- E2E startup repro found a Spring bean construction issue:
|
||||||
|
- Command: `mvn spring-boot:run "-Dspring-boot.run.profiles=mvp-demo"`
|
||||||
|
- Failure: `ExecutorGatekeeperService` had two public constructors and no annotated constructor, so Spring attempted a no-arg constructor and failed with `No default constructor found`.
|
||||||
|
- Classification: code deviation from OpenSpec implementation intent, not a spec gap.
|
||||||
|
- Fix: annotate the production constructor with `@Autowired`.
|
||||||
|
- Post-fix focused regression passed:
|
||||||
|
- `mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest" test`
|
||||||
|
- Result: 23 tests, 0 failures, 0 errors.
|
||||||
|
- Live E2E passed for demo compatibility:
|
||||||
|
- Start: `mvn spring-boot:run "-Dspring-boot.run.profiles=mvp-demo"`
|
||||||
|
- Run: `powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1`
|
||||||
|
- Result: `/api/chat`, `/api/diagnosis/{sessionId}/trace`, and `/api/feedback` completed successfully.
|
||||||
|
- Trace summary included `hasVerifierEvaluation=true`.
|
||||||
|
- Persisted Gatekeeper audit included `rule_set_version=gatekeeper-rules-v1`.
|
||||||
|
- Residual quality note: the live payment-timeout response remained `LOW_CONFID` because some model-produced evidence bindings still lacked explicit `source_invocation_id`; deterministic PASS/LOW_CONFID/REJECT claims are covered by fixture-backed eval.
|
||||||
|
|
||||||
|
## Archive Readiness
|
||||||
|
|
||||||
|
- OpenSpec tasks 1-4 completed.
|
||||||
|
- Verification is recorded in devflow acceptance artifacts.
|
||||||
|
- Remaining known risk: live LLM output is not deterministic and may still produce LOW_CONFID on the payment-timeout path; this is intentionally documented as demo compatibility, not a fixed PASS guarantee.
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
# diagnosis-eval-demo-gatekeeper-closure Evidence
|
||||||
|
|
||||||
|
## Code And Artifact Evidence
|
||||||
|
|
||||||
|
- Gatekeeper rule metadata lives in `src/main/resources/gatekeeper/gatekeeper-rules.json`.
|
||||||
|
- `ExecutorGatekeeperService` loads the local catalog, uses configured threshold parameters, and emits `rule_set_version` plus enabled rule metadata.
|
||||||
|
- `VerifierInputHook` and `ChatService` preserve Gatekeeper audit metadata in fallback/default paths.
|
||||||
|
- `DiagnosisTraceEvaluator` can optionally validate expected Gatekeeper rule set version.
|
||||||
|
- `mvp/eval/cases/diagnosis-cases.json` now includes narrow-scope and no-evidence matrix cases.
|
||||||
|
- `mvp/eval/reports/baseline-report.json` and `.md` were regenerated for the expanded fixed matrix.
|
||||||
|
- `mvp/demo/evidence-pipeline-scenarios.md` documents live vs fixture-backed demo scenarios.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
- Keep this phase internal: no public API, no DB schema, no Planner output change.
|
||||||
|
- Keep Gatekeeper deterministic Java validation; the catalog is metadata/config only.
|
||||||
|
- Treat live demo as compatibility evidence and fixture-backed eval as deterministic regression evidence.
|
||||||
|
- Persist audit under the existing `self_evaluation.verifier_evaluation.gatekeeper_result` structure.
|
||||||
|
|
||||||
|
## Runtime Finding
|
||||||
|
|
||||||
|
The first Maven E2E startup found a real integration issue: `ExecutorGatekeeperService` had multiple public constructors without an annotated constructor, so Spring could not instantiate the service. The fix was to annotate the production constructor with `@Autowired`.
|
||||||
@@ -397,7 +397,8 @@ Prompt、Hook、Gatekeeper、Verifier、Composer 和评测门禁的完整说明
|
|||||||
- Chat verifier 和 AIOps rule evaluation 合并进 `self_evaluation`。
|
- Chat verifier 和 AIOps rule evaluation 合并进 `self_evaluation`。
|
||||||
- Chat Executor 结构化输出 `executor_evidence_v2`,不再直接承担最终用户答复。
|
- Chat Executor 结构化输出 `executor_evidence_v2`,不再直接承担最终用户答复。
|
||||||
- `tool_invocation.retrieval_details.evidence_refs` 支持 `raw_path` 精确引用和 `$.no_evidence` 负向证据。
|
- `tool_invocation.retrieval_details.evidence_refs` 支持 `raw_path` 精确引用和 `$.no_evidence` 负向证据。
|
||||||
- Gatekeeper 对 Executor 引用做代码级验真,Verifier 只判断可推导性。
|
- Gatekeeper 对 Executor 引用做代码级验真,并在审计中记录 `rule_set_version` 和规则元数据摘要。
|
||||||
|
- Verifier 只判断可推导性。
|
||||||
- Composer 在 Verifier 之后生成最终用户表达,并限制 negative observation 过度表述。
|
- Composer 在 Verifier 之后生成最终用户表达,并限制 negative observation 过度表述。
|
||||||
- RAG offline baseline 和 live acceptance 脚本。
|
- RAG offline baseline 和 live acceptance 脚本。
|
||||||
|
|
||||||
|
|||||||
@@ -267,7 +267,7 @@ category
|
|||||||
|---|---|
|
|---|---|
|
||||||
| `executor_output_parse_status` | Executor 输出是否能解析为 `executor_evidence_v2` |
|
| `executor_output_parse_status` | Executor 输出是否能解析为 `executor_evidence_v2` |
|
||||||
| `executor_structured_output` | Executor 结构化 claims、hypotheses、recommended_actions、missing_info |
|
| `executor_structured_output` | Executor 结构化 claims、hypotheses、recommended_actions、missing_info |
|
||||||
| `gatekeeper_result` | 引用真实性校验结果,包括 checked bindings、failed rules、warnings、errors |
|
| `gatekeeper_result` | 引用真实性校验结果,包括 rule set version、checked bindings、failed rules、warnings、errors |
|
||||||
| `composer_output` | Composer 最终表达及解析状态 |
|
| `composer_output` | Composer 最终表达及解析状态 |
|
||||||
| `tool_trace_summary` | Verifier 调用时使用的工具调用导航索引,不是唯一证据源 |
|
| `tool_trace_summary` | Verifier 调用时使用的工具调用导航索引,不是唯一证据源 |
|
||||||
|
|
||||||
|
|||||||
@@ -233,6 +233,15 @@ Gatekeeper 位于 Verifier 前,由 `VerifierInputHook` 触发,负责代码
|
|||||||
{
|
{
|
||||||
"status": "pass",
|
"status": "pass",
|
||||||
"severity": "none",
|
"severity": "none",
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
}
|
||||||
|
],
|
||||||
"checked_bindings": [
|
"checked_bindings": [
|
||||||
{
|
{
|
||||||
"claim_id": "claim-1",
|
"claim_id": "claim-1",
|
||||||
@@ -255,6 +264,8 @@ Gatekeeper 位于 Verifier 前,由 `VerifierInputHook` 触发,负责代码
|
|||||||
|---|---|---|
|
|---|---|---|
|
||||||
| `status` | string | `pass` 或 `fail` |
|
| `status` | string | `pass` 或 `fail` |
|
||||||
| `severity` | string | `none`、`low_confid`、`reject` |
|
| `severity` | string | `none`、`low_confid`、`reject` |
|
||||||
|
| `rule_set_version` | string | 当前加载的 Gatekeeper 规则集版本 |
|
||||||
|
| `rules` | array | 已启用规则的轻量元数据摘要 |
|
||||||
| `checked_bindings` | array | 每条证据绑定的校验结果 |
|
| `checked_bindings` | array | 每条证据绑定的校验结果 |
|
||||||
| `failed_rules` | array | 失败规则 id |
|
| `failed_rules` | array | 失败规则 id |
|
||||||
| `warnings` | array | 自动回填等非阻断信息 |
|
| `warnings` | array | 自动回填等非阻断信息 |
|
||||||
@@ -271,6 +282,12 @@ Gatekeeper 位于 Verifier 前,由 `VerifierInputHook` 触发,负责代码
|
|||||||
- `evidence_excerpt` 必须由 `evidence_refs[].text` 支撑。
|
- `evidence_excerpt` 必须由 `evidence_refs[].text` 支撑。
|
||||||
- `negative_observation` 只能绑定 `$.no_evidence`。
|
- `negative_observation` 只能绑定 `$.no_evidence`。
|
||||||
|
|
||||||
|
规则配置:
|
||||||
|
|
||||||
|
- 当前规则元数据位于 `src/main/resources/gatekeeper/gatekeeper-rules.json`。
|
||||||
|
- 规则实现仍是确定性 Java 代码,不执行动态脚本。
|
||||||
|
- 当前配置只承载规则 id、描述、默认 severity、启用状态和简单参数,例如 excerpt token overlap 阈值。
|
||||||
|
|
||||||
失败分级:
|
失败分级:
|
||||||
|
|
||||||
| 场景 | severity |
|
| 场景 | severity |
|
||||||
@@ -389,7 +406,9 @@ Composer 位于 Verifier 之后,输入是 ChatService 过滤后的允许表达
|
|||||||
"traceability_version": "v1",
|
"traceability_version": "v1",
|
||||||
"executor_output_parse_status": {},
|
"executor_output_parse_status": {},
|
||||||
"executor_structured_output": {},
|
"executor_structured_output": {},
|
||||||
"gatekeeper_result": {},
|
"gatekeeper_result": {
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1"
|
||||||
|
},
|
||||||
"composer_output": {},
|
"composer_output": {},
|
||||||
"tool_trace_summary": []
|
"tool_trace_summary": []
|
||||||
}
|
}
|
||||||
@@ -420,6 +439,5 @@ Trace API 可用于回放:
|
|||||||
当前架构文档已记录主链路、数据契约和语义边界。后续如果继续实现,建议再补:
|
当前架构文档已记录主链路、数据契约和语义边界。后续如果继续实现,建议再补:
|
||||||
|
|
||||||
1. Planner `scope_contract` 的 ADR:只有当 Prompt-first 无法稳定控制越界时再引入。
|
1. Planner `scope_contract` 的 ADR:只有当 Prompt-first 无法稳定控制越界时再引入。
|
||||||
2. Gatekeeper 规则配置化文档:如果后续把规则做成索引层、元数据层、规则层,需要单独记录加载顺序和审计字段。
|
2. 更完整的 Gatekeeper 规则配置化:当前只有本地轻量 metadata/catalog,后续如果做索引层、元数据层、远程规则层,需要单独记录加载顺序、变更审批和回滚策略。
|
||||||
3. E2E fixture 矩阵:把 ISS-008/ISS-009 的用例固化到诊断评测集,而不是只存在 issue 验证记录。
|
3. Prompt version 记录:当前 prompt 变更没有版本号,后续如果需要回滚和对比,应记录 prompt version。
|
||||||
4. Prompt version 记录:当前 prompt 变更没有版本号,后续如果需要回滚和对比,应记录 prompt version。
|
|
||||||
|
|||||||
@@ -173,6 +173,8 @@ Gatekeeper 检查:
|
|||||||
| `evidence_excerpt` 由 `evidence_refs[].text` 支撑 | excerpt 编造或错配直接拒绝 |
|
| `evidence_excerpt` 由 `evidence_refs[].text` 支撑 | excerpt 编造或错配直接拒绝 |
|
||||||
| `negative_observation` 只能引用 `$.no_evidence` | 用正向日志证明“没查到”直接拒绝 |
|
| `negative_observation` 只能引用 `$.no_evidence` | 用正向日志证明“没查到”直接拒绝 |
|
||||||
|
|
||||||
|
Gatekeeper 审计还会记录 `rule_set_version` 和已启用规则元数据摘要。当前规则元数据来自本地 `gatekeeper-rules.json`,规则执行仍是确定性 Java 代码。
|
||||||
|
|
||||||
Verifier 输出:
|
Verifier 输出:
|
||||||
|
|
||||||
```json
|
```json
|
||||||
@@ -230,7 +232,7 @@ diagnosis_session.self_evaluation.aiops_rule_evaluation
|
|||||||
- 工具参数 schema 校验。
|
- 工具参数 schema 校验。
|
||||||
- 同一工具调用次数上限。
|
- 同一工具调用次数上限。
|
||||||
- 工具超时的统一熔断。
|
- 工具超时的统一熔断。
|
||||||
- Gatekeeper 规则三层分离:索引层、元数据层、规则实现层。
|
- Gatekeeper 规则远程化或三层分离:索引层、元数据层、规则实现层。
|
||||||
- Prompt 版本记录和回滚。
|
- Prompt 版本记录和回滚。
|
||||||
- Verifier 对 AIOps 报告的 LLM 级事实校验。
|
- Verifier 对 AIOps 报告的 LLM 级事实校验。
|
||||||
|
|
||||||
|
|||||||
@@ -6,9 +6,13 @@
|
|||||||
|
|
||||||
- `ten-minute-interview-demo.md`:10 分钟现场演示脚本。
|
- `ten-minute-interview-demo.md`:10 分钟现场演示脚本。
|
||||||
- `interview-walkthrough.md`:面试讲解话术。
|
- `interview-walkthrough.md`:面试讲解话术。
|
||||||
|
- `evidence-pipeline-scenarios.md`:PASS / LOW_CONFID / REJECT / no-evidence 场景矩阵。
|
||||||
- `trace-inspection-checklist.md`:Trace 字段检查清单。
|
- `trace-inspection-checklist.md`:Trace 字段检查清单。
|
||||||
- `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。
|
- `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。
|
||||||
- `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。
|
- `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。
|
||||||
|
- `requests/narrow-highcpu-chat.json`:窄范围正向观察请求。
|
||||||
|
- `requests/hikari-no-evidence-chat.json`:no-evidence 负向观察请求。
|
||||||
|
- `requests/safety-unsupported-claim-chat.json`:安全降级讨论请求。
|
||||||
|
|
||||||
## 1. 前置条件
|
## 1. 前置条件
|
||||||
|
|
||||||
@@ -167,3 +171,13 @@ AIOps 主线:
|
|||||||
-> AIOps rule evaluation
|
-> AIOps rule evaluation
|
||||||
-> Trace API 回放
|
-> Trace API 回放
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## 8. Evidence Pipeline 场景矩阵
|
||||||
|
|
||||||
|
面试时不要把所有安全场景都压到 live LLM 现场表现上。建议使用:
|
||||||
|
|
||||||
|
- `scripts/run-payment-timeout-demo.ps1` 跑主路径。
|
||||||
|
- `evidence-pipeline-scenarios.md` 讲解 PASS / LOW_CONFID / REJECT / no-evidence 矩阵。
|
||||||
|
- `mvp/eval/reports/baseline-report.md` 证明固定 fixture 10/10 通过。
|
||||||
|
|
||||||
|
这样可以同时展示真实链路和确定性回归能力。
|
||||||
|
|||||||
@@ -0,0 +1,63 @@
|
|||||||
|
# Evidence Pipeline Demo Scenarios
|
||||||
|
|
||||||
|
这份清单用于面试时说明 Chat 证据链路如何覆盖 `PASS`、`LOW_CONFID`、`REJECT` 和 no-evidence 场景。
|
||||||
|
|
||||||
|
重点区别:
|
||||||
|
|
||||||
|
- Live demo 证明本地服务、工具、Trace、Feedback 主链路能跑通。
|
||||||
|
- Fixture-backed eval 证明固定安全场景可以确定性回归,不依赖 LLM 当场随机输出。
|
||||||
|
|
||||||
|
## Scenario Matrix
|
||||||
|
|
||||||
|
| 场景 | 类型 | 输入/证据 | 期望讲点 |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Payment timeout | Live 主路径 | `requests/payment-timeout-chat.json` | 完整 Chat -> Trace -> Feedback 闭环 |
|
||||||
|
| Narrow HighCPU observation | Live 可尝试 + fixture-backed | `requests/narrow-highcpu-chat.json` / `mvp/eval/fixtures/narrow-highcpu-observation-pass.json` | Executor 只输出观察类 claim,Gatekeeper 验引用,Verifier PASS |
|
||||||
|
| Hikari no-evidence | Live 可尝试 + fixture-backed | `requests/hikari-no-evidence-chat.json` / `mvp/eval/fixtures/hikari-no-evidence-negative-observation-pass.json` | `$.no_evidence` 只表示本次查询无匹配证据,Composer 不说“已排除” |
|
||||||
|
| Unsupported claim filtering | Fixture-backed | `requests/safety-unsupported-claim-chat.json` / `mvp/eval/fixtures/unsupported-claim-filtering-low-confid.json` | Verifier 将 unsupported claim 降为 LOW_CONFID,最终答案不确认“主库故障” |
|
||||||
|
| Fabricated invocation reject | Fixture-backed | `mvp/eval/fixtures/gatekeeper-fabricated-invocation-reject.json` | Gatekeeper 拦截伪造 invocation,最终 REJECT/降级 |
|
||||||
|
| Composer fallback | Fixture-backed | `mvp/eval/fixtures/composer-fallback-no-raw-json-low-confid.json` | 即使 Composer 输出异常,也不能把 Executor JSON 泄漏给用户 |
|
||||||
|
|
||||||
|
## Trace Fields To Inspect
|
||||||
|
|
||||||
|
| 能力 | JSON path |
|
||||||
|
|---|---|
|
||||||
|
| Executor V2 输出 | `data.session.selfEvaluation.verifier_evaluation.executor_structured_output` |
|
||||||
|
| Gatekeeper 结果 | `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` |
|
||||||
|
| Gatekeeper 规则版本 | `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` |
|
||||||
|
| 证据绑定校验 | `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.checked_bindings` |
|
||||||
|
| Verifier claim checks | `data.session.selfEvaluation.verifier_evaluation.claim_checks` |
|
||||||
|
| Composer 输出 | `data.session.selfEvaluation.verifier_evaluation.composer_output` |
|
||||||
|
| 工具证据引用 | `data.toolInvocations[*].retrievalDetails.evidence_refs` |
|
||||||
|
|
||||||
|
## How To Present It
|
||||||
|
|
||||||
|
```text
|
||||||
|
我把现场 demo 和固定 eval 分开。
|
||||||
|
现场 demo 证明系统能跑通真实链路;
|
||||||
|
fixture-backed eval 证明反幻觉安全场景可以稳定回归。
|
||||||
|
Gatekeeper 的规则版本也进入 trace,所以后续调整阈值或规则时可以审计。
|
||||||
|
```
|
||||||
|
|
||||||
|
## Optional Live Requests
|
||||||
|
|
||||||
|
手动发送某个请求样例:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
$body = Get-Content -Raw -Encoding UTF8 "mvp/demo/requests/narrow-highcpu-chat.json"
|
||||||
|
Invoke-RestMethod `
|
||||||
|
-Method Post `
|
||||||
|
-Uri "http://localhost:9900/api/chat" `
|
||||||
|
-ContentType "application/json" `
|
||||||
|
-Body $body
|
||||||
|
```
|
||||||
|
|
||||||
|
然后查询同一 session:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
Invoke-RestMethod `
|
||||||
|
-Method Get `
|
||||||
|
-Uri "http://localhost:9900/api/diagnosis/mvp-demo-narrow-highcpu-001/trace"
|
||||||
|
```
|
||||||
|
|
||||||
|
注意:除 payment-timeout 主路径外,其它 live 请求是“可尝试”的演示入口;稳定验收以 `mvp/eval` fixture 和 baseline 为准。
|
||||||
@@ -24,6 +24,7 @@
|
|||||||
4. 打开 `mvp/demo/output/trace-response.json`。
|
4. 打开 `mvp/demo/output/trace-response.json`。
|
||||||
5. 指出证据工具和 verifier evaluation。
|
5. 指出证据工具和 verifier evaluation。
|
||||||
6. 提交 feedback,并展示它挂在同一个 session 上。
|
6. 提交 feedback,并展示它挂在同一个 session 上。
|
||||||
|
7. 打开 `evidence-pipeline-scenarios.md`,说明 PASS / LOW_CONFID / REJECT / no-evidence 的固定回归矩阵。
|
||||||
|
|
||||||
## 3. 命令
|
## 3. 命令
|
||||||
|
|
||||||
@@ -125,6 +126,14 @@ Demo 证明真实链路能跑通,offline eval baseline 证明固定 case 可
|
|||||||
这两者分开是有意的:Demo 面向人类审阅,eval 面向自动化信号。
|
这两者分开是有意的:Demo 面向人类审阅,eval 面向自动化信号。
|
||||||
```
|
```
|
||||||
|
|
||||||
|
如果被问到怎么防止证据归因幻觉,可以补充:
|
||||||
|
|
||||||
|
```text
|
||||||
|
Executor 的 claim 必须绑定 source_invocation_id、raw_path 和 evidence_excerpt。
|
||||||
|
Gatekeeper 用代码核验这些引用,并把 rule_set_version 写进 trace。
|
||||||
|
Verifier 只判断已核验证据能否推出 claim,Composer 只表达允许输出的内容。
|
||||||
|
```
|
||||||
|
|
||||||
## 5. 强面试表达
|
## 5. 强面试表达
|
||||||
|
|
||||||
```text
|
```text
|
||||||
@@ -141,4 +150,3 @@ traceability、evidence persistence、verifier gating、feedback 和 regression
|
|||||||
mvp-demo profile mock 了日志和指标,但不是完整生产运行环境。
|
mvp-demo profile mock 了日志和指标,但不是完整生产运行环境。
|
||||||
密钥清理、默认隔离测试和生产可靠性是后续 hardening 工作。
|
密钥清理、默认隔离测试和生产可靠性是后续 hardening 工作。
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"Id": "mvp-demo-hikari-no-evidence-001",
|
||||||
|
"Question": "确认 inventory-service 当前是否有 HikariCP 连接池耗尽日志。"
|
||||||
|
}
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"Id": "mvp-demo-narrow-highcpu-001",
|
||||||
|
"Question": "确认 payment-service 当前是否存在 HighCPUUsage 告警。"
|
||||||
|
}
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"Id": "mvp-demo-safety-unsupported-001",
|
||||||
|
"Question": "订单超时是否可以确认由数据库主库故障导致?请只基于当前证据回答。"
|
||||||
|
}
|
||||||
@@ -10,6 +10,7 @@
|
|||||||
| `data.session.query` | 是否包含支付超时问题 | Trace 记录了原始用户意图 |
|
| `data.session.query` | 是否包含支付超时问题 | Trace 记录了原始用户意图 |
|
||||||
| `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace |
|
| `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace |
|
||||||
| `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 |
|
| `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 |
|
||||||
|
| `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 如果是 Chat V2 链路,是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 |
|
||||||
| `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在同一次诊断上 |
|
| `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在同一次诊断上 |
|
||||||
|
|
||||||
## 2. Agent 步骤
|
## 2. Agent 步骤
|
||||||
@@ -30,6 +31,7 @@
|
|||||||
| `data.toolInvocations[*].outputPreview` | 是否有受控长度的证据预览 | 保留证据但不倾倒巨大 payload |
|
| `data.toolInvocations[*].outputPreview` | 是否有受控长度的证据预览 | 保留证据但不倾倒巨大 payload |
|
||||||
| `data.toolInvocations[*].success` | 是否区分成功和失败 | 工具失败对 Verifier 和 reviewer 可见 |
|
| `data.toolInvocations[*].success` | 是否区分成功和失败 | 工具失败对 Verifier 和 reviewer 可见 |
|
||||||
| `data.toolInvocations[*].retrievalDetails` | 是否包含检索 metadata | 检索质量可事后检查 |
|
| `data.toolInvocations[*].retrievalDetails` | 是否包含检索 metadata | 检索质量可事后检查 |
|
||||||
|
| `data.toolInvocations[*].retrievalDetails.evidence_refs` | 是否包含 `raw_path + text` | Gatekeeper 可以用代码核对 Executor 引用 |
|
||||||
| `data.toolInvocations[*].relevanceLevel` | 是否有相关性等级 | 可解释检索结果强弱 |
|
| `data.toolInvocations[*].relevanceLevel` | 是否有相关性等级 | 可解释检索结果强弱 |
|
||||||
|
|
||||||
## 4. Summary
|
## 4. Summary
|
||||||
|
|||||||
+8
-4
@@ -29,18 +29,21 @@ The baseline evaluates saved trace fixtures. It does not start the application a
|
|||||||
The committed baseline currently contains:
|
The committed baseline currently contains:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
8 fixed cases
|
10 fixed cases
|
||||||
8 passing fixture evaluations
|
10 passing fixture evaluations
|
||||||
2 PASS verdicts
|
4 PASS verdicts
|
||||||
5 LOW_CONFID verdicts
|
5 LOW_CONFID verdicts
|
||||||
1 REJECT verdict
|
1 REJECT verdict
|
||||||
```
|
```
|
||||||
|
|
||||||
The three V2 audit-closure cases cover:
|
The V2 evidence-pipeline matrix covers:
|
||||||
|
|
||||||
|
- Positive supported evidence for a narrow HighCPU observation.
|
||||||
|
- No-evidence `negative_observation` using `$.no_evidence`.
|
||||||
- Gatekeeper failure for a fabricated tool invocation reference.
|
- Gatekeeper failure for a fabricated tool invocation reference.
|
||||||
- Unsupported claim filtering before the final answer.
|
- Unsupported claim filtering before the final answer.
|
||||||
- Composer fallback rendering without raw Executor JSON leakage.
|
- Composer fallback rendering without raw Executor JSON leakage.
|
||||||
|
- Gatekeeper rule set version audit for new matrix fixtures.
|
||||||
|
|
||||||
## Verification
|
## Verification
|
||||||
|
|
||||||
@@ -76,6 +79,7 @@ Stage 5 adds these V2 checks:
|
|||||||
- Required V2 fixtures must include `gatekeeper_result`, `claim_checks`, and `composer_output`.
|
- Required V2 fixtures must include `gatekeeper_result`, `claim_checks`, and `composer_output`.
|
||||||
- `claim_checks` must be structurally auditable.
|
- `claim_checks` must be structurally auditable.
|
||||||
- Composer output must record whether normal parsing or fallback rendering was used.
|
- Composer output must record whether normal parsing or fallback rendering was used.
|
||||||
|
- Gatekeeper rule set version can be asserted per fixture.
|
||||||
- Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`.
|
- Final answers must not leak raw Executor protocol markers such as `executor_evidence_v2`, `answer_version`, `evidence_bindings`, or `claim_id`.
|
||||||
- Configured unsupported claim keywords must not appear as confirmed final-answer content.
|
- Configured unsupported claim keywords must not appear as confirmed final-answer content.
|
||||||
|
|
||||||
|
|||||||
@@ -1,4 +1,40 @@
|
|||||||
[
|
[
|
||||||
|
{
|
||||||
|
"id": "narrow-highcpu-observation",
|
||||||
|
"title": "Narrow HighCPU observation",
|
||||||
|
"question": "确认 payment-service 当前是否存在 HighCPUUsage 告警。",
|
||||||
|
"traceFixture": "narrow-highcpu-observation-pass.json",
|
||||||
|
"expectedRootCauseKeywords": ["payment-service", "HighCPUUsage", "92%"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["query_metrics"],
|
||||||
|
"allowedVerdicts": ["PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["根因", "修复建议", "通常情况下"],
|
||||||
|
"requireV2AuditClosure": true,
|
||||||
|
"requireClaimChecks": true,
|
||||||
|
"requireComposerOutput": true,
|
||||||
|
"expectedGatekeeperStatuses": ["pass"],
|
||||||
|
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||||
|
"expectedComposerStatuses": ["valid"],
|
||||||
|
"forbiddenConfirmedClaimKeywords": ["数据库连接池"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "hikari-no-evidence-negative-observation",
|
||||||
|
"title": "Hikari no-evidence negative observation",
|
||||||
|
"question": "确认 inventory-service 当前是否有 HikariCP 连接池耗尽日志。",
|
||||||
|
"traceFixture": "hikari-no-evidence-negative-observation-pass.json",
|
||||||
|
"expectedRootCauseKeywords": ["未检索到", "HikariCP", "匹配证据"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["query_logs"],
|
||||||
|
"allowedVerdicts": ["PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["已排除", "确认没有", "日志层面已排除"],
|
||||||
|
"requireV2AuditClosure": true,
|
||||||
|
"requireClaimChecks": true,
|
||||||
|
"requireComposerOutput": true,
|
||||||
|
"expectedGatekeeperStatuses": ["pass"],
|
||||||
|
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||||
|
"expectedComposerStatuses": ["valid"],
|
||||||
|
"forbiddenConfirmedClaimKeywords": ["已排除 HikariCP"]
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"id": "payment-timeout",
|
"id": "payment-timeout",
|
||||||
"title": "Payment API timeout",
|
"title": "Payment API timeout",
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-hikari-no-evidence-negative-observation",
|
||||||
|
"query": "确认 inventory-service 当前是否有 HikariCP 连接池耗尽日志。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 21000,
|
||||||
|
"toolCallCount": 1,
|
||||||
|
"answer": "本次查询未检索到 inventory-service 的 HikariCP 连接池耗尽日志;这只表示当前查询没有匹配证据,仍不能据此判断系统一定健康。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "PASS",
|
||||||
|
"groundedness_score": 1.0,
|
||||||
|
"critical_fact_count": 1,
|
||||||
|
"gatekeeper_result": {
|
||||||
|
"status": "pass",
|
||||||
|
"severity": "none",
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"checked_bindings": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"source_invocation_id": 12,
|
||||||
|
"raw_path": "$.no_evidence",
|
||||||
|
"matched_text": "query_logs returned no evidence; evidence_status=no_evidence; query=inventory-service HikariCP; topic=application-logs; total=0; message=未找到匹配的日志",
|
||||||
|
"status": "pass"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"failed_rules": [],
|
||||||
|
"warnings": [],
|
||||||
|
"errors": []
|
||||||
|
},
|
||||||
|
"executor_structured_output": {
|
||||||
|
"answer_version": "executor_evidence_v2",
|
||||||
|
"claims": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"claim_type": "negative_observation",
|
||||||
|
"claim_text": "当前查询未检索到 inventory-service 的 HikariCP 连接池耗尽日志。",
|
||||||
|
"support_level": "direct",
|
||||||
|
"evidence_bindings": [
|
||||||
|
{
|
||||||
|
"source_type": "tool_trace",
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"source_invocation_id": 12,
|
||||||
|
"raw_path": "$.no_evidence",
|
||||||
|
"evidence_excerpt": "query_logs returned no evidence; query=inventory-service HikariCP; total=0; evidence_status=no_evidence"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"hypotheses": [],
|
||||||
|
"recommended_actions": [],
|
||||||
|
"missing_info": [
|
||||||
|
"仅查询了 application-logs 中 inventory-service HikariCP 相关日志"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"claim_checks": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"claim_text": "当前查询未检索到 inventory-service 的 HikariCP 连接池耗尽日志。",
|
||||||
|
"claim_type": "negative_observation",
|
||||||
|
"verification": "direct_observation",
|
||||||
|
"detail": "$.no_evidence 只支持本次查询未检索到匹配证据。",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"source_invocation_id": 12,
|
||||||
|
"raw_path": "$.no_evidence"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"facts_checked": [],
|
||||||
|
"composer_output": {
|
||||||
|
"status": "valid",
|
||||||
|
"answer_summary": "本次查询未检索到匹配日志。",
|
||||||
|
"recommended_actions": [],
|
||||||
|
"user_facing_answer": "本次查询未检索到 inventory-service 的 HikariCP 连接池耗尽日志;这只表示当前查询没有匹配证据,仍不能据此判断系统一定健康。"
|
||||||
|
},
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"source_invocation_ids": [12],
|
||||||
|
"evidence_level": "no_evidence"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 12,
|
||||||
|
"sessionId": "eval-hikari-no-evidence-negative-observation",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"outputPreview": "query_logs returned no evidence; evidence_status=no_evidence; query=inventory-service HikariCP; topic=application-logs; total=0; message=未找到匹配的日志",
|
||||||
|
"retrievalDetails": {
|
||||||
|
"evidence_status": "no_evidence",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"raw_path": "$.no_evidence",
|
||||||
|
"text": "query_logs returned no evidence; evidence_status=no_evidence; query=inventory-service HikariCP; topic=application-logs; total=0; message=未找到匹配的日志"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 1,
|
||||||
|
"returnedToolCallCount": 1,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-narrow-highcpu-observation",
|
||||||
|
"query": "确认 payment-service 当前是否存在 HighCPUUsage 告警。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 18000,
|
||||||
|
"toolCallCount": 1,
|
||||||
|
"answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "PASS",
|
||||||
|
"groundedness_score": 1.0,
|
||||||
|
"critical_fact_count": 1,
|
||||||
|
"gatekeeper_result": {
|
||||||
|
"status": "pass",
|
||||||
|
"severity": "none",
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"checked_bindings": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"source_invocation_id": 11,
|
||||||
|
"raw_path": "$.alerts[0]",
|
||||||
|
"matched_text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
|
||||||
|
"status": "pass"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"failed_rules": [],
|
||||||
|
"warnings": [],
|
||||||
|
"errors": []
|
||||||
|
},
|
||||||
|
"executor_structured_output": {
|
||||||
|
"answer_version": "executor_evidence_v2",
|
||||||
|
"claims": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"claim_type": "observation",
|
||||||
|
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||||
|
"support_level": "direct",
|
||||||
|
"evidence_bindings": [
|
||||||
|
{
|
||||||
|
"source_type": "tool_trace",
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"source_invocation_id": 11,
|
||||||
|
"raw_path": "$.alerts[0]",
|
||||||
|
"evidence_excerpt": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"hypotheses": [],
|
||||||
|
"recommended_actions": [],
|
||||||
|
"missing_info": []
|
||||||
|
},
|
||||||
|
"claim_checks": [
|
||||||
|
{
|
||||||
|
"claim_id": "claim-1",
|
||||||
|
"claim_text": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。",
|
||||||
|
"claim_type": "observation",
|
||||||
|
"verification": "direct_observation",
|
||||||
|
"detail": "已核验的指标证据直接包含服务名、告警名和 CPU 当前值。",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"source_invocation_id": 11,
|
||||||
|
"raw_path": "$.alerts[0]"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"facts_checked": [],
|
||||||
|
"composer_output": {
|
||||||
|
"status": "valid",
|
||||||
|
"answer_summary": "payment-service 当前存在 HighCPUUsage 告警。",
|
||||||
|
"recommended_actions": [],
|
||||||
|
"user_facing_answer": "payment-service 当前存在 HighCPUUsage 告警,CPU 使用率为 92%。"
|
||||||
|
},
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"success": true,
|
||||||
|
"source_invocation_ids": [11],
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 11,
|
||||||
|
"sessionId": "eval-narrow-highcpu-observation",
|
||||||
|
"toolName": "query_metrics",
|
||||||
|
"outputPreview": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m",
|
||||||
|
"retrievalDetails": {
|
||||||
|
"evidence_status": "supported",
|
||||||
|
"evidence_refs": [
|
||||||
|
{
|
||||||
|
"raw_path": "$.alerts[0]",
|
||||||
|
"text": "HighCPUUsage firing, service=payment-service, current=92%, duration=25m"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 1,
|
||||||
|
"returnedToolCallCount": 1,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,15 +1,49 @@
|
|||||||
{
|
{
|
||||||
"totalCases" : 8,
|
"totalCases" : 10,
|
||||||
"passedCases" : 8,
|
"passedCases" : 10,
|
||||||
"passRate" : 1.0,
|
"passRate" : 1.0,
|
||||||
"verdictDistribution" : {
|
"verdictDistribution" : {
|
||||||
"PASS" : 2,
|
"PASS" : 4,
|
||||||
"LOW_CONFID" : 5,
|
"LOW_CONFID" : 5,
|
||||||
"REJECT" : 1
|
"REJECT" : 1
|
||||||
},
|
},
|
||||||
"averageToolCallCount" : 1.625,
|
"averageToolCallCount" : 1.5,
|
||||||
"averageDurationMs" : 44875.0,
|
"averageDurationMs" : 39800.0,
|
||||||
"results" : [ {
|
"results" : [ {
|
||||||
|
"caseId" : "narrow-highcpu-observation",
|
||||||
|
"title" : "Narrow HighCPU observation",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "PASS",
|
||||||
|
"matchedKeywordCount" : 3,
|
||||||
|
"requiredKeywordCount" : 3,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_metrics" : true
|
||||||
|
},
|
||||||
|
"gatekeeperStatus" : "pass",
|
||||||
|
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||||
|
"composerStatus" : "valid",
|
||||||
|
"claimCheckCount" : 1,
|
||||||
|
"toolCallCount" : 1,
|
||||||
|
"durationMs" : 18000
|
||||||
|
}, {
|
||||||
|
"caseId" : "hikari-no-evidence-negative-observation",
|
||||||
|
"title" : "Hikari no-evidence negative observation",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "PASS",
|
||||||
|
"matchedKeywordCount" : 3,
|
||||||
|
"requiredKeywordCount" : 3,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_logs" : true
|
||||||
|
},
|
||||||
|
"gatekeeperStatus" : "pass",
|
||||||
|
"gatekeeperRuleSetVersion" : "gatekeeper-rules-v1",
|
||||||
|
"composerStatus" : "valid",
|
||||||
|
"claimCheckCount" : 1,
|
||||||
|
"toolCallCount" : 1,
|
||||||
|
"durationMs" : 21000
|
||||||
|
}, {
|
||||||
"caseId" : "payment-timeout",
|
"caseId" : "payment-timeout",
|
||||||
"title" : "Payment API timeout",
|
"title" : "Payment API timeout",
|
||||||
"passed" : true,
|
"passed" : true,
|
||||||
@@ -23,6 +57,7 @@
|
|||||||
"query_metrics" : true
|
"query_metrics" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
"toolCallCount" : 3,
|
"toolCallCount" : 3,
|
||||||
@@ -40,6 +75,7 @@
|
|||||||
"query_logs" : true
|
"query_logs" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
"toolCallCount" : 2,
|
"toolCallCount" : 2,
|
||||||
@@ -56,6 +92,7 @@
|
|||||||
"query_logs" : true
|
"query_logs" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
@@ -73,6 +110,7 @@
|
|||||||
"query_logs" : true
|
"query_logs" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
"toolCallCount" : 2,
|
"toolCallCount" : 2,
|
||||||
@@ -90,6 +128,7 @@
|
|||||||
"query_logs" : true
|
"query_logs" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : null,
|
"gatekeeperStatus" : null,
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : null,
|
"composerStatus" : null,
|
||||||
"claimCheckCount" : null,
|
"claimCheckCount" : null,
|
||||||
"toolCallCount" : 2,
|
"toolCallCount" : 2,
|
||||||
@@ -106,6 +145,7 @@
|
|||||||
"query_logs" : true
|
"query_logs" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : "fail",
|
"gatekeeperStatus" : "fail",
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : "valid",
|
"composerStatus" : "valid",
|
||||||
"claimCheckCount" : 1,
|
"claimCheckCount" : 1,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
@@ -122,6 +162,7 @@
|
|||||||
"query_logs" : true
|
"query_logs" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : "pass",
|
"gatekeeperStatus" : "pass",
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : "valid",
|
"composerStatus" : "valid",
|
||||||
"claimCheckCount" : 2,
|
"claimCheckCount" : 2,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
@@ -138,6 +179,7 @@
|
|||||||
"query_metrics" : true
|
"query_metrics" : true
|
||||||
},
|
},
|
||||||
"gatekeeperStatus" : "pass",
|
"gatekeeperStatus" : "pass",
|
||||||
|
"gatekeeperRuleSetVersion" : null,
|
||||||
"composerStatus" : "composer_malformed",
|
"composerStatus" : "composer_malformed",
|
||||||
"claimCheckCount" : 2,
|
"claimCheckCount" : 2,
|
||||||
"toolCallCount" : 1,
|
"toolCallCount" : 1,
|
||||||
|
|||||||
@@ -1,26 +1,28 @@
|
|||||||
# Diagnosis Eval Report
|
# Diagnosis Eval Report
|
||||||
|
|
||||||
- Total cases: 8
|
- Total cases: 10
|
||||||
- Passed cases: 8
|
- Passed cases: 10
|
||||||
- Pass rate: 100.00%
|
- Pass rate: 100.00%
|
||||||
- Average tool calls: 1.63
|
- Average tool calls: 1.50
|
||||||
- Average duration ms: 44875.00
|
- Average duration ms: 39800.00
|
||||||
|
|
||||||
## Verdict Distribution
|
## Verdict Distribution
|
||||||
|
|
||||||
- PASS: 2
|
- PASS: 4
|
||||||
- LOW_CONFID: 5
|
- LOW_CONFID: 5
|
||||||
- REJECT: 1
|
- REJECT: 1
|
||||||
|
|
||||||
## Cases
|
## Cases
|
||||||
|
|
||||||
| Case | Result | Verdict | Gatekeeper | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |
|
| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |
|
||||||
| --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |
|
| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |
|
||||||
| payment-timeout | PASS | PASS | - | - | - | 3/3 | 3 | 42000 | - |
|
| narrow-highcpu-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 18000 | - |
|
||||||
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | 3/3 | 2 | 51000 | - |
|
| hikari-no-evidence-negative-observation | PASS | PASS | pass | gatekeeper-rules-v1 | valid | 1 | 3/3 | 1 | 21000 | - |
|
||||||
| redis-timeout | PASS | LOW_CONFID | - | - | - | 2/2 | 1 | 36000 | - |
|
| payment-timeout | PASS | PASS | - | - | - | - | 3/3 | 3 | 42000 | - |
|
||||||
| slow-response | PASS | PASS | - | - | - | 2/2 | 2 | 47000 | - |
|
| mysql-pool-exhausted | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 51000 | - |
|
||||||
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | 3/3 | 2 | 53000 | - |
|
| redis-timeout | PASS | LOW_CONFID | - | - | - | - | 2/2 | 1 | 36000 | - |
|
||||||
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | valid | 1 | 3/3 | 1 | 39000 | - |
|
| slow-response | PASS | PASS | - | - | - | - | 2/2 | 2 | 47000 | - |
|
||||||
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | valid | 2 | 2/2 | 1 | 44000 | - |
|
| jvm-memory-risk | PASS | LOW_CONFID | - | - | - | - | 3/3 | 2 | 53000 | - |
|
||||||
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | composer_malformed | 2 | 2/2 | 1 | 47000 | - |
|
| gatekeeper-fabricated-invocation | PASS | REJECT | fail | - | valid | 1 | 3/3 | 1 | 39000 | - |
|
||||||
|
| unsupported-claim-filtering | PASS | LOW_CONFID | pass | - | valid | 2 | 2/2 | 1 | 44000 | - |
|
||||||
|
| composer-fallback-no-raw-json | PASS | LOW_CONFID | pass | - | composer_malformed | 2 | 2/2 | 1 | 47000 | - |
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ baseline report:整套固定集当前认可的结果
|
|||||||
"requireClaimChecks": true,
|
"requireClaimChecks": true,
|
||||||
"requireComposerOutput": true,
|
"requireComposerOutput": true,
|
||||||
"expectedGatekeeperStatuses": ["pass"],
|
"expectedGatekeeperStatuses": ["pass"],
|
||||||
|
"expectedGatekeeperRuleSetVersion": "gatekeeper-rules-v1",
|
||||||
"expectedComposerStatuses": ["valid"],
|
"expectedComposerStatuses": ["valid"],
|
||||||
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
"forbiddenConfirmedClaimKeywords": ["主库故障"]
|
||||||
}
|
}
|
||||||
@@ -54,6 +55,7 @@ baseline report:整套固定集当前认可的结果
|
|||||||
| `requireClaimChecks` | 是否要求 `claim_checks` | 要求 claim check 数组存在且非空 |
|
| `requireClaimChecks` | 是否要求 `claim_checks` | 要求 claim check 数组存在且非空 |
|
||||||
| `requireComposerOutput` | 是否要求 `composer_output` | 要求 Composer 审计存在并带 `status` |
|
| `requireComposerOutput` | 是否要求 `composer_output` | 要求 Composer 审计存在并带 `status` |
|
||||||
| `expectedGatekeeperStatuses` | 允许的 Gatekeeper 状态 | 实际 `gatekeeper_result.status` 不在列表中则失败 |
|
| `expectedGatekeeperStatuses` | 允许的 Gatekeeper 状态 | 实际 `gatekeeper_result.status` 不在列表中则失败 |
|
||||||
|
| `expectedGatekeeperRuleSetVersion` | 期望的 Gatekeeper 规则集版本 | 配置后校验 `gatekeeper_result.rule_set_version` |
|
||||||
| `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 |
|
| `expectedComposerStatuses` | 允许的 Composer 状态 | 实际 `composer_output.status` 不在列表中则失败 |
|
||||||
| `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 |
|
| `forbiddenConfirmedClaimKeywords` | 不得进入最终答案的未支持结论关键词 | 用于证明 unsupported/external_unknown claim 被过滤 |
|
||||||
|
|
||||||
@@ -69,6 +71,7 @@ fixture 是一次 Agent 运行后的 trace 快照。评测器只读取当前规
|
|||||||
| `session.totalDurationMs` | 运行耗时 | 进入报告 |
|
| `session.totalDurationMs` | 运行耗时 | 进入报告 |
|
||||||
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` |
|
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 判定 | 必须存在并符合 case 的 `allowedVerdicts` |
|
||||||
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` |
|
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.status` | Gatekeeper 结果 | V2 case 必须存在;`fail` 不允许搭配 `PASS` |
|
||||||
|
| `session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | Gatekeeper 规则集版本 | 新矩阵 case 可显式断言该版本 |
|
||||||
| `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` |
|
| `session.selfEvaluation.verifier_evaluation.claim_checks` | Verifier V2 claim 级校验 | V2 case 必须存在;每项需要 `claim_id`、`verification`、`detail` |
|
||||||
| `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 |
|
| `session.selfEvaluation.verifier_evaluation.composer_output.status` | Composer 渲染状态 | V2 case 必须存在;记录 `valid`、`composer_malformed` 等 |
|
||||||
| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 |
|
| `session.selfEvaluation.verifier_evaluation.executor_structured_output.claims[*].evidence_bindings` | Executor claim 证据绑定 | 如果结构化输出存在,每条 claim 需要证据绑定 |
|
||||||
@@ -91,6 +94,7 @@ Java 类型:`DiagnosisEvalResult`
|
|||||||
| `requiredKeywordCount` | case 配置的关键词数量 |
|
| `requiredKeywordCount` | case 配置的关键词数量 |
|
||||||
| `evidenceCoverage` | 每个必需工具是否出现 |
|
| `evidenceCoverage` | 每个必需工具是否出现 |
|
||||||
| `gatekeeperStatus` | 读到的 `gatekeeper_result.status` |
|
| `gatekeeperStatus` | 读到的 `gatekeeper_result.status` |
|
||||||
|
| `gatekeeperRuleSetVersion` | 读到的 `gatekeeper_result.rule_set_version` |
|
||||||
| `composerStatus` | 读到的 `composer_output.status` |
|
| `composerStatus` | 读到的 `composer_output.status` |
|
||||||
| `claimCheckCount` | `claim_checks` 数量 |
|
| `claimCheckCount` | `claim_checks` 数量 |
|
||||||
| `toolCallCount` | trace 中工具调用总数 |
|
| `toolCallCount` | trace 中工具调用总数 |
|
||||||
@@ -125,6 +129,7 @@ Executor structured output
|
|||||||
|
|
||||||
- V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。
|
- V2 case 必须有 `gatekeeper_result`、`claim_checks`、`composer_output`。
|
||||||
- `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。
|
- `gatekeeper_result.status = fail` 时,Verifier verdict 不能是 `PASS`。
|
||||||
|
- 配置 `expectedGatekeeperRuleSetVersion` 的 case 必须匹配 `gatekeeper_result.rule_set_version`。
|
||||||
- `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。
|
- `claim_checks[*].verification` 只能是 `direct_observation`、`reasonable_inference`、`overstated`、`unsupported`、`external_unknown`、`contradicted`。
|
||||||
- Composer 输出必须记录 `status`。
|
- Composer 输出必须记录 `status`。
|
||||||
- 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。
|
- 最终答案不能泄漏 `executor_evidence_v2`、`answer_version`、`evidence_bindings`、`claim_id`。
|
||||||
|
|||||||
+1
@@ -0,0 +1 @@
|
|||||||
|
archive-ready
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
committed
|
||||||
@@ -0,0 +1,110 @@
|
|||||||
|
# Design
|
||||||
|
|
||||||
|
## Data Flow
|
||||||
|
|
||||||
|
```text
|
||||||
|
Gatekeeper rule catalog
|
||||||
|
-> ExecutorGatekeeperService.validate(...)
|
||||||
|
-> gatekeeper_result.rule_set_version + rule metadata summary
|
||||||
|
-> DiagnosisTraceEvaluator fixture checks
|
||||||
|
-> baseline report / demo runbook
|
||||||
|
```
|
||||||
|
|
||||||
|
```text
|
||||||
|
Stable demo scenarios
|
||||||
|
-> request payloads and docs
|
||||||
|
-> optional live run for main path
|
||||||
|
-> saved trace fixtures for deterministic matrix
|
||||||
|
-> mvp/eval baseline report
|
||||||
|
```
|
||||||
|
|
||||||
|
## Eval Matrix
|
||||||
|
|
||||||
|
The diagnosis eval matrix remains offline and deterministic. It should cover these rows:
|
||||||
|
|
||||||
|
| Matrix row | Expected signal |
|
||||||
|
|---|---|
|
||||||
|
| Positive supported evidence | `PASS`, Gatekeeper `pass`, Composer `valid` |
|
||||||
|
| Narrow-scope observation | `PASS`, one or minimal claims, no forbidden over-expansion |
|
||||||
|
| No-evidence negative observation | `PASS` or allowed non-reject verdict, `$.no_evidence`, no overstatement |
|
||||||
|
| Unsupported claim filtering | `LOW_CONFID`, unsupported claim not in final answer |
|
||||||
|
| Gatekeeper fabricated reference | `REJECT` or `LOW_CONFID`, Gatekeeper `fail` |
|
||||||
|
| Composer fallback | no raw Executor protocol leakage |
|
||||||
|
|
||||||
|
The evaluator should validate rule set version only for cases that opt into the new check. This keeps old fixtures readable while allowing the new matrix to prove Gatekeeper metadata persistence.
|
||||||
|
|
||||||
|
## Stable Demo Scenarios
|
||||||
|
|
||||||
|
Stable demo scenarios are source-controlled payloads and documentation, not a live-only test harness. The demo set should include:
|
||||||
|
|
||||||
|
- A supported positive path.
|
||||||
|
- A no-evidence / negative-observation path.
|
||||||
|
- A safety/reject path explained through fixed fixture or evaluator output.
|
||||||
|
|
||||||
|
Only the main path needs a live script in this phase. Other scenarios may be represented by payloads, fixture names, and expected trace fields.
|
||||||
|
|
||||||
|
## Gatekeeper Rule Catalog
|
||||||
|
|
||||||
|
Gatekeeper rules stay local and lightweight:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"default_severity": "reject",
|
||||||
|
"enabled": true
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The first implementation may use an in-memory default catalog or a classpath JSON resource. It must expose:
|
||||||
|
|
||||||
|
- rule set version
|
||||||
|
- enabled rule ids
|
||||||
|
- rule descriptions
|
||||||
|
- severity defaults or threshold parameters when present
|
||||||
|
|
||||||
|
Gatekeeper validation logic remains deterministic Java code. The catalog is metadata/config, not a dynamic scripting engine.
|
||||||
|
|
||||||
|
## Audit Persistence
|
||||||
|
|
||||||
|
`gatekeeper_result` should include:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"rule_set_version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Existing fields remain:
|
||||||
|
|
||||||
|
- `status`
|
||||||
|
- `severity`
|
||||||
|
- `checked_bindings`
|
||||||
|
- `failed_rules`
|
||||||
|
- `warnings`
|
||||||
|
- `errors`
|
||||||
|
|
||||||
|
## Interface Impact
|
||||||
|
|
||||||
|
Impact level: L2 internal contract change.
|
||||||
|
|
||||||
|
This expands internal audit JSON and eval case/result fields. It does not change external HTTP APIs, database schema, Planner output, or public DTO contracts.
|
||||||
|
|
||||||
|
## Risks
|
||||||
|
|
||||||
|
- Too much Gatekeeper flexibility could weaken safety. This phase only adds metadata/configuration and keeps rule implementations fixed in code.
|
||||||
|
- Demo scenarios should not promise deterministic LLM behavior. Deterministic claims should point to fixture-backed eval results.
|
||||||
|
- Baseline updates must be made together with new fixtures and tests.
|
||||||
+66
@@ -0,0 +1,66 @@
|
|||||||
|
# diagnosis-eval-demo-gatekeeper-closure
|
||||||
|
|
||||||
|
## Problem
|
||||||
|
|
||||||
|
The Chat evidence pipeline now has Executor V2 structured output, deterministic Gatekeeper validation, Verifier claim checks, and Composer final rendering. The runtime path is stronger than the project-level acceptance story around it.
|
||||||
|
|
||||||
|
Current gaps:
|
||||||
|
|
||||||
|
- Diagnosis eval fixtures cover several V2 audit behaviors, but they do not yet form an explicit interview-ready matrix for positive evidence, no-evidence, reject, low-confidence, narrow-scope, and raw-output leakage.
|
||||||
|
- Demo assets are still centered on the original payment-timeout path. They do not clearly package the stable PASS / LOW_CONFID / REJECT examples needed for an Agent engineering interview.
|
||||||
|
- Gatekeeper rules are hard-coded constants and thresholds. Audit output identifies failed rules, but it does not expose a rule set version or rule metadata that can be discussed, tested, and evolved.
|
||||||
|
|
||||||
|
## Proposed Change
|
||||||
|
|
||||||
|
Create an acceptance closure layer around the existing evidence pipeline:
|
||||||
|
|
||||||
|
```text
|
||||||
|
stable demo scenarios
|
||||||
|
-> saved offline diagnosis fixtures
|
||||||
|
-> deterministic diagnosis eval matrix
|
||||||
|
-> Gatekeeper rule metadata/version
|
||||||
|
-> persisted audit result that records the rule set version
|
||||||
|
```
|
||||||
|
|
||||||
|
This change keeps the current runtime architecture. It does not introduce new Agents or new database tables. It strengthens the project as an interview-ready Agent engineering artifact by making the anti-hallucination behavior demonstrable and regressable.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Expand `mvp/eval` case definitions and trace fixtures into an explicit evidence-pipeline matrix.
|
||||||
|
- Add or update baseline reports so the fixed matrix remains deterministic.
|
||||||
|
- Add stable demo request payloads and runbook docs for PASS, LOW_CONFID / no-evidence, and REJECT / fabricated-reference style scenarios.
|
||||||
|
- Add Gatekeeper rule metadata/configuration with a small rule set version.
|
||||||
|
- Include Gatekeeper rule set version and loaded rule metadata summary in `gatekeeper_result`.
|
||||||
|
- Add focused tests for eval matrix behavior and Gatekeeper rule metadata/audit version.
|
||||||
|
- Update architecture/demo/eval docs as needed.
|
||||||
|
|
||||||
|
## Non-goals
|
||||||
|
|
||||||
|
- No new public HTTP endpoint.
|
||||||
|
- No new database table.
|
||||||
|
- No Planner `scope_contract`.
|
||||||
|
- No retry loop from Gatekeeper back to Executor.
|
||||||
|
- No full JSONPath engine.
|
||||||
|
- No LLM-as-judge.
|
||||||
|
- No production-grade remote rule registry.
|
||||||
|
- No automatic prompt optimization.
|
||||||
|
|
||||||
|
## Context Constraints
|
||||||
|
|
||||||
|
- Diagnosis eval should remain offline and deterministic.
|
||||||
|
- Stable demo data should reuse existing mock tools and `knowledge_base` where possible.
|
||||||
|
- Gatekeeper audit should remain under `diagnosis_session.self_evaluation.verifier_evaluation.gatekeeper_result`.
|
||||||
|
- Gatekeeper rules should stay lightweight: rule id, description, severity, enabled flag, and config parameters are enough for this phase.
|
||||||
|
- `$.no_evidence` means only "this tool query returned no matching evidence"; it must not become a confirmed absence of the underlying problem.
|
||||||
|
|
||||||
|
## Assumptions
|
||||||
|
|
||||||
|
- This is a `standard` sm-flow change because it touches eval assets, demo assets, Gatekeeper internals, tests, and docs.
|
||||||
|
- Interface impact is expected to be L2 internal contract change: internal JSON audit fields expand, but no external HTTP contract or database schema changes.
|
||||||
|
- Existing V2 audit fixtures and ISS-008 / ISS-009 validation records are acceptable seeds for the matrix.
|
||||||
|
|
||||||
|
## Risks
|
||||||
|
|
||||||
|
- If demo scenarios depend on live LLM output, they may still be nondeterministic. The fixed eval matrix must use saved fixtures.
|
||||||
|
- If Gatekeeper config becomes too flexible, it could obscure deterministic safety rules. This phase should only expose metadata and simple parameters.
|
||||||
|
- Baseline reports must be updated together with case/fixture changes, or the evaluator tests will become noisy.
|
||||||
+26
@@ -0,0 +1,26 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Gatekeeper SHALL expose rule catalog metadata in audit output
|
||||||
|
Gatekeeper SHALL include the rule catalog version and enabled rule metadata in its validation result.
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper pass includes rule metadata
|
||||||
|
- **WHEN** Gatekeeper returns `status=pass`
|
||||||
|
- **THEN** the result SHALL include `rule_set_version`
|
||||||
|
- **AND** it SHALL include a rule metadata summary
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper fail includes rule metadata
|
||||||
|
- **WHEN** Gatekeeper returns `status=fail`
|
||||||
|
- **THEN** the result SHALL include `rule_set_version`
|
||||||
|
- **AND** it SHALL include a rule metadata summary
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper fallback pass includes rule metadata
|
||||||
|
- **WHEN** the Verifier input hook returns a fallback Gatekeeper pass because no Gatekeeper service is available
|
||||||
|
- **THEN** the result SHOULD still include the default rule set version and an empty or default rule metadata summary
|
||||||
|
|
||||||
|
### Requirement: Gatekeeper rule catalog SHALL remain deterministic
|
||||||
|
The Gatekeeper rule catalog SHALL configure metadata and simple parameters only; validation behavior SHALL remain deterministic Java code.
|
||||||
|
|
||||||
|
#### Scenario: Rule metadata is lightweight
|
||||||
|
- **WHEN** rule metadata is loaded
|
||||||
|
- **THEN** each enabled rule SHOULD expose an id, description, enabled flag, and default severity or relevant parameter
|
||||||
|
- **AND** rule metadata SHALL NOT execute dynamic scripts
|
||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL expose an evidence-pipeline matrix
|
||||||
|
The evaluation harness SHALL include fixed cases that demonstrate the current Chat evidence pipeline across positive evidence, narrow-scope observation, no-evidence negative observation, low-confidence filtering, reject safety, and composer fallback behavior.
|
||||||
|
|
||||||
|
#### Scenario: Matrix cases are fixture backed
|
||||||
|
- **WHEN** the fixed diagnosis case file is evaluated
|
||||||
|
- **THEN** each matrix case SHALL resolve to an offline trace fixture
|
||||||
|
- **AND** evaluation SHALL not require a live LLM or running application
|
||||||
|
|
||||||
|
#### Scenario: Matrix cases preserve V2 audit closure
|
||||||
|
- **WHEN** a matrix case requires V2 audit closure
|
||||||
|
- **THEN** its fixture SHALL include `gatekeeper_result`
|
||||||
|
- **AND** it SHALL include `claim_checks`
|
||||||
|
- **AND** it SHALL include `composer_output`
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL validate Gatekeeper rule set version when requested
|
||||||
|
The evaluation harness SHALL be able to assert the Gatekeeper rule set version recorded in a fixture.
|
||||||
|
|
||||||
|
#### Scenario: Expected rule set version matches
|
||||||
|
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
|
||||||
|
- **AND** the fixture has the same `selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version`
|
||||||
|
- **THEN** the rule set version check SHALL pass
|
||||||
|
|
||||||
|
#### Scenario: Expected rule set version mismatches
|
||||||
|
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
|
||||||
|
- **AND** the fixture has a different or missing rule set version
|
||||||
|
- **THEN** the case SHALL fail with a clear failed check
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide stable evidence-pipeline scenarios
|
||||||
|
The MVP demo SHALL provide stable scenarios that explain how to demonstrate positive evidence, no-evidence, and safety/reject behavior for an Agent engineering interview.
|
||||||
|
|
||||||
|
#### Scenario: Scenario guide maps demo inputs to evidence claims
|
||||||
|
- **WHEN** a reviewer opens the demo scenario guide
|
||||||
|
- **THEN** it SHALL list the supported positive, no-evidence, and safety/reject scenarios
|
||||||
|
- **AND** it SHALL map each scenario to a request payload or fixture id
|
||||||
|
- **AND** it SHALL describe the expected Gatekeeper, Verifier, Composer, and trace fields to inspect
|
||||||
|
|
||||||
|
#### Scenario: Live and fixture-backed scenarios are distinguished
|
||||||
|
- **WHEN** a demo scenario is fixture-backed rather than live-scripted
|
||||||
|
- **THEN** the documentation SHALL say so explicitly
|
||||||
|
- **AND** it SHALL avoid promising deterministic live LLM output for that scenario
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
# Tasks
|
||||||
|
|
||||||
|
## 1. Diagnosis eval matrix
|
||||||
|
|
||||||
|
- [x] Add matrix-oriented eval cases for narrow-scope supported evidence and no-evidence negative observation.
|
||||||
|
- [x] Add matching offline trace fixtures with V2 audit closure fields.
|
||||||
|
- [x] Extend evaluator case/result model to optionally check `gatekeeper_result.rule_set_version`.
|
||||||
|
- [x] Regenerate baseline JSON and Markdown reports.
|
||||||
|
- [x] Update eval README/schema to describe the matrix and rule set version check.
|
||||||
|
|
||||||
|
Acceptance:
|
||||||
|
|
||||||
|
- `DiagnosisTraceEvaluatorTest` passes with the new total case count and verdict distribution.
|
||||||
|
- Every fixed case resolves to an existing fixture.
|
||||||
|
- V2 matrix cases include Gatekeeper, claim checks, Composer output, and final-answer leakage checks where relevant.
|
||||||
|
|
||||||
|
## 2. Stable demo scenarios
|
||||||
|
|
||||||
|
- [x] Add stable demo request payloads for supported positive, no-evidence, and safety/reject discussion scenarios.
|
||||||
|
- [x] Add a demo scenario guide that maps each payload or fixture to interview claims and expected trace fields.
|
||||||
|
- [x] Update existing demo README/runbook references so reviewers know which path is live and which paths are fixture-backed.
|
||||||
|
|
||||||
|
Acceptance:
|
||||||
|
|
||||||
|
- Demo docs clearly distinguish live script path from deterministic fixture-backed scenarios.
|
||||||
|
- Each scenario has a stable session id or fixture id.
|
||||||
|
- No demo doc claims unsupported production behavior.
|
||||||
|
|
||||||
|
## 3. Gatekeeper rule catalog and audit version
|
||||||
|
|
||||||
|
- [x] Add a lightweight Gatekeeper rule catalog with version and rule metadata.
|
||||||
|
- [x] Include `rule_set_version` and enabled rule metadata summary in every Gatekeeper result, including pass/fallback/internal-error results.
|
||||||
|
- [x] Add tests for catalog loading and audit fields.
|
||||||
|
- [x] Keep validation logic deterministic; do not add dynamic script execution or remote config.
|
||||||
|
|
||||||
|
Acceptance:
|
||||||
|
|
||||||
|
- `ExecutorGatekeeperServiceTest` proves Gatekeeper output contains the rule set version and rule metadata.
|
||||||
|
- Existing Gatekeeper pass/low_confid/reject behavior remains unchanged.
|
||||||
|
|
||||||
|
## 4. Verification
|
||||||
|
|
||||||
|
- [x] Run focused tests for eval and Gatekeeper changes.
|
||||||
|
- [x] Run broader relevant regression tests if focused changes touch shared code.
|
||||||
|
- [x] Run at least one live end-to-end check if unit/fixture evidence is insufficient to prove demo path compatibility.
|
||||||
|
- [x] Record verification commands and results in devflow acceptance.
|
||||||
|
|
||||||
|
Acceptance:
|
||||||
|
|
||||||
|
- All required tests pass, or failures are classified and fixed before archive.
|
||||||
|
- If live E2E is skipped, the reason is documented and fixture coverage must prove the requested behavior.
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
# chat-verifier-agent Specification
|
# chat-verifier-agent Specification
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
TBD - created by archiving change chat-verifier-agent. Update Purpose after archive.
|
TBD - created by archiving change chat-verifier-agent. Update Purpose after archive.
|
||||||
@@ -531,3 +531,28 @@ The Executor prompt SHALL instruct Executor to keep narrow confirmation question
|
|||||||
- **THEN** Executor SHOULD output the minimum necessary claims, normally one and at most two
|
- **THEN** Executor SHOULD output the minimum necessary claims, normally one and at most two
|
||||||
- **AND** those claims SHALL be `observation` or `negative_observation` unless current-session evidence proves more
|
- **AND** those claims SHALL be `observation` or `negative_observation` unless current-session evidence proves more
|
||||||
- **AND** Executor SHALL NOT emit unrelated root-cause, remediation, or excluded-topic claims as confirmed facts
|
- **AND** Executor SHALL NOT emit unrelated root-cause, remediation, or excluded-topic claims as confirmed facts
|
||||||
|
|
||||||
|
### Requirement: Gatekeeper SHALL expose rule catalog metadata in audit output
|
||||||
|
Gatekeeper SHALL include the rule catalog version and enabled rule metadata in its validation result.
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper pass includes rule metadata
|
||||||
|
- **WHEN** Gatekeeper returns `status=pass`
|
||||||
|
- **THEN** the result SHALL include `rule_set_version`
|
||||||
|
- **AND** it SHALL include a rule metadata summary
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper fail includes rule metadata
|
||||||
|
- **WHEN** Gatekeeper returns `status=fail`
|
||||||
|
- **THEN** the result SHALL include `rule_set_version`
|
||||||
|
- **AND** it SHALL include a rule metadata summary
|
||||||
|
|
||||||
|
#### Scenario: Gatekeeper fallback pass includes rule metadata
|
||||||
|
- **WHEN** the Verifier input hook returns a fallback Gatekeeper pass because no Gatekeeper service is available
|
||||||
|
- **THEN** the result SHOULD still include the default rule set version and an empty or default rule metadata summary
|
||||||
|
|
||||||
|
### Requirement: Gatekeeper rule catalog SHALL remain deterministic
|
||||||
|
The Gatekeeper rule catalog SHALL configure metadata and simple parameters only; validation behavior SHALL remain deterministic Java code.
|
||||||
|
|
||||||
|
#### Scenario: Rule metadata is lightweight
|
||||||
|
- **WHEN** rule metadata is loaded
|
||||||
|
- **THEN** each enabled rule SHOULD expose an id, description, enabled flag, and default severity or relevant parameter
|
||||||
|
- **AND** rule metadata SHALL NOT execute dynamic scripts
|
||||||
|
|||||||
@@ -2,9 +2,7 @@
|
|||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Provide a repeatable offline evaluation harness for MVP diagnosis Agent behavior, so prompt, tool, retrieval, and verifier changes can be checked against fixed trace-based regression cases.
|
Provide a repeatable offline evaluation harness for MVP diagnosis Agent behavior, so prompt, tool, retrieval, and verifier changes can be checked against fixed trace-based regression cases.
|
||||||
|
|
||||||
## Requirements
|
## Requirements
|
||||||
|
|
||||||
### Requirement: Evaluation harness SHALL define fixed diagnosis cases
|
### Requirement: Evaluation harness SHALL define fixed diagnosis cases
|
||||||
The system SHALL provide a small fixed set of MVP diagnosis evaluation cases with explicit expected trace and answer criteria.
|
The system SHALL provide a small fixed set of MVP diagnosis evaluation cases with explicit expected trace and answer criteria.
|
||||||
|
|
||||||
@@ -170,3 +168,30 @@ The evaluation harness SHALL detect configured unsafe or unsupported claim text
|
|||||||
#### Scenario: Composer fallback still avoids raw JSON leakage
|
#### Scenario: Composer fallback still avoids raw JSON leakage
|
||||||
- **WHEN** a trace records Composer fallback rendering
|
- **WHEN** a trace records Composer fallback rendering
|
||||||
- **THEN** the evaluator SHALL still enforce final-answer raw JSON leakage checks
|
- **THEN** the evaluator SHALL still enforce final-answer raw JSON leakage checks
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL expose an evidence-pipeline matrix
|
||||||
|
The evaluation harness SHALL include fixed cases that demonstrate the current Chat evidence pipeline across positive evidence, narrow-scope observation, no-evidence negative observation, low-confidence filtering, reject safety, and composer fallback behavior.
|
||||||
|
|
||||||
|
#### Scenario: Matrix cases are fixture backed
|
||||||
|
- **WHEN** the fixed diagnosis case file is evaluated
|
||||||
|
- **THEN** each matrix case SHALL resolve to an offline trace fixture
|
||||||
|
- **AND** evaluation SHALL not require a live LLM or running application
|
||||||
|
|
||||||
|
#### Scenario: Matrix cases preserve V2 audit closure
|
||||||
|
- **WHEN** a matrix case requires V2 audit closure
|
||||||
|
- **THEN** its fixture SHALL include `gatekeeper_result`
|
||||||
|
- **AND** it SHALL include `claim_checks`
|
||||||
|
- **AND** it SHALL include `composer_output`
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL validate Gatekeeper rule set version when requested
|
||||||
|
The evaluation harness SHALL be able to assert the Gatekeeper rule set version recorded in a fixture.
|
||||||
|
|
||||||
|
#### Scenario: Expected rule set version matches
|
||||||
|
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
|
||||||
|
- **AND** the fixture has the same `selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version`
|
||||||
|
- **THEN** the rule set version check SHALL pass
|
||||||
|
|
||||||
|
#### Scenario: Expected rule set version mismatches
|
||||||
|
- **WHEN** an evaluation case declares `expectedGatekeeperRuleSetVersion`
|
||||||
|
- **AND** the fixture has a different or missing rule set version
|
||||||
|
- **THEN** the case SHALL fail with a clear failed check
|
||||||
|
|||||||
@@ -98,3 +98,17 @@ without requiring raw JSON inspection first.
|
|||||||
planner `read_skill` text mentions, executor `read_skill` text mentions, and
|
planner `read_skill` text mentions, executor `read_skill` text mentions, and
|
||||||
verifier `read_skill` text mentions
|
verifier `read_skill` text mentions
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide stable evidence-pipeline scenarios
|
||||||
|
The MVP demo SHALL provide stable scenarios that explain how to demonstrate positive evidence, no-evidence, and safety/reject behavior for an Agent engineering interview.
|
||||||
|
|
||||||
|
#### Scenario: Scenario guide maps demo inputs to evidence claims
|
||||||
|
- **WHEN** a reviewer opens the demo scenario guide
|
||||||
|
- **THEN** it SHALL list the supported positive, no-evidence, and safety/reject scenarios
|
||||||
|
- **AND** it SHALL map each scenario to a request payload or fixture id
|
||||||
|
- **AND** it SHALL describe the expected Gatekeeper, Verifier, Composer, and trace fields to inspect
|
||||||
|
|
||||||
|
#### Scenario: Live and fixture-backed scenarios are distinguished
|
||||||
|
- **WHEN** a demo scenario is fixture-backed rather than live-scripted
|
||||||
|
- **THEN** the documentation SHALL say so explicitly
|
||||||
|
- **AND** it SHALL avoid promising deterministic live LLM output for that scenario
|
||||||
|
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ public class DiagnosisEvalCase {
|
|||||||
private Boolean requireClaimChecks;
|
private Boolean requireClaimChecks;
|
||||||
private Boolean requireComposerOutput;
|
private Boolean requireComposerOutput;
|
||||||
private List<String> expectedGatekeeperStatuses;
|
private List<String> expectedGatekeeperStatuses;
|
||||||
|
private String expectedGatekeeperRuleSetVersion;
|
||||||
private List<String> expectedComposerStatuses;
|
private List<String> expectedComposerStatuses;
|
||||||
private List<String> forbiddenConfirmedClaimKeywords;
|
private List<String> forbiddenConfirmedClaimKeywords;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -46,8 +46,8 @@ public class DiagnosisEvalReportWriter {
|
|||||||
}
|
}
|
||||||
|
|
||||||
builder.append("## Cases\n\n");
|
builder.append("## Cases\n\n");
|
||||||
builder.append("| Case | Result | Verdict | Gatekeeper | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
|
builder.append("| Case | Result | Verdict | Gatekeeper | Rule Set | Composer | Claim Checks | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
|
||||||
builder.append("| --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |\n");
|
builder.append("| --- | --- | --- | --- | --- | --- | ---: | --- | ---: | ---: | --- |\n");
|
||||||
for (DiagnosisEvalResult result : report.getResults()) {
|
for (DiagnosisEvalResult result : report.getResults()) {
|
||||||
builder.append("| ")
|
builder.append("| ")
|
||||||
.append(result.getCaseId())
|
.append(result.getCaseId())
|
||||||
@@ -58,6 +58,8 @@ public class DiagnosisEvalReportWriter {
|
|||||||
.append(" | ")
|
.append(" | ")
|
||||||
.append(valueOrDash(result.getGatekeeperStatus()))
|
.append(valueOrDash(result.getGatekeeperStatus()))
|
||||||
.append(" | ")
|
.append(" | ")
|
||||||
|
.append(valueOrDash(result.getGatekeeperRuleSetVersion()))
|
||||||
|
.append(" | ")
|
||||||
.append(valueOrDash(result.getComposerStatus()))
|
.append(valueOrDash(result.getComposerStatus()))
|
||||||
.append(" | ")
|
.append(" | ")
|
||||||
.append(result.getClaimCheckCount() == null ? "-" : result.getClaimCheckCount())
|
.append(result.getClaimCheckCount() == null ? "-" : result.getClaimCheckCount())
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ public class DiagnosisEvalResult {
|
|||||||
private int requiredKeywordCount;
|
private int requiredKeywordCount;
|
||||||
private Map<String, Boolean> evidenceCoverage;
|
private Map<String, Boolean> evidenceCoverage;
|
||||||
private String gatekeeperStatus;
|
private String gatekeeperStatus;
|
||||||
|
private String gatekeeperRuleSetVersion;
|
||||||
private String composerStatus;
|
private String composerStatus;
|
||||||
private Integer claimCheckCount;
|
private Integer claimCheckCount;
|
||||||
private Integer toolCallCount;
|
private Integer toolCallCount;
|
||||||
|
|||||||
@@ -66,6 +66,7 @@ public class DiagnosisTraceEvaluator {
|
|||||||
.requiredKeywordCount(size(evalCase.getExpectedRootCauseKeywords()))
|
.requiredKeywordCount(size(evalCase.getExpectedRootCauseKeywords()))
|
||||||
.evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools()))
|
.evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools()))
|
||||||
.gatekeeperStatus(null)
|
.gatekeeperStatus(null)
|
||||||
|
.gatekeeperRuleSetVersion(null)
|
||||||
.composerStatus(null)
|
.composerStatus(null)
|
||||||
.claimCheckCount(null)
|
.claimCheckCount(null)
|
||||||
.toolCallCount(null)
|
.toolCallCount(null)
|
||||||
@@ -120,10 +121,12 @@ public class DiagnosisTraceEvaluator {
|
|||||||
|
|
||||||
failedChecks.addAll(validateExecutorStructuredOutput(trace));
|
failedChecks.addAll(validateExecutorStructuredOutput(trace));
|
||||||
String gatekeeperStatus = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "status");
|
String gatekeeperStatus = extractNestedString(trace, "verifier_evaluation", "gatekeeper_result", "status");
|
||||||
|
String gatekeeperRuleSetVersion = extractNestedString(trace, "verifier_evaluation",
|
||||||
|
"gatekeeper_result", "rule_set_version");
|
||||||
String composerStatus = extractNestedString(trace, "verifier_evaluation", "composer_output", "status");
|
String composerStatus = extractNestedString(trace, "verifier_evaluation", "composer_output", "status");
|
||||||
Integer claimCheckCount = countList(trace, "verifier_evaluation", "claim_checks");
|
Integer claimCheckCount = countList(trace, "verifier_evaluation", "claim_checks");
|
||||||
failedChecks.addAll(validateV2AuditClosure(evalCase, trace, normalizedAnswer, verdict,
|
failedChecks.addAll(validateV2AuditClosure(evalCase, trace, normalizedAnswer, verdict,
|
||||||
gatekeeperStatus, composerStatus));
|
gatekeeperStatus, gatekeeperRuleSetVersion, composerStatus));
|
||||||
|
|
||||||
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
|
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
|
||||||
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
|
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
|
||||||
@@ -138,6 +141,7 @@ public class DiagnosisTraceEvaluator {
|
|||||||
.requiredKeywordCount(requiredKeywordCount)
|
.requiredKeywordCount(requiredKeywordCount)
|
||||||
.evidenceCoverage(evidenceCoverage)
|
.evidenceCoverage(evidenceCoverage)
|
||||||
.gatekeeperStatus(gatekeeperStatus)
|
.gatekeeperStatus(gatekeeperStatus)
|
||||||
|
.gatekeeperRuleSetVersion(gatekeeperRuleSetVersion)
|
||||||
.composerStatus(composerStatus)
|
.composerStatus(composerStatus)
|
||||||
.claimCheckCount(claimCheckCount)
|
.claimCheckCount(claimCheckCount)
|
||||||
.toolCallCount(toolCallCount)
|
.toolCallCount(toolCallCount)
|
||||||
@@ -232,6 +236,7 @@ public class DiagnosisTraceEvaluator {
|
|||||||
String normalizedAnswer,
|
String normalizedAnswer,
|
||||||
String verdict,
|
String verdict,
|
||||||
String gatekeeperStatus,
|
String gatekeeperStatus,
|
||||||
|
String gatekeeperRuleSetVersion,
|
||||||
String composerStatus) {
|
String composerStatus) {
|
||||||
List<String> failedChecks = new ArrayList<>();
|
List<String> failedChecks = new ArrayList<>();
|
||||||
boolean requireV2AuditClosure = Boolean.TRUE.equals(evalCase.getRequireV2AuditClosure());
|
boolean requireV2AuditClosure = Boolean.TRUE.equals(evalCase.getRequireV2AuditClosure());
|
||||||
@@ -249,6 +254,11 @@ public class DiagnosisTraceEvaluator {
|
|||||||
&& !safeList(evalCase.getExpectedGatekeeperStatuses()).contains(gatekeeperStatus)) {
|
&& !safeList(evalCase.getExpectedGatekeeperStatuses()).contains(gatekeeperStatus)) {
|
||||||
failedChecks.add("gatekeeper status not expected: " + valueOrMissing(gatekeeperStatus));
|
failedChecks.add("gatekeeper status not expected: " + valueOrMissing(gatekeeperStatus));
|
||||||
}
|
}
|
||||||
|
if (!isBlank(evalCase.getExpectedGatekeeperRuleSetVersion())
|
||||||
|
&& !evalCase.getExpectedGatekeeperRuleSetVersion().equals(gatekeeperRuleSetVersion)) {
|
||||||
|
failedChecks.add("gatekeeper rule set version not expected: "
|
||||||
|
+ valueOrMissing(gatekeeperRuleSetVersion));
|
||||||
|
}
|
||||||
|
|
||||||
failedChecks.addAll(validateClaimChecks(trace, requireClaimChecks));
|
failedChecks.addAll(validateClaimChecks(trace, requireClaimChecks));
|
||||||
|
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ import com.fasterxml.jackson.core.type.TypeReference;
|
|||||||
import com.fasterxml.jackson.databind.JsonNode;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
import com.superbiz.agent.service.ExecutorGatekeeperService;
|
import com.superbiz.agent.service.ExecutorGatekeeperService;
|
||||||
|
import com.superbiz.agent.service.GatekeeperRuleCatalog;
|
||||||
import com.superbiz.agent.service.ToolTraceSummaryService;
|
import com.superbiz.agent.service.ToolTraceSummaryService;
|
||||||
import com.superbiz.agent.util.SessionContextHolder;
|
import com.superbiz.agent.util.SessionContextHolder;
|
||||||
import com.superbiz.agent.util.VerifierContextHolder;
|
import com.superbiz.agent.util.VerifierContextHolder;
|
||||||
@@ -110,9 +111,12 @@ public class VerifierInputHook extends MessagesModelHook {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private Map<String, Object> passGatekeeperResult() {
|
private Map<String, Object> passGatekeeperResult() {
|
||||||
|
GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback();
|
||||||
Map<String, Object> result = new LinkedHashMap<>();
|
Map<String, Object> result = new LinkedHashMap<>();
|
||||||
result.put("status", "pass");
|
result.put("status", "pass");
|
||||||
result.put("severity", "none");
|
result.put("severity", "none");
|
||||||
|
result.put("rule_set_version", catalog.version());
|
||||||
|
result.put("rules", catalog.auditRules());
|
||||||
result.put("checked_bindings", List.of());
|
result.put("checked_bindings", List.of());
|
||||||
result.put("failed_rules", List.of());
|
result.put("failed_rules", List.of());
|
||||||
result.put("warnings", List.of());
|
result.put("warnings", List.of());
|
||||||
|
|||||||
@@ -892,8 +892,7 @@ public class ChatService {
|
|||||||
Optional.ofNullable(VerifierContextHolder.getToolTraceSummary()).orElse(List.of()));
|
Optional.ofNullable(VerifierContextHolder.getToolTraceSummary()).orElse(List.of()));
|
||||||
verifierEvaluation.put("gatekeeper_result",
|
verifierEvaluation.put("gatekeeper_result",
|
||||||
Optional.ofNullable(VerifierContextHolder.getGatekeeperResult())
|
Optional.ofNullable(VerifierContextHolder.getGatekeeperResult())
|
||||||
.orElse(Map.of("status", "pass", "severity", "none", "checked_bindings", List.of(),
|
.orElse(defaultGatekeeperPass()));
|
||||||
"failed_rules", List.of(), "warnings", List.of(), "errors", List.of())));
|
|
||||||
if (composerOutput != null) {
|
if (composerOutput != null) {
|
||||||
verifierEvaluation.put("composer_output", composerOutput);
|
verifierEvaluation.put("composer_output", composerOutput);
|
||||||
}
|
}
|
||||||
@@ -903,6 +902,20 @@ public class ChatService {
|
|||||||
diagnosisSessionRepository.save(session);
|
diagnosisSessionRepository.save(session);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private Map<String, Object> defaultGatekeeperPass() {
|
||||||
|
GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback();
|
||||||
|
return Map.of(
|
||||||
|
"status", "pass",
|
||||||
|
"severity", "none",
|
||||||
|
"rule_set_version", catalog.version(),
|
||||||
|
"rules", catalog.auditRules(),
|
||||||
|
"checked_bindings", List.of(),
|
||||||
|
"failed_rules", List.of(),
|
||||||
|
"warnings", List.of(),
|
||||||
|
"errors", List.of()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
private ComposerRenderResult composeFinalAnswer(ChatModel chatModel, String originalQuery,
|
private ComposerRenderResult composeFinalAnswer(ChatModel chatModel, String originalQuery,
|
||||||
VerifierDecision decision, RunnableConfig config) {
|
VerifierDecision decision, RunnableConfig config) {
|
||||||
Map<String, Object> composerInput = buildComposerInput(originalQuery, decision);
|
Map<String, Object> composerInput = buildComposerInput(originalQuery, decision);
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import com.fasterxml.jackson.core.type.TypeReference;
|
|||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||||
import com.superbiz.agent.repository.ToolInvocationRepository;
|
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||||||
|
import org.springframework.beans.factory.annotation.Autowired;
|
||||||
import org.springframework.stereotype.Service;
|
import org.springframework.stereotype.Service;
|
||||||
|
|
||||||
import java.util.ArrayList;
|
import java.util.ArrayList;
|
||||||
@@ -35,19 +36,29 @@ public class ExecutorGatekeeperService {
|
|||||||
|
|
||||||
private static final TypeReference<Map<String, Object>> MAP_TYPE = new TypeReference<>() {
|
private static final TypeReference<Map<String, Object>> MAP_TYPE = new TypeReference<>() {
|
||||||
};
|
};
|
||||||
private static final double MIN_TOKEN_OVERLAP = 0.5;
|
private static final double DEFAULT_MIN_TOKEN_OVERLAP = 0.5;
|
||||||
|
|
||||||
private final ToolInvocationRepository toolInvocationRepository;
|
private final ToolInvocationRepository toolInvocationRepository;
|
||||||
|
private final GatekeeperRuleCatalog ruleCatalog;
|
||||||
private final ObjectMapper objectMapper = new ObjectMapper();
|
private final ObjectMapper objectMapper = new ObjectMapper();
|
||||||
|
|
||||||
|
@Autowired
|
||||||
public ExecutorGatekeeperService(ToolInvocationRepository toolInvocationRepository) {
|
public ExecutorGatekeeperService(ToolInvocationRepository toolInvocationRepository) {
|
||||||
|
this(toolInvocationRepository, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
public ExecutorGatekeeperService(ToolInvocationRepository toolInvocationRepository,
|
||||||
|
GatekeeperRuleCatalog ruleCatalog) {
|
||||||
this.toolInvocationRepository = toolInvocationRepository;
|
this.toolInvocationRepository = toolInvocationRepository;
|
||||||
|
this.ruleCatalog = ruleCatalog == null
|
||||||
|
? GatekeeperRuleCatalog.loadDefault(objectMapper)
|
||||||
|
: ruleCatalog;
|
||||||
}
|
}
|
||||||
|
|
||||||
public Map<String, Object> validate(String sessionId,
|
public Map<String, Object> validate(String sessionId,
|
||||||
Map<String, Object> structuredOutput,
|
Map<String, Object> structuredOutput,
|
||||||
Map<String, Object> parseStatus) {
|
Map<String, Object> parseStatus) {
|
||||||
GatekeeperResult result = new GatekeeperResult();
|
GatekeeperResult result = new GatekeeperResult(ruleCatalog);
|
||||||
validateSchema(structuredOutput, parseStatus, result);
|
validateSchema(structuredOutput, parseStatus, result);
|
||||||
if (structuredOutput != null) {
|
if (structuredOutput != null) {
|
||||||
validateInvocationRefs(sessionId, structuredOutput, result);
|
validateInvocationRefs(sessionId, structuredOutput, result);
|
||||||
@@ -57,11 +68,11 @@ public class ExecutorGatekeeperService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
public Map<String, Object> pass() {
|
public Map<String, Object> pass() {
|
||||||
return new GatekeeperResult().toMap();
|
return new GatekeeperResult(ruleCatalog).toMap();
|
||||||
}
|
}
|
||||||
|
|
||||||
public Map<String, Object> fail(String ruleId, String target, String message) {
|
public Map<String, Object> fail(String ruleId, String target, String message) {
|
||||||
GatekeeperResult result = new GatekeeperResult();
|
GatekeeperResult result = new GatekeeperResult(ruleCatalog);
|
||||||
result.fail(ruleId, target, message, SEVERITY_REJECT);
|
result.fail(ruleId, target, message, SEVERITY_REJECT);
|
||||||
return result.toMap();
|
return result.toMap();
|
||||||
}
|
}
|
||||||
@@ -474,7 +485,9 @@ public class ExecutorGatekeeperService {
|
|||||||
overlap++;
|
overlap++;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return (double) overlap / excerptTokens.size() >= MIN_TOKEN_OVERLAP;
|
double minTokenOverlap = ruleCatalog.doubleParameter(RULE_EXCERPT_MISMATCH,
|
||||||
|
"min_token_overlap", DEFAULT_MIN_TOKEN_OVERLAP);
|
||||||
|
return (double) overlap / excerptTokens.size() >= minTokenOverlap;
|
||||||
}
|
}
|
||||||
|
|
||||||
private String normalized(String value) {
|
private String normalized(String value) {
|
||||||
@@ -527,12 +540,17 @@ public class ExecutorGatekeeperService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private static final class GatekeeperResult {
|
private static final class GatekeeperResult {
|
||||||
|
private final GatekeeperRuleCatalog ruleCatalog;
|
||||||
private final List<String> failedRules = new ArrayList<>();
|
private final List<String> failedRules = new ArrayList<>();
|
||||||
private final List<Map<String, Object>> checkedBindings = new ArrayList<>();
|
private final List<Map<String, Object>> checkedBindings = new ArrayList<>();
|
||||||
private final List<Map<String, Object>> warnings = new ArrayList<>();
|
private final List<Map<String, Object>> warnings = new ArrayList<>();
|
||||||
private final List<Map<String, Object>> errors = new ArrayList<>();
|
private final List<Map<String, Object>> errors = new ArrayList<>();
|
||||||
private String severity = SEVERITY_NONE;
|
private String severity = SEVERITY_NONE;
|
||||||
|
|
||||||
|
GatekeeperResult(GatekeeperRuleCatalog ruleCatalog) {
|
||||||
|
this.ruleCatalog = ruleCatalog == null ? GatekeeperRuleCatalog.fallback() : ruleCatalog;
|
||||||
|
}
|
||||||
|
|
||||||
void fail(String ruleId, String target, String message, String failureSeverity) {
|
void fail(String ruleId, String target, String message, String failureSeverity) {
|
||||||
if (!failedRules.contains(ruleId)) {
|
if (!failedRules.contains(ruleId)) {
|
||||||
failedRules.add(ruleId);
|
failedRules.add(ruleId);
|
||||||
@@ -563,6 +581,8 @@ public class ExecutorGatekeeperService {
|
|||||||
Map<String, Object> result = new LinkedHashMap<>();
|
Map<String, Object> result = new LinkedHashMap<>();
|
||||||
result.put("status", failedRules.isEmpty() ? STATUS_PASS : STATUS_FAIL);
|
result.put("status", failedRules.isEmpty() ? STATUS_PASS : STATUS_FAIL);
|
||||||
result.put("severity", failedRules.isEmpty() ? SEVERITY_NONE : severity);
|
result.put("severity", failedRules.isEmpty() ? SEVERITY_NONE : severity);
|
||||||
|
result.put("rule_set_version", ruleCatalog.version());
|
||||||
|
result.put("rules", ruleCatalog.auditRules());
|
||||||
result.put("checked_bindings", checkedBindings);
|
result.put("checked_bindings", checkedBindings);
|
||||||
result.put("failed_rules", failedRules);
|
result.put("failed_rules", failedRules);
|
||||||
result.put("warnings", warnings);
|
result.put("warnings", warnings);
|
||||||
|
|||||||
@@ -0,0 +1,142 @@
|
|||||||
|
package com.superbiz.agent.service;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.core.type.TypeReference;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
|
||||||
|
import java.io.InputStream;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Lightweight metadata catalog for deterministic Gatekeeper rules.
|
||||||
|
*/
|
||||||
|
public class GatekeeperRuleCatalog {
|
||||||
|
|
||||||
|
public static final String DEFAULT_RESOURCE = "gatekeeper/gatekeeper-rules.json";
|
||||||
|
public static final String FALLBACK_VERSION = "gatekeeper-rules-v1";
|
||||||
|
|
||||||
|
private static final TypeReference<Map<String, Object>> MAP_TYPE = new TypeReference<>() {
|
||||||
|
};
|
||||||
|
|
||||||
|
private final String version;
|
||||||
|
private final List<RuleMetadata> rules;
|
||||||
|
|
||||||
|
public GatekeeperRuleCatalog(String version, List<RuleMetadata> rules) {
|
||||||
|
this.version = version == null || version.isBlank() ? FALLBACK_VERSION : version;
|
||||||
|
this.rules = List.copyOf(rules == null ? List.of() : rules);
|
||||||
|
}
|
||||||
|
|
||||||
|
public static GatekeeperRuleCatalog loadDefault(ObjectMapper objectMapper) {
|
||||||
|
try (InputStream input = GatekeeperRuleCatalog.class.getClassLoader()
|
||||||
|
.getResourceAsStream(DEFAULT_RESOURCE)) {
|
||||||
|
if (input == null) {
|
||||||
|
return fallback();
|
||||||
|
}
|
||||||
|
Map<String, Object> root = objectMapper.readValue(input, MAP_TYPE);
|
||||||
|
String version = stringValue(root.get("version"));
|
||||||
|
List<RuleMetadata> rules = new ArrayList<>();
|
||||||
|
Object rulesValue = root.get("rules");
|
||||||
|
if (rulesValue instanceof List<?> ruleList) {
|
||||||
|
for (Object ruleValue : ruleList) {
|
||||||
|
if (ruleValue instanceof Map<?, ?> ruleMap) {
|
||||||
|
rules.add(RuleMetadata.from(ruleMap));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return new GatekeeperRuleCatalog(version, rules);
|
||||||
|
} catch (Exception ignored) {
|
||||||
|
return fallback();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
public static GatekeeperRuleCatalog fallback() {
|
||||||
|
return new GatekeeperRuleCatalog(FALLBACK_VERSION, List.of(
|
||||||
|
new RuleMetadata("schema.executor_v2", "Executor output must match executor_evidence_v2 schema",
|
||||||
|
true, "low_confid", Map.of()),
|
||||||
|
new RuleMetadata("evidence.invocation_ref", "source_invocation_id must refer to a real current-session tool invocation",
|
||||||
|
true, "reject", Map.of()),
|
||||||
|
new RuleMetadata("evidence.raw_path", "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
true, "reject", Map.of()),
|
||||||
|
new RuleMetadata("evidence.excerpt_mismatch", "evidence_excerpt must be supported by the matched evidence ref text",
|
||||||
|
true, "reject", Map.of("min_token_overlap", 0.5)),
|
||||||
|
new RuleMetadata("evidence.missing", "claims must include usable evidence bindings",
|
||||||
|
true, "low_confid", Map.of())
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
public String version() {
|
||||||
|
return version;
|
||||||
|
}
|
||||||
|
|
||||||
|
public List<Map<String, Object>> auditRules() {
|
||||||
|
List<Map<String, Object>> result = new ArrayList<>();
|
||||||
|
for (RuleMetadata rule : rules) {
|
||||||
|
if (rule.enabled()) {
|
||||||
|
result.add(rule.toAuditMap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
public double doubleParameter(String ruleId, String parameterName, double fallback) {
|
||||||
|
for (RuleMetadata rule : rules) {
|
||||||
|
if (!rule.id().equals(ruleId) || !rule.enabled()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
Object value = rule.parameters().get(parameterName);
|
||||||
|
if (value instanceof Number number) {
|
||||||
|
return number.doubleValue();
|
||||||
|
}
|
||||||
|
if (value instanceof String text) {
|
||||||
|
try {
|
||||||
|
return Double.parseDouble(text);
|
||||||
|
} catch (NumberFormatException ignored) {
|
||||||
|
return fallback;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return fallback;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String stringValue(Object value) {
|
||||||
|
return value == null ? "" : String.valueOf(value);
|
||||||
|
}
|
||||||
|
|
||||||
|
public record RuleMetadata(String id,
|
||||||
|
String description,
|
||||||
|
boolean enabled,
|
||||||
|
String defaultSeverity,
|
||||||
|
Map<String, Object> parameters) {
|
||||||
|
|
||||||
|
static RuleMetadata from(Map<?, ?> raw) {
|
||||||
|
String id = stringValue(raw.get("id"));
|
||||||
|
String description = stringValue(raw.get("description"));
|
||||||
|
boolean enabled = !(raw.get("enabled") instanceof Boolean value) || value;
|
||||||
|
String defaultSeverity = stringValue(raw.get("default_severity"));
|
||||||
|
Map<String, Object> parameters = new LinkedHashMap<>();
|
||||||
|
Object parametersValue = raw.get("parameters");
|
||||||
|
if (parametersValue instanceof Map<?, ?> parameterMap) {
|
||||||
|
for (Map.Entry<?, ?> entry : parameterMap.entrySet()) {
|
||||||
|
if (entry.getKey() != null) {
|
||||||
|
parameters.put(String.valueOf(entry.getKey()), entry.getValue());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return new RuleMetadata(id, description, enabled, defaultSeverity, parameters);
|
||||||
|
}
|
||||||
|
|
||||||
|
Map<String, Object> toAuditMap() {
|
||||||
|
Map<String, Object> result = new LinkedHashMap<>();
|
||||||
|
result.put("id", id);
|
||||||
|
result.put("description", description);
|
||||||
|
result.put("enabled", enabled);
|
||||||
|
result.put("default_severity", defaultSeverity);
|
||||||
|
if (!parameters.isEmpty()) {
|
||||||
|
result.put("parameters", parameters);
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
{
|
||||||
|
"version": "gatekeeper-rules-v1",
|
||||||
|
"rules": [
|
||||||
|
{
|
||||||
|
"id": "schema.executor_v2",
|
||||||
|
"description": "Executor output must match executor_evidence_v2 schema",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "low_confid"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "evidence.invocation_ref",
|
||||||
|
"description": "source_invocation_id must refer to a real current-session tool invocation",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "evidence.raw_path",
|
||||||
|
"description": "raw_path must exist in retrieval_details.evidence_refs",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "evidence.excerpt_mismatch",
|
||||||
|
"description": "evidence_excerpt must be supported by the matched evidence ref text",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "reject",
|
||||||
|
"parameters": {
|
||||||
|
"min_token_overlap": 0.5
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "evidence.missing",
|
||||||
|
"description": "claims must include usable evidence bindings",
|
||||||
|
"enabled": true,
|
||||||
|
"default_severity": "low_confid"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -100,12 +100,12 @@ class DiagnosisEvalBaselineDiffTest {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private void degradeRedisCase(DiagnosisEvalReport report) {
|
private void degradeRedisCase(DiagnosisEvalReport report) {
|
||||||
report.setPassedCases(7);
|
report.setPassedCases(9);
|
||||||
report.setPassRate(0.875);
|
report.setPassRate(0.9);
|
||||||
report.setAverageToolCallCount(3.0);
|
report.setAverageToolCallCount(3.0);
|
||||||
report.setAverageDurationMs(44875.0);
|
report.setAverageDurationMs(39800.0);
|
||||||
report.setVerdictDistribution(new LinkedHashMap<>());
|
report.setVerdictDistribution(new LinkedHashMap<>());
|
||||||
report.getVerdictDistribution().put("PASS", 2L);
|
report.getVerdictDistribution().put("PASS", 4L);
|
||||||
report.getVerdictDistribution().put("LOW_CONFID", 4L);
|
report.getVerdictDistribution().put("LOW_CONFID", 4L);
|
||||||
report.getVerdictDistribution().put("REJECT", 2L);
|
report.getVerdictDistribution().put("REJECT", 2L);
|
||||||
|
|
||||||
|
|||||||
@@ -24,13 +24,23 @@ class DiagnosisTraceEvaluatorTest {
|
|||||||
|
|
||||||
DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures"));
|
DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures"));
|
||||||
|
|
||||||
assertEquals(8, report.getTotalCases());
|
assertEquals(10, report.getTotalCases());
|
||||||
assertEquals(8, report.getPassedCases());
|
assertEquals(10, report.getPassedCases());
|
||||||
assertEquals(1.0, report.getPassRate(), 0.001);
|
assertEquals(1.0, report.getPassRate(), 0.001);
|
||||||
assertEquals(2L, report.getVerdictDistribution().get("PASS"));
|
assertEquals(4L, report.getVerdictDistribution().get("PASS"));
|
||||||
assertEquals(5L, report.getVerdictDistribution().get("LOW_CONFID"));
|
assertEquals(5L, report.getVerdictDistribution().get("LOW_CONFID"));
|
||||||
assertEquals(1L, report.getVerdictDistribution().get("REJECT"));
|
assertEquals(1L, report.getVerdictDistribution().get("REJECT"));
|
||||||
|
|
||||||
|
DiagnosisEvalResult narrowHighCpu = result(report, "narrow-highcpu-observation");
|
||||||
|
assertTrue(narrowHighCpu.isPassed());
|
||||||
|
assertEquals("gatekeeper-rules-v1", narrowHighCpu.getGatekeeperRuleSetVersion());
|
||||||
|
assertEquals("pass", narrowHighCpu.getGatekeeperStatus());
|
||||||
|
|
||||||
|
DiagnosisEvalResult hikariNoEvidence = result(report, "hikari-no-evidence-negative-observation");
|
||||||
|
assertTrue(hikariNoEvidence.isPassed());
|
||||||
|
assertEquals("gatekeeper-rules-v1", hikariNoEvidence.getGatekeeperRuleSetVersion());
|
||||||
|
assertEquals("pass", hikariNoEvidence.getGatekeeperStatus());
|
||||||
|
|
||||||
DiagnosisEvalResult payment = result(report, "payment-timeout");
|
DiagnosisEvalResult payment = result(report, "payment-timeout");
|
||||||
assertTrue(payment.isPassed());
|
assertTrue(payment.isPassed());
|
||||||
assertTrue(payment.getEvidenceCoverage().get("lookup_knowledge"));
|
assertTrue(payment.getEvidenceCoverage().get("lookup_knowledge"));
|
||||||
@@ -153,6 +163,39 @@ class DiagnosisTraceEvaluatorTest {
|
|||||||
assertTrue(result.getFailedChecks().contains("gatekeeper fail cannot have PASS verdict"));
|
assertTrue(result.getFailedChecks().contains("gatekeeper fail cannot have PASS verdict"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void evaluateFailsWhenGatekeeperRuleSetVersionMismatches() {
|
||||||
|
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
|
||||||
|
.id("rule-version")
|
||||||
|
.title("Rule version")
|
||||||
|
.expectedRootCauseKeywords(List.of())
|
||||||
|
.requiredEvidenceTools(List.of())
|
||||||
|
.allowedVerdicts(List.of("PASS"))
|
||||||
|
.expectedGatekeeperRuleSetVersion("gatekeeper-rules-v1")
|
||||||
|
.build();
|
||||||
|
DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder()
|
||||||
|
.session(DiagnosisTraceResponse.SessionTrace.builder()
|
||||||
|
.answer("安全回答")
|
||||||
|
.selfEvaluation(java.util.Map.of(
|
||||||
|
"verifier_evaluation", java.util.Map.of(
|
||||||
|
"verdict", "PASS",
|
||||||
|
"gatekeeper_result", java.util.Map.of(
|
||||||
|
"status", "pass",
|
||||||
|
"rule_set_version", "old-rules"
|
||||||
|
)
|
||||||
|
)))
|
||||||
|
.build())
|
||||||
|
.toolInvocations(List.of())
|
||||||
|
.build();
|
||||||
|
|
||||||
|
DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace);
|
||||||
|
|
||||||
|
assertFalse(result.isPassed());
|
||||||
|
assertTrue(result.getFailedChecks().contains(
|
||||||
|
"gatekeeper rule set version not expected: old-rules"));
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void evaluateFailsWhenUnsupportedClaimLeaksIntoFinalAnswer() {
|
void evaluateFailsWhenUnsupportedClaimLeaksIntoFinalAnswer() {
|
||||||
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
|
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
|
||||||
|
|||||||
@@ -8,12 +8,25 @@ import java.util.List;
|
|||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
import static org.mockito.Mockito.mock;
|
import static org.mockito.Mockito.mock;
|
||||||
import static org.mockito.Mockito.when;
|
import static org.mockito.Mockito.when;
|
||||||
|
|
||||||
class ExecutorGatekeeperServiceTest {
|
class ExecutorGatekeeperServiceTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void ruleCatalogLoadsDefaultMetadata() {
|
||||||
|
GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.loadDefault(new com.fasterxml.jackson.databind.ObjectMapper());
|
||||||
|
|
||||||
|
assertEquals("gatekeeper-rules-v1", catalog.version());
|
||||||
|
assertFalse(catalog.auditRules().isEmpty());
|
||||||
|
assertTrue(catalog.auditRules().stream()
|
||||||
|
.anyMatch(rule -> "evidence.raw_path".equals(rule.get("id"))));
|
||||||
|
assertEquals(0.5, catalog.doubleParameter("evidence.excerpt_mismatch",
|
||||||
|
"min_token_overlap", 0.0), 0.001);
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void validatePassesForExecutorEvidenceV2WithMatchingInvocation() {
|
void validatePassesForExecutorEvidenceV2WithMatchingInvocation() {
|
||||||
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
||||||
@@ -30,6 +43,7 @@ class ExecutorGatekeeperServiceTest {
|
|||||||
|
|
||||||
assertEquals("pass", result.get("status"));
|
assertEquals("pass", result.get("status"));
|
||||||
assertEquals("none", result.get("severity"));
|
assertEquals("none", result.get("severity"));
|
||||||
|
assertRuleAudit(result);
|
||||||
assertTrue(((List<?>) result.get("failed_rules")).isEmpty());
|
assertTrue(((List<?>) result.get("failed_rules")).isEmpty());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -49,6 +63,7 @@ class ExecutorGatekeeperServiceTest {
|
|||||||
|
|
||||||
assertEquals("pass", result.get("status"));
|
assertEquals("pass", result.get("status"));
|
||||||
assertEquals("none", result.get("severity"));
|
assertEquals("none", result.get("severity"));
|
||||||
|
assertRuleAudit(result);
|
||||||
assertTrue(((List<?>) result.get("failed_rules")).isEmpty());
|
assertTrue(((List<?>) result.get("failed_rules")).isEmpty());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -115,6 +130,7 @@ class ExecutorGatekeeperServiceTest {
|
|||||||
|
|
||||||
assertEquals("fail", result.get("status"));
|
assertEquals("fail", result.get("status"));
|
||||||
assertEquals("reject", result.get("severity"));
|
assertEquals("reject", result.get("severity"));
|
||||||
|
assertRuleAudit(result);
|
||||||
assertTrue(((List<?>) result.get("failed_rules")).contains("evidence.excerpt_mismatch"));
|
assertTrue(((List<?>) result.get("failed_rules")).contains("evidence.excerpt_mismatch"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -167,9 +183,22 @@ class ExecutorGatekeeperServiceTest {
|
|||||||
|
|
||||||
assertEquals("fail", result.get("status"));
|
assertEquals("fail", result.get("status"));
|
||||||
assertEquals("reject", result.get("severity"));
|
assertEquals("reject", result.get("severity"));
|
||||||
|
assertRuleAudit(result);
|
||||||
assertTrue(((List<?>) result.get("failed_rules")).contains("evidence.invocation_ref"));
|
assertTrue(((List<?>) result.get("failed_rules")).contains("evidence.invocation_ref"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void failResultIncludesRuleAuditMetadata() {
|
||||||
|
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
||||||
|
ExecutorGatekeeperService service = new ExecutorGatekeeperService(repository);
|
||||||
|
|
||||||
|
Map<String, Object> result = service.fail("gatekeeper.internal_error", "gatekeeper", "boom");
|
||||||
|
|
||||||
|
assertEquals("fail", result.get("status"));
|
||||||
|
assertEquals("reject", result.get("severity"));
|
||||||
|
assertRuleAudit(result);
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void validateFailsForToolNameMismatch() {
|
void validateFailsForToolNameMismatch() {
|
||||||
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
||||||
@@ -329,4 +358,13 @@ class ExecutorGatekeeperServiceTest {
|
|||||||
"missing_info", List.of()
|
"missing_info", List.of()
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private void assertRuleAudit(Map<String, Object> result) {
|
||||||
|
assertEquals("gatekeeper-rules-v1", result.get("rule_set_version"));
|
||||||
|
assertTrue(result.get("rules") instanceof List<?>);
|
||||||
|
List<?> rules = (List<?>) result.get("rules");
|
||||||
|
assertFalse(rules.isEmpty());
|
||||||
|
assertTrue(rules.stream().anyMatch(rule ->
|
||||||
|
rule instanceof Map<?, ?> map && "evidence.raw_path".equals(map.get("id"))));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user