Merge branch 'emdash/shy-items-fry-f4zze' into refactor/mvp1.0
# Conflicts: # mvp/issues/README.md
This commit is contained in:
@@ -60,3 +60,7 @@ uploads/
|
|||||||
### Windows / Runtime Artifacts
|
### Windows / Runtime Artifacts
|
||||||
*.stackdump
|
*.stackdump
|
||||||
NUL
|
NUL
|
||||||
|
|
||||||
|
### MVP Demo Generated Outputs
|
||||||
|
mvp/demo/output/*.json
|
||||||
|
!mvp/demo/output/README.md
|
||||||
|
|||||||
@@ -4,6 +4,11 @@
|
|||||||
|
|
||||||
| 日期 | slug | 领域 | 关键词 | 状态 |
|
| 日期 | slug | 领域 | 关键词 | 状态 |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|---|
|
||||||
|
| 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived |
|
||||||
|
| 2026-07-05 | diagnosis-eval-baseline-diff | Agent 评测/回归 Diff | baseline diff, regression detection, evidence coverage, cost signal, markdown report | openspec/changes/archive/2026-07-05-diagnosis-eval-baseline-diff | archived |
|
||||||
|
| 2026-07-04 | expand-diagnosis-eval-fixtures | Agent 评测/回归 Baseline | fixture coverage, baseline report, redis timeout, slow response, jvm memory risk | openspec/changes/archive/2026-07-05-expand-diagnosis-eval-fixtures | archived |
|
||||||
|
| 2026-07-04 | diagnosis-eval-harness | Agent 评测/回归 Harness | fixed cases, trace validation, evidence coverage, verdict distribution, markdown report | openspec/changes/archive/2026-07-04-diagnosis-eval-harness | archived |
|
||||||
|
| 2026-07-04 | evidence-trace-hardening | 证据链/降级契约/离线验证 | ToolInvocationRecorder, ToolTraceSummaryService, lookup_knowledge, query_logs, query_metrics, LOW_CONFID, REJECT | openspec/changes/archive/2026-07-04-evidence-trace-hardening | archived |
|
||||||
| 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived |
|
| 2026-07-03 | mvp-demo-trace-acceptance | MVP Demo/trace/acceptance | mvp-demo, trace API, diagnosis_session, agent_step, tool_invocation, feedback | openspec/changes/archive/2026-07-03-mvp-demo-trace-acceptance | archived |
|
||||||
| 2026-07-04 | aiops-traceable-diagnosis-entry | AIOps/trace/alert diagnosis | ai_ops, SSE, alert input, sessionId, diagnosis_session, trace API | openspec/changes/archive/2026-07-04-aiops-traceable-diagnosis-entry | archived |
|
| 2026-07-04 | aiops-traceable-diagnosis-entry | AIOps/trace/alert diagnosis | ai_ops, SSE, alert input, sessionId, diagnosis_session, trace API | openspec/changes/archive/2026-07-04-aiops-traceable-diagnosis-entry | archived |
|
||||||
| 2026-07-04 | aiops-alert-scope-control | AIOps/scope/prompt control | payload mode, auto-discovery mode, queryPrometheusAlerts, HighCPUUsage | openspec/changes/archive/2026-07-04-aiops-alert-scope-control | archived |
|
| 2026-07-04 | aiops-alert-scope-control | AIOps/scope/prompt control | payload mode, auto-discovery mode, queryPrometheusAlerts, HighCPUUsage | openspec/changes/archive/2026-07-04-aiops-alert-scope-control | archived |
|
||||||
|
|||||||
@@ -0,0 +1,36 @@
|
|||||||
|
# Acceptance: diagnosis-eval-harness
|
||||||
|
|
||||||
|
## Classification
|
||||||
|
|
||||||
|
standard-light
|
||||||
|
|
||||||
|
## Task Status
|
||||||
|
|
||||||
|
| Task | Status | Notes |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Issue and OpenSpec setup | Done | `ISS-006` and initial OpenSpec artifacts were created. |
|
||||||
|
| Implementation | Done | Added fixed cases, fixture-mode trace evaluation, aggregate metrics, and JSON / Markdown report writer. |
|
||||||
|
| Verification | Done | Targeted evaluator tests, compile verification, and OpenSpec validation passed. |
|
||||||
|
|
||||||
|
## Current State
|
||||||
|
|
||||||
|
- First implementation uses fixture-mode evaluation.
|
||||||
|
- Live trace API polling remains a follow-up option.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
### Script Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`
|
||||||
|
- Result: passed
|
||||||
|
- Notes: Covers fixed case loading, fixture evaluation, missing fixture reporting, reject degraded-output validation, and report writing.
|
||||||
|
|
||||||
|
### Static Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q -DskipTests compile`
|
||||||
|
- Result: passed
|
||||||
|
|
||||||
|
### OpenSpec Verification
|
||||||
|
|
||||||
|
- Command: `openspec validate diagnosis-eval-harness --strict`
|
||||||
|
- Result: passed
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
# Brief: diagnosis-eval-harness
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The MVP has a runnable demo and hardened evidence trace semantics, but it still lacks a fixed regression baseline for Agent diagnosis quality. P1-B creates a small evaluation harness that can validate diagnosis traces against fixed cases and produce repeatable reports.
|
||||||
|
|
||||||
|
## Goals
|
||||||
|
|
||||||
|
1. Define fixed diagnosis cases for the MVP demo domain.
|
||||||
|
2. Validate trace evidence, verifier verdicts, answer keywords, and degraded-output behavior.
|
||||||
|
3. Produce JSON and Markdown reports for interview and regression use.
|
||||||
|
4. Keep the first version offline by supporting trace fixtures.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Evaluation case definitions
|
||||||
|
- Trace fixture shape
|
||||||
|
- Rule-based evaluator
|
||||||
|
- JSON / Markdown report output
|
||||||
|
- Focused offline tests and docs
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
- No LLM-as-judge
|
||||||
|
- No live end-to-end runtime requirement
|
||||||
|
- No production API
|
||||||
|
- No chat or verifier runtime change
|
||||||
|
|
||||||
|
## Related OpenSpec
|
||||||
|
|
||||||
|
`openspec/changes/diagnosis-eval-harness/`
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
# Diagnosis Eval Harness Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: build P1-B fixed case evaluation after evidence trace hardening.
|
||||||
|
- Slug: `diagnosis-eval-harness`
|
||||||
|
- Devflow scale: standard-light
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- P1-A `evidence-trace-hardening` created stable evidence semantics for supported, no-evidence, deduped, and failed tool calls.
|
||||||
|
- The MVP demo trace API already provides an aggregate trace shape suitable for evaluation.
|
||||||
|
- The first evaluator should avoid depending on external infrastructure so it can run in regular development.
|
||||||
|
|
||||||
|
## Key Decisions
|
||||||
|
|
||||||
|
- Decision: Start with rule-based trace validation instead of LLM-as-judge.
|
||||||
|
- Reason: The first regression signal should be deterministic and tied to trace contracts.
|
||||||
|
|
||||||
|
- Decision: Support offline fixture traces first.
|
||||||
|
- Reason: This makes the harness usable without MySQL, Redis, Milvus, or a real LLM.
|
||||||
|
|
||||||
|
- Decision: Output both JSON and Markdown.
|
||||||
|
- Reason: JSON supports automation; Markdown is easier to discuss in interviews.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- Whether live trace API polling belongs in this change or a follow-up after fixture mode lands.
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
# Diagnosis Eval Harness Evidence
|
||||||
|
|
||||||
|
## Evidence
|
||||||
|
|
||||||
|
| Source | Evidence | Conclusion | Reported |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `openspec/specs/evidence-trace-hardening/spec.md` | Defines stable evidence states and summary behavior | Evaluation can rely on trace semantics rather than ad hoc log parsing | Yes |
|
||||||
|
| `mvp/demo/README.md` | Documents an end-to-end demo flow with chat, trace, and feedback | Existing demo flow provides the runtime story, but not a reusable evaluation baseline | Yes |
|
||||||
|
| `DiagnosisTraceService` | Aggregates session, steps, tools, and self-evaluation | Trace response shape can be reused as evaluation input | Yes |
|
||||||
|
| `ToolTraceSummaryService` | Builds verifier-facing evidence summaries from persisted tool rows | Evaluator can check evidence coverage through persisted trace artifacts | Yes |
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
# Acceptance: evidence-trace-hardening
|
||||||
|
|
||||||
|
## Classification
|
||||||
|
|
||||||
|
standard-light
|
||||||
|
|
||||||
|
## Task Status
|
||||||
|
|
||||||
|
| Task | Status | Notes |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Issue and OpenSpec setup | Done | `ISS-005` and the initial OpenSpec artifacts were created. |
|
||||||
|
| Implementation | Done | Recorder contract, lookup persistence path, evidence summary semantics, and degraded-path tests were implemented. |
|
||||||
|
| Verification | Done | Targeted offline tests and compile verification passed. |
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
### Script Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q "-Dtest=ToolInvocationRecorderTest,ToolTraceSummaryServiceTest,ChatServiceSequentialAgentTest,LookupKnowledgeToolTest" test`
|
||||||
|
- Result: passed
|
||||||
|
- Notes: Covers recorder contract, summary semantics for success/failure/no-evidence, and `ChatService` fallback / degraded paths.
|
||||||
|
|
||||||
|
### Static Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q -DskipTests compile`
|
||||||
|
- Result: passed
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
| Question | Current position |
|
||||||
|
| --- | --- |
|
||||||
|
| Should deduped retrievals be counted separately from generic no-hit events in future evaluation metrics? | Deferred to P1-B; this change preserves enough structure to decide later. |
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
# Brief: evidence-trace-hardening
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The MVP already has persisted tool traces and a verifier, but the evidence contract is still only partially standardized. For interview-focused hardening, the project now needs a tighter contract for evidence persistence, no-evidence / failure semantics, and degraded-output behavior.
|
||||||
|
|
||||||
|
## Goals
|
||||||
|
|
||||||
|
1. Standardize the persisted evidence-tool contract across `lookup_knowledge`, `query_logs`, and `query_metrics`.
|
||||||
|
2. Make verifier-facing summaries distinguish failed calls, no-hit calls, deduped retrievals, and actual supporting evidence.
|
||||||
|
3. Add offline tests for verifier fallback and degraded-output paths.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- `ToolInvocationRecorder`
|
||||||
|
- `LookupKnowledgeTool`
|
||||||
|
- `QueryLogsTools`
|
||||||
|
- `QueryMetricsTools`
|
||||||
|
- `ToolTraceSummaryService`
|
||||||
|
- `ChatService`
|
||||||
|
- Focused offline tests
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
- No new API or schema
|
||||||
|
- No evaluation harness yet
|
||||||
|
- No trace UI
|
||||||
|
- No security/config cleanup
|
||||||
|
|
||||||
|
## Related OpenSpec
|
||||||
|
|
||||||
|
`openspec/changes/evidence-trace-hardening/`
|
||||||
@@ -0,0 +1,24 @@
|
|||||||
|
# Evidence Trace Hardening Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: harden the MVP evidence contract before building the P1-B evaluation harness.
|
||||||
|
- Slug: `evidence-trace-hardening`
|
||||||
|
- Devflow scale: standard-light
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- `ISS-003` raised verifier traceability and failure-path concerns.
|
||||||
|
- Current code inspection shows `QueryLogsTools` and `QueryMetricsTools` already use `ToolInvocationRecorder`, while `LookupKnowledgeTool` still persists rows through a local helper.
|
||||||
|
- `ChatService` already contains fallback behavior for missing/invalid `verifier_output`, but coverage is narrow.
|
||||||
|
|
||||||
|
## Key Decisions
|
||||||
|
|
||||||
|
- Decision: Treat this as a contract-hardening change, not a new feature change.
|
||||||
|
- Reason: The project already has the necessary runtime pieces; the gap is semantic consistency and testability.
|
||||||
|
|
||||||
|
- Decision: Keep the scope before P1-B.
|
||||||
|
- Reason: The evaluation harness will rely on stable evidence semantics, so this contract slice should land first.
|
||||||
|
|
||||||
|
- Decision: Preserve schema and API stability.
|
||||||
|
- Reason: The interview value here is engineering rigor, not more surface area.
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Evidence Trace Hardening Evidence
|
||||||
|
|
||||||
|
## Evidence
|
||||||
|
|
||||||
|
| Source | Evidence | Conclusion | Reported |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `ToolInvocationRecorder` | Provides a common persistence seam for evidence tools | Contract hardening should build on the existing recorder instead of introducing a new store path | Yes |
|
||||||
|
| `LookupKnowledgeTool` | Still constructs `ToolInvocation` rows through a local helper | Retrieval-aware evidence persistence is not yet unified with the recorder contract | Yes |
|
||||||
|
| `QueryLogsTools` / `QueryMetricsTools` | Already record evidence invocations through `recordEvidenceTool(...)` | Current gap is semantic alignment, not missing persistence | Yes |
|
||||||
|
| `ToolTraceSummaryService` | Merges rows by tool and topic domain and infers evidence level heuristically | Summary rules need explicit handling for failure, no-hit, and dedup cases | Yes |
|
||||||
|
| `ChatService` | Falls back to `LOW_CONFID` when verifier output is missing or invalid | These degraded paths exist and should now be covered by focused offline tests | Yes |
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
# Acceptance: expand-diagnosis-eval-fixtures
|
||||||
|
|
||||||
|
## Classification
|
||||||
|
|
||||||
|
standard-light
|
||||||
|
|
||||||
|
## Task Status
|
||||||
|
|
||||||
|
| Task | Status | Notes |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Issue and OpenSpec setup | Done | Created slug-based issue and initial OpenSpec artifacts. |
|
||||||
|
| Implementation | Done | Added remaining fixtures, full baseline reports, and documentation updates. |
|
||||||
|
| Verification | Done | Evaluator tests, compile verification, and OpenSpec validation passed. |
|
||||||
|
|
||||||
|
## Current State
|
||||||
|
|
||||||
|
- Fixture coverage is complete for the five fixed diagnosis cases.
|
||||||
|
- Baseline reports are saved under `mvp/eval/reports`.
|
||||||
|
- No production runtime behavior has been changed.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
### Script Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`
|
||||||
|
- Result: passed
|
||||||
|
- Notes: Covers full fixture coverage, baseline report matching, reject degraded-output validation, and report writing.
|
||||||
|
|
||||||
|
### Static Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q -DskipTests compile`
|
||||||
|
- Result: passed
|
||||||
|
|
||||||
|
### OpenSpec Verification
|
||||||
|
|
||||||
|
- Command: `openspec validate expand-diagnosis-eval-fixtures --strict`
|
||||||
|
- Result: passed
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
# Brief: expand-diagnosis-eval-fixtures
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The diagnosis eval harness is implemented and archived, but the fixed baseline is incomplete because three of the five diagnosis cases still reference missing fixtures.
|
||||||
|
|
||||||
|
## Goals
|
||||||
|
|
||||||
|
1. Add representative trace fixtures for all remaining fixed diagnosis cases.
|
||||||
|
2. Save a reproducible baseline report in JSON and Markdown.
|
||||||
|
3. Document how to regenerate and interpret the baseline.
|
||||||
|
4. Keep evaluation offline and deterministic.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Redis timeout fixture
|
||||||
|
- Slow response fixture
|
||||||
|
- JVM memory risk fixture
|
||||||
|
- Baseline reports under `mvp/eval/reports`
|
||||||
|
- Focused tests for full fixture coverage and report generation
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
- No new diagnosis cases
|
||||||
|
- No production Agent runtime changes
|
||||||
|
- No LLM-as-judge
|
||||||
|
- No live infrastructure requirement
|
||||||
|
|
||||||
|
## Related OpenSpec
|
||||||
|
|
||||||
|
`openspec/changes/expand-diagnosis-eval-fixtures/`
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
# Expand Diagnosis Eval Fixtures Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: complete the fixed diagnosis eval baseline after the harness is in place.
|
||||||
|
- Slug: `expand-diagnosis-eval-fixtures`
|
||||||
|
- Devflow scale: standard-light
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- `diagnosis-eval-harness` created the evaluator, case file, fixture mode, and report writer.
|
||||||
|
- The first baseline still has missing fixtures by design.
|
||||||
|
- This follow-up turns that partial baseline into a full fixed-case baseline.
|
||||||
|
|
||||||
|
## Key Decisions
|
||||||
|
|
||||||
|
- Decision: Keep this change data-focused.
|
||||||
|
- Reason: the evaluator rules already landed; this change should not blur fixture expansion with harness behavior changes.
|
||||||
|
|
||||||
|
- Decision: Save baseline reports in the repository.
|
||||||
|
- Reason: interview review and future diffs are easier when the expected baseline is visible.
|
||||||
|
|
||||||
|
- Decision: Use deterministic fixture traces instead of live trace generation.
|
||||||
|
- Reason: this baseline should run without infrastructure or external model calls.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- Whether a future change should add a CLI or Maven goal for report regeneration.
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Evidence: expand-diagnosis-eval-fixtures
|
||||||
|
|
||||||
|
## Evidence Log
|
||||||
|
|
||||||
|
- 2026-07-04: Created slug-based issue `expand-diagnosis-eval-fixtures.md`.
|
||||||
|
- 2026-07-04: Created OpenSpec change `expand-diagnosis-eval-fixtures`.
|
||||||
|
- 2026-07-04: Added Redis timeout, slow response, and JVM memory risk fixtures.
|
||||||
|
- 2026-07-04: Added baseline JSON and Markdown reports under `mvp/eval/reports`.
|
||||||
|
- 2026-07-04: Verification passed with `mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test`.
|
||||||
|
- 2026-07-04: Verification passed with `mvn -q -DskipTests compile`.
|
||||||
|
- 2026-07-04: Verification passed with `openspec validate expand-diagnosis-eval-fixtures --strict`.
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
# Acceptance: diagnosis-eval-baseline-diff
|
||||||
|
|
||||||
|
## Classification
|
||||||
|
|
||||||
|
standard-light
|
||||||
|
|
||||||
|
## Task Status
|
||||||
|
|
||||||
|
| Task | Status | Notes |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Issue and OpenSpec setup | Done | Created slug-based issue and initial OpenSpec artifacts. |
|
||||||
|
| Implementation | Done | Added diff model, comparator, writer, docs, sample outputs, and focused tests. |
|
||||||
|
| Verification | Done | Diff/evaluator tests, compile verification, and OpenSpec validation passed. |
|
||||||
|
|
||||||
|
## Current State
|
||||||
|
|
||||||
|
- Baseline diff is implemented for aggregate metrics, verdict distribution, case-level state, keyword coverage, evidence coverage, missing cases, and new cases.
|
||||||
|
- JSON and Markdown diff output are available.
|
||||||
|
- No production runtime behavior has been changed.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
### Script Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q "-Dtest=DiagnosisEvalBaselineDiffTest" test`
|
||||||
|
- Result: passed
|
||||||
|
- Notes: Also verified with `DiagnosisTraceEvaluatorTest`.
|
||||||
|
|
||||||
|
### Static Verification
|
||||||
|
|
||||||
|
- Command: `mvn -q -DskipTests compile`
|
||||||
|
- Result: passed
|
||||||
|
|
||||||
|
### OpenSpec Verification
|
||||||
|
|
||||||
|
- Command: `openspec validate diagnosis-eval-baseline-diff --strict`
|
||||||
|
- Result: passed
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
# Brief: diagnosis-eval-baseline-diff
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The eval harness now has a complete saved baseline. This change adds the comparison layer that turns the baseline into an actionable regression signal.
|
||||||
|
|
||||||
|
## Goals
|
||||||
|
|
||||||
|
1. Compare baseline and current `DiagnosisEvalReport` objects.
|
||||||
|
2. Detect aggregate and per-case regressions.
|
||||||
|
3. Output JSON and Markdown diff reports.
|
||||||
|
4. Document how to read the diff in interview and engineering terms.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Diff data structures
|
||||||
|
- Deterministic report comparison
|
||||||
|
- JSON / Markdown diff output
|
||||||
|
- Focused tests and eval docs
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
- No live Agent execution
|
||||||
|
- No LLM-as-judge
|
||||||
|
- No evaluator scoring rule changes
|
||||||
|
- No production API changes
|
||||||
|
|
||||||
|
## Related OpenSpec
|
||||||
|
|
||||||
|
`openspec/changes/diagnosis-eval-baseline-diff/`
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
# Diagnosis Eval Baseline Diff Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: add report diffing on top of the completed diagnosis eval baseline.
|
||||||
|
- Slug: `diagnosis-eval-baseline-diff`
|
||||||
|
- Devflow scale: standard-light
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- `diagnosis-eval-harness` created deterministic fixture evaluation.
|
||||||
|
- `expand-diagnosis-eval-fixtures` created a complete saved baseline.
|
||||||
|
- This change compares new reports against that baseline.
|
||||||
|
|
||||||
|
## Key Decisions
|
||||||
|
|
||||||
|
- Decision: Diff report DTOs instead of raw traces.
|
||||||
|
- Reason: the report is the stable contract for regression review.
|
||||||
|
|
||||||
|
- Decision: Use deterministic code rules instead of LLM-as-judge.
|
||||||
|
- Reason: baseline regression checks should be repeatable and explainable.
|
||||||
|
|
||||||
|
- Decision: Output both JSON and Markdown.
|
||||||
|
- Reason: JSON supports automation; Markdown is useful in reviews and interviews.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- Whether a future change should expose this through a CLI or Maven goal.
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Evidence: diagnosis-eval-baseline-diff
|
||||||
|
|
||||||
|
## Evidence Log
|
||||||
|
|
||||||
|
- 2026-07-05: Created slug-based issue `diagnosis-eval-baseline-diff.md`.
|
||||||
|
- 2026-07-05: Created OpenSpec change `diagnosis-eval-baseline-diff`.
|
||||||
|
- 2026-07-05: Added baseline diff DTOs, deterministic comparer, and JSON / Markdown writer.
|
||||||
|
- 2026-07-05: Added sample baseline diff JSON and Markdown reports.
|
||||||
|
- 2026-07-05: Verification passed with `mvn -q "-Dtest=DiagnosisEvalBaselineDiffTest,DiagnosisTraceEvaluatorTest" test`.
|
||||||
|
- 2026-07-05: Verification passed with `mvn -q -DskipTests compile`.
|
||||||
|
- 2026-07-05: Verification passed with `openspec validate diagnosis-eval-baseline-diff --strict`.
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
# Acceptance: mvp-demo-interview-runbook
|
||||||
|
|
||||||
|
## Classification
|
||||||
|
|
||||||
|
standard-light
|
||||||
|
|
||||||
|
## Task Status
|
||||||
|
|
||||||
|
| Task | Status | Notes |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Issue and OpenSpec setup | Done | Created slug-based issue and OpenSpec artifacts. |
|
||||||
|
| Implementation | Done | Added request payload, runnable script, output directory docs, interview walkthrough, and trace checklist. |
|
||||||
|
| Verification | Done | OpenSpec validation passed. |
|
||||||
|
|
||||||
|
## Current State
|
||||||
|
|
||||||
|
- No backend runtime behavior has been changed.
|
||||||
|
- Demo is packaged under `mvp/demo` for interview use.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
### OpenSpec Verification
|
||||||
|
|
||||||
|
- Command: `openspec validate mvp-demo-interview-runbook --strict`
|
||||||
|
- Result: passed
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
# Brief: mvp-demo-interview-runbook
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
Plan C is the interview-facing demo package. The project has the engineering pieces, but needs a single place to run and explain the MVP flow.
|
||||||
|
|
||||||
|
## Goals
|
||||||
|
|
||||||
|
1. Provide a fixed payment-timeout request payload.
|
||||||
|
2. Provide a PowerShell script that runs chat, trace, and feedback.
|
||||||
|
3. Save demo responses under `mvp/demo/output`.
|
||||||
|
4. Add interview walkthrough and trace checklist.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Demo docs and scripts only
|
||||||
|
- Existing local APIs only
|
||||||
|
- Existing `mvp-demo` profile only
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
- No backend code changes
|
||||||
|
- No eval extension
|
||||||
|
- No secret cleanup
|
||||||
|
- No full offline runtime
|
||||||
|
|
||||||
|
## Related OpenSpec
|
||||||
|
|
||||||
|
`openspec/changes/mvp-demo-interview-runbook/`
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
# MVP Demo Interview Runbook Decisions
|
||||||
|
|
||||||
|
## Clarify
|
||||||
|
|
||||||
|
- Entry summary: package existing MVP capabilities into a repeatable interview demo.
|
||||||
|
- Slug: `mvp-demo-interview-runbook`
|
||||||
|
- Devflow scale: standard-light
|
||||||
|
|
||||||
|
## Context
|
||||||
|
|
||||||
|
- Evidence trace and eval baseline work are already done.
|
||||||
|
- The next useful step is not more eval tooling, but a runnable demo path.
|
||||||
|
|
||||||
|
## Key Decisions
|
||||||
|
|
||||||
|
- Decision: Keep this change documentation/script-only.
|
||||||
|
- Reason: Plan C is about demo packaging, not new runtime capability.
|
||||||
|
|
||||||
|
- Decision: Use a stable session id.
|
||||||
|
- Reason: it makes trace lookup and saved output predictable.
|
||||||
|
|
||||||
|
- Decision: Save outputs to `mvp/demo/output`.
|
||||||
|
- Reason: generated artifacts should be easy to review without mixing into source fixtures.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- Whether a later change should add a truly offline stubbed demo mode.
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
# Evidence: mvp-demo-interview-runbook
|
||||||
|
|
||||||
|
## Evidence Log
|
||||||
|
|
||||||
|
- 2026-07-05: Created Plan C demo packaging issue and OpenSpec change.
|
||||||
|
- 2026-07-05: Added fixed payment-timeout request payload.
|
||||||
|
- 2026-07-05: Added PowerShell demo script for chat, trace, and feedback.
|
||||||
|
- 2026-07-05: Added interview walkthrough and trace inspection checklist.
|
||||||
|
- 2026-07-05: Verification passed with `openspec validate mvp-demo-interview-runbook --strict`.
|
||||||
@@ -2,6 +2,13 @@
|
|||||||
|
|
||||||
This demo proves the MVP flow from user question to persisted diagnosis trace.
|
This demo proves the MVP flow from user question to persisted diagnosis trace.
|
||||||
|
|
||||||
|
For interview use, start with:
|
||||||
|
|
||||||
|
- `interview-walkthrough.md` for the talk track
|
||||||
|
- `trace-inspection-checklist.md` for fields to inspect
|
||||||
|
- `scripts/run-payment-timeout-demo.ps1` for the runnable local demo
|
||||||
|
- `requests/payment-timeout-chat.json` for the fixed request payload
|
||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
- MySQL, Redis, Milvus/Zilliz, and LLM/embedding configuration are available through the current project configuration.
|
- MySQL, Redis, Milvus/Zilliz, and LLM/embedding configuration are available through the current project configuration.
|
||||||
@@ -22,6 +29,22 @@ http://localhost:9900
|
|||||||
|
|
||||||
## 1. Run Chat Diagnosis
|
## 1. Run Chat Diagnosis
|
||||||
|
|
||||||
|
Fast path:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1
|
||||||
|
```
|
||||||
|
|
||||||
|
This writes:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mvp/demo/output/chat-response.json
|
||||||
|
mvp/demo/output/trace-response.json
|
||||||
|
mvp/demo/output/feedback-response.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Manual path:
|
||||||
|
|
||||||
```powershell
|
```powershell
|
||||||
$sessionId = "mvp-demo-payment-timeout-001"
|
$sessionId = "mvp-demo-payment-timeout-001"
|
||||||
$body = @{
|
$body = @{
|
||||||
|
|||||||
@@ -0,0 +1,146 @@
|
|||||||
|
# Interview Walkthrough: MVP Diagnosis Agent
|
||||||
|
|
||||||
|
This walkthrough is the Plan C demo story. It is meant for a short Agent Engineer interview, not as exhaustive system documentation.
|
||||||
|
|
||||||
|
## 30-Second Summary
|
||||||
|
|
||||||
|
```text
|
||||||
|
This is an enterprise diagnosis Agent MVP.
|
||||||
|
It takes a payment-timeout question, plans the investigation, calls evidence tools,
|
||||||
|
checks the answer through a verifier, persists the full trace, and accepts feedback.
|
||||||
|
```
|
||||||
|
|
||||||
|
The important claim is not "the model answered once." The claim is:
|
||||||
|
|
||||||
|
```text
|
||||||
|
The system can show what evidence was used, how the answer was checked, and how to replay the session.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Demo Flow
|
||||||
|
|
||||||
|
1. Start the service with the `mvp-demo` profile.
|
||||||
|
2. Run the fixed payment-timeout request.
|
||||||
|
3. Open `mvp/demo/output/chat-response.json`.
|
||||||
|
4. Open `mvp/demo/output/trace-response.json`.
|
||||||
|
5. Point to evidence tools and verifier evaluation.
|
||||||
|
6. Submit feedback and show it is attached to the same session.
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
Start service:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
mvn spring-boot:run "-Dspring-boot.run.profiles=mvp-demo"
|
||||||
|
```
|
||||||
|
|
||||||
|
Run the demo from another terminal:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1
|
||||||
|
```
|
||||||
|
|
||||||
|
Optional custom session:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
powershell -ExecutionPolicy Bypass -File mvp/demo/scripts/run-payment-timeout-demo.ps1 -SessionId "mvp-demo-payment-timeout-002"
|
||||||
|
```
|
||||||
|
|
||||||
|
## What To Show
|
||||||
|
|
||||||
|
### 1. User-Facing Answer
|
||||||
|
|
||||||
|
File:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mvp/demo/output/chat-response.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Say:
|
||||||
|
|
||||||
|
```text
|
||||||
|
This is the answer the user sees. The session id is stable, so I can trace this exact answer later.
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Evidence Trace
|
||||||
|
|
||||||
|
File:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mvp/demo/output/trace-response.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Say:
|
||||||
|
|
||||||
|
```text
|
||||||
|
This is the important Agent engineering part.
|
||||||
|
I can inspect which tools were called, what inputs they received,
|
||||||
|
whether they succeeded, and what evidence preview was persisted.
|
||||||
|
```
|
||||||
|
|
||||||
|
Point to:
|
||||||
|
|
||||||
|
- `data.toolInvocations[*].toolName`
|
||||||
|
- `data.toolInvocations[*].inputParams`
|
||||||
|
- `data.toolInvocations[*].outputPreview`
|
||||||
|
- `data.toolInvocations[*].success`
|
||||||
|
|
||||||
|
### 3. Verifier / Self-Evaluation
|
||||||
|
|
||||||
|
Point to:
|
||||||
|
|
||||||
|
- `data.session.selfEvaluation`
|
||||||
|
- `data.summary.hasVerifierEvaluation`
|
||||||
|
|
||||||
|
Say:
|
||||||
|
|
||||||
|
```text
|
||||||
|
The final answer is not just raw Executor output.
|
||||||
|
It is checked by a verifier or self-evaluation layer using the persisted trace.
|
||||||
|
That lets the system return PASS, LOW_CONFID, or REJECT-style behavior instead of pretending all answers are equally certain.
|
||||||
|
```
|
||||||
|
|
||||||
|
### 4. Feedback Loop
|
||||||
|
|
||||||
|
File:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mvp/demo/output/feedback-response.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Then re-query trace if needed.
|
||||||
|
|
||||||
|
Say:
|
||||||
|
|
||||||
|
```text
|
||||||
|
Feedback is attached to the same diagnosis session.
|
||||||
|
That makes it possible to mine useful / not useful cases later.
|
||||||
|
```
|
||||||
|
|
||||||
|
### 5. Regression Story
|
||||||
|
|
||||||
|
Mention, do not deep dive unless asked:
|
||||||
|
|
||||||
|
```text
|
||||||
|
For repeatability, I also built an offline eval baseline.
|
||||||
|
The demo proves the runtime trace; the eval baseline proves fixed-case regression.
|
||||||
|
The two are separate on purpose: demo for human review, eval for automated signal.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Strong Interview Framing
|
||||||
|
|
||||||
|
Use this phrasing:
|
||||||
|
|
||||||
|
```text
|
||||||
|
I focused on the Agent engineering surface:
|
||||||
|
traceability, evidence persistence, verifier gating, feedback, and regression checks.
|
||||||
|
The model answer is only one part of the system.
|
||||||
|
The more important part is whether we can audit and improve the answer after it is produced.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Known Limits To Say Proactively
|
||||||
|
|
||||||
|
```text
|
||||||
|
This MVP still depends on configured MySQL, Redis, Milvus, and model credentials.
|
||||||
|
The mvp-demo profile mocks logs and metrics, but not the full application runtime.
|
||||||
|
Secret cleanup and fully isolated default tests are separate production-hardening tasks.
|
||||||
|
```
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Demo Output
|
||||||
|
|
||||||
|
This directory is the default output location for local demo responses.
|
||||||
|
|
||||||
|
Generated files are intentionally ignored by Git:
|
||||||
|
|
||||||
|
- `chat-response.json`
|
||||||
|
- `trace-response.json`
|
||||||
|
- `feedback-response.json`
|
||||||
|
|
||||||
|
Keep this README so the directory exists in the repository.
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"Id": "mvp-demo-payment-timeout-001",
|
||||||
|
"Question": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。"
|
||||||
|
}
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
param(
|
||||||
|
[string]$BaseUrl = "http://localhost:9900",
|
||||||
|
[string]$SessionId = "mvp-demo-payment-timeout-001",
|
||||||
|
[string]$RequestFile = "$PSScriptRoot/../requests/payment-timeout-chat.json",
|
||||||
|
[string]$OutputDir = "$PSScriptRoot/../output"
|
||||||
|
)
|
||||||
|
|
||||||
|
$ErrorActionPreference = "Stop"
|
||||||
|
|
||||||
|
New-Item -ItemType Directory -Force -Path $OutputDir | Out-Null
|
||||||
|
|
||||||
|
$request = Get-Content -Raw -Encoding UTF8 -Path $RequestFile | ConvertFrom-Json
|
||||||
|
$request.Id = $SessionId
|
||||||
|
$body = $request | ConvertTo-Json -Depth 8
|
||||||
|
|
||||||
|
Write-Host "Running payment-timeout chat demo..."
|
||||||
|
Write-Host "BaseUrl: $BaseUrl"
|
||||||
|
Write-Host "SessionId: $SessionId"
|
||||||
|
|
||||||
|
$chat = Invoke-RestMethod `
|
||||||
|
-Method Post `
|
||||||
|
-Uri "$BaseUrl/api/chat" `
|
||||||
|
-ContentType "application/json; charset=utf-8" `
|
||||||
|
-Body $body
|
||||||
|
|
||||||
|
$chat | ConvertTo-Json -Depth 20 | Set-Content -Encoding UTF8 -Path "$OutputDir/chat-response.json"
|
||||||
|
Write-Host "Saved chat response: $OutputDir/chat-response.json"
|
||||||
|
|
||||||
|
$trace = Invoke-RestMethod `
|
||||||
|
-Method Get `
|
||||||
|
-Uri "$BaseUrl/api/diagnosis/$SessionId/trace"
|
||||||
|
|
||||||
|
$trace | ConvertTo-Json -Depth 50 | Set-Content -Encoding UTF8 -Path "$OutputDir/trace-response.json"
|
||||||
|
Write-Host "Saved trace response: $OutputDir/trace-response.json"
|
||||||
|
|
||||||
|
$feedbackBody = @{
|
||||||
|
sessionId = $SessionId
|
||||||
|
feedback = "useful"
|
||||||
|
} | ConvertTo-Json
|
||||||
|
|
||||||
|
$feedback = Invoke-RestMethod `
|
||||||
|
-Method Post `
|
||||||
|
-Uri "$BaseUrl/api/feedback" `
|
||||||
|
-ContentType "application/json; charset=utf-8" `
|
||||||
|
-Body $feedbackBody
|
||||||
|
|
||||||
|
$feedback | ConvertTo-Json -Depth 20 | Set-Content -Encoding UTF8 -Path "$OutputDir/feedback-response.json"
|
||||||
|
Write-Host "Saved feedback response: $OutputDir/feedback-response.json"
|
||||||
|
|
||||||
|
Write-Host ""
|
||||||
|
Write-Host "Demo completed. Review:"
|
||||||
|
Write-Host "- mvp/demo/output/chat-response.json"
|
||||||
|
Write-Host "- mvp/demo/output/trace-response.json"
|
||||||
|
Write-Host "- mvp/demo/output/feedback-response.json"
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
# Trace Inspection Checklist
|
||||||
|
|
||||||
|
Use this checklist after running `scripts/run-payment-timeout-demo.ps1`.
|
||||||
|
|
||||||
|
## Session
|
||||||
|
|
||||||
|
| JSON path | What to check | Interview point |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `data.session.sessionId` | Matches `mvp-demo-payment-timeout-001` | One session id connects chat, tools, verifier, feedback, and trace. |
|
||||||
|
| `data.session.query` | Contains the payment-timeout question | The trace records the original user intent. |
|
||||||
|
| `data.session.answer` | Contains the final diagnosis answer | The final answer is not detached from the trace. |
|
||||||
|
| `data.session.selfEvaluation` | Contains verifier or rule evaluation | The answer has a quality gate, not just raw model output. |
|
||||||
|
| `data.session.feedback` | Becomes `useful` after feedback submission | User feedback is attached to the same diagnosis session. |
|
||||||
|
|
||||||
|
## Agent Steps
|
||||||
|
|
||||||
|
| JSON path | What to check | Interview point |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `data.steps[*].agentName` | Planner / Executor / Verifier or equivalent step names | The flow is decomposed into inspectable Agent steps. |
|
||||||
|
| `data.steps[*].thought` | High-level step reasoning where available | Internal reasoning is auditable without relying only on final text. |
|
||||||
|
| `data.steps[*].durationMs` | Step duration | The trace can support cost and latency review. |
|
||||||
|
| `data.steps[*].tokenCount` | Token count where available | The trace can support model-cost review. |
|
||||||
|
|
||||||
|
## Tool Evidence
|
||||||
|
|
||||||
|
| JSON path | What to check | Interview point |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `data.toolInvocations[*].toolName` | Includes evidence tools such as `lookup_knowledge`, `query_logs`, `query_metrics` | The Agent uses tools, not unsupported guesses. |
|
||||||
|
| `data.toolInvocations[*].inputParams` | Shows what each tool was asked | Inputs are inspectable for debugging and audit. |
|
||||||
|
| `data.toolInvocations[*].outputPreview` | Shows a bounded preview of evidence | Evidence is preserved without dumping huge payloads. |
|
||||||
|
| `data.toolInvocations[*].success` | Distinguishes success from failure | Tool failure is visible to verifier and reviewers. |
|
||||||
|
| `data.toolInvocations[*].retrievalDetails` | Shows retrieval metadata when available | Retrieval quality can be reviewed after the fact. |
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
| JSON path | What to check | Interview point |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `data.summary.persistedStepCount` | Step rows were persisted | The trace is backed by storage, not only response memory. |
|
||||||
|
| `data.summary.persistedToolCallCount` | Tool rows were persisted | Evidence survives the request. |
|
||||||
|
| `data.summary.hasVerifierEvaluation` | Verifier evaluation exists | The final answer passed through a quality gate. |
|
||||||
|
| `data.summary.hasFeedback` | Feedback exists after feedback step | Human feedback closes the loop. |
|
||||||
|
|
||||||
|
## What Good Looks Like
|
||||||
|
|
||||||
|
```text
|
||||||
|
same session id
|
||||||
|
-> final answer
|
||||||
|
-> persisted agent steps
|
||||||
|
-> persisted evidence tool calls
|
||||||
|
-> verifier/self-evaluation
|
||||||
|
-> feedback attached to the same session
|
||||||
|
```
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# Diagnosis Eval Harness
|
||||||
|
|
||||||
|
This folder contains the first fixed-case evaluation set for the MVP diagnosis Agent.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
- Case definitions: `cases/diagnosis-cases.json`
|
||||||
|
- Offline trace fixtures: `fixtures/*.json`
|
||||||
|
- Field definitions: `schema.md`
|
||||||
|
- Baseline reports: `reports/baseline-report.json` and `reports/baseline-report.md`
|
||||||
|
- Baseline diff sample: `reports/baseline-diff-sample.json` and `reports/baseline-diff-sample.md`
|
||||||
|
- Evaluator implementation: `DiagnosisTraceEvaluator`
|
||||||
|
- Report writer: `DiagnosisEvalReportWriter`
|
||||||
|
|
||||||
|
## Current Mode
|
||||||
|
|
||||||
|
The first version evaluates saved trace fixtures. It does not start the application and does not require MySQL, Redis, Milvus, or a real LLM.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
Run the focused evaluator test:
|
||||||
|
|
||||||
|
```powershell
|
||||||
|
mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test
|
||||||
|
```
|
||||||
|
|
||||||
|
The committed baseline report represents the current fixed fixture set:
|
||||||
|
|
||||||
|
```text
|
||||||
|
5 fixed cases
|
||||||
|
5 passing fixture evaluations
|
||||||
|
2 PASS verdicts
|
||||||
|
3 LOW_CONFID verdicts
|
||||||
|
```
|
||||||
|
|
||||||
|
When fixtures or evaluator rules change, regenerate the report from the same case file and fixture directory, then update both JSON and Markdown outputs together.
|
||||||
|
|
||||||
|
## Interview Story
|
||||||
|
|
||||||
|
The harness gives the MVP a repeatable baseline:
|
||||||
|
|
||||||
|
```text
|
||||||
|
fixed diagnosis case
|
||||||
|
-> saved or runtime trace
|
||||||
|
-> rule-based trace validation
|
||||||
|
-> JSON / Markdown report
|
||||||
|
-> regression signal for prompts, tools, retrieval, and verifier behavior
|
||||||
|
```
|
||||||
|
|
||||||
|
## Baseline Diff
|
||||||
|
|
||||||
|
Baseline diff compares a current report against `reports/baseline-report.json`.
|
||||||
|
|
||||||
|
```text
|
||||||
|
baseline report
|
||||||
|
current report
|
||||||
|
-> deterministic diff
|
||||||
|
-> regressions, improvements, and changed signals
|
||||||
|
```
|
||||||
|
|
||||||
|
Use it to answer: did a prompt, tool, retrieval, or verifier change make the Agent worse than the fixed baseline?
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
[
|
||||||
|
{
|
||||||
|
"id": "payment-timeout",
|
||||||
|
"title": "Payment API timeout",
|
||||||
|
"question": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。",
|
||||||
|
"traceFixture": "payment-timeout-pass.json",
|
||||||
|
"expectedRootCauseKeywords": ["支付", "超时", "连接池"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["lookup_knowledge", "query_logs", "query_metrics"],
|
||||||
|
"allowedVerdicts": ["PASS", "LOW_CONFID"],
|
||||||
|
"forbiddenAnswerKeywords": ["无证据确定"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "mysql-pool-exhausted",
|
||||||
|
"title": "MySQL connection pool exhausted",
|
||||||
|
"question": "订单服务大量请求超时,请判断是否和 MySQL 连接池有关。",
|
||||||
|
"traceFixture": "mysql-pool-low-confid.json",
|
||||||
|
"expectedRootCauseKeywords": ["mysql", "连接池", "超时"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["lookup_knowledge", "query_logs"],
|
||||||
|
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["已经完全确认"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "redis-timeout",
|
||||||
|
"title": "Redis timeout",
|
||||||
|
"question": "支付服务出现 Redis 连接超时,请定位可能原因。",
|
||||||
|
"traceFixture": "redis-timeout-low-confid.json",
|
||||||
|
"expectedRootCauseKeywords": ["redis", "超时"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["query_logs"],
|
||||||
|
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["无需进一步排查"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "slow-response",
|
||||||
|
"title": "Slow response",
|
||||||
|
"question": "用户服务 P99 响应时间升高,请结合指标和日志分析。",
|
||||||
|
"traceFixture": "slow-response-pass.json",
|
||||||
|
"expectedRootCauseKeywords": ["p99", "慢响应"],
|
||||||
|
"minKeywordMatches": 1,
|
||||||
|
"requiredEvidenceTools": ["query_metrics", "query_logs"],
|
||||||
|
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["没有风险"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "jvm-memory-risk",
|
||||||
|
"title": "JVM memory risk",
|
||||||
|
"question": "订单服务内存使用率过高,请判断是否存在 OOM 风险。",
|
||||||
|
"traceFixture": "jvm-memory-risk-low-confid.json",
|
||||||
|
"expectedRootCauseKeywords": ["jvm", "内存", "oom"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["query_metrics", "query_logs"],
|
||||||
|
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||||
|
"forbiddenAnswerKeywords": ["可以忽略"]
|
||||||
|
}
|
||||||
|
]
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-jvm-memory-risk",
|
||||||
|
"query": "订单服务内存使用率过高,请判断是否存在 OOM 风险。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 53000,
|
||||||
|
"toolCallCount": 2,
|
||||||
|
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\n订单服务存在 JVM 内存风险,但还不能完全确认会发生 OOM。指标显示 heap 使用率持续高于 88%,日志出现多次 Full GC 和 allocation pressure,需要继续观察对象增长来源并检查最近发布。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "LOW_CONFID",
|
||||||
|
"groundedness_score": 0.52,
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "indirect"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"sessionId": "eval-jvm-memory-risk",
|
||||||
|
"toolName": "query_metrics",
|
||||||
|
"success": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"sessionId": "eval-jvm-memory-risk",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 2,
|
||||||
|
"returnedToolCallCount": 2,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-mysql-pool",
|
||||||
|
"query": "订单服务大量请求超时,请判断是否和 MySQL 连接池有关。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 51000,
|
||||||
|
"toolCallCount": 2,
|
||||||
|
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\nMySQL 连接池可能参与了本次超时问题。日志中出现 connection pool exhausted,但当前缺少完整指标证据,因此只能作为低置信结论处理。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "LOW_CONFID",
|
||||||
|
"groundedness_score": 0.48,
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "lookup_knowledge",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"sessionId": "eval-mysql-pool",
|
||||||
|
"toolName": "lookup_knowledge",
|
||||||
|
"success": true,
|
||||||
|
"relevanceLevel": "PRECISE"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"sessionId": "eval-mysql-pool",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 2,
|
||||||
|
"returnedToolCallCount": 2,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-payment-timeout",
|
||||||
|
"query": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 42000,
|
||||||
|
"toolCallCount": 3,
|
||||||
|
"answer": "支付接口超时与连接池等待有关。知识库说明支付超时需要同时检查连接池、日志和指标;日志出现 connection pool exhausted;指标显示支付服务延迟升高。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "PASS",
|
||||||
|
"groundedness_score": 0.86,
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "lookup_knowledge",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"sessionId": "eval-payment-timeout",
|
||||||
|
"toolName": "lookup_knowledge",
|
||||||
|
"success": true,
|
||||||
|
"relevanceLevel": "PRECISE"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"sessionId": "eval-payment-timeout",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"success": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 3,
|
||||||
|
"sessionId": "eval-payment-timeout",
|
||||||
|
"toolName": "query_metrics",
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 3,
|
||||||
|
"returnedToolCallCount": 3,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-redis-timeout",
|
||||||
|
"query": "支付服务出现 Redis 连接超时,请定位可能原因。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 36000,
|
||||||
|
"toolCallCount": 1,
|
||||||
|
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\nRedis 连接超时可能和支付服务到 Redis 的网络抖动或连接池等待有关。日志中出现 redis timeout 和 command timeout 记录,但当前缺少指标侧证据,因此只能作为低置信结论处理。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "LOW_CONFID",
|
||||||
|
"groundedness_score": 0.46,
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"sessionId": "eval-redis-timeout",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 2,
|
||||||
|
"returnedStepCount": 2,
|
||||||
|
"persistedToolCallCount": 1,
|
||||||
|
"returnedToolCallCount": 1,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
{
|
||||||
|
"session": {
|
||||||
|
"sessionId": "eval-slow-response",
|
||||||
|
"query": "用户服务 P99 响应时间升高,请结合指标和日志分析。",
|
||||||
|
"status": "SUCCESS",
|
||||||
|
"agentFlow": "CHAT",
|
||||||
|
"totalDurationMs": 47000,
|
||||||
|
"toolCallCount": 2,
|
||||||
|
"answer": "用户服务 P99 升高主要表现为慢响应。指标显示 P99 latency 从 280ms 上升到 1800ms,日志中同时出现 slow request 和 downstream timeout,因此优先排查下游依赖耗时和线程池排队。",
|
||||||
|
"selfEvaluation": {
|
||||||
|
"verifier_evaluation": {
|
||||||
|
"verdict": "PASS",
|
||||||
|
"groundedness_score": 0.78,
|
||||||
|
"tool_trace_summary": [
|
||||||
|
{
|
||||||
|
"tool_name": "query_metrics",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"tool_name": "query_logs",
|
||||||
|
"success": true,
|
||||||
|
"evidence_level": "direct"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"steps": [],
|
||||||
|
"toolInvocations": [
|
||||||
|
{
|
||||||
|
"id": 1,
|
||||||
|
"sessionId": "eval-slow-response",
|
||||||
|
"toolName": "query_metrics",
|
||||||
|
"success": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": 2,
|
||||||
|
"sessionId": "eval-slow-response",
|
||||||
|
"toolName": "query_logs",
|
||||||
|
"success": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"summary": {
|
||||||
|
"persistedStepCount": 3,
|
||||||
|
"returnedStepCount": 3,
|
||||||
|
"persistedToolCallCount": 2,
|
||||||
|
"returnedToolCallCount": 2,
|
||||||
|
"hasVerifierEvaluation": true,
|
||||||
|
"hasFeedback": false
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,85 @@
|
|||||||
|
{
|
||||||
|
"baselineTotalCases" : 5,
|
||||||
|
"currentTotalCases" : 5,
|
||||||
|
"baselinePassedCases" : 5,
|
||||||
|
"currentPassedCases" : 4,
|
||||||
|
"baselinePassRate" : 1.0,
|
||||||
|
"currentPassRate" : 0.8,
|
||||||
|
"regressionCount" : 6,
|
||||||
|
"improvementCount" : 0,
|
||||||
|
"changedCount" : 2,
|
||||||
|
"hasRegression" : true,
|
||||||
|
"items" : [ {
|
||||||
|
"type" : "REGRESSION",
|
||||||
|
"scope" : "aggregate",
|
||||||
|
"caseId" : null,
|
||||||
|
"metric" : "passRate",
|
||||||
|
"baselineValue" : "1.0",
|
||||||
|
"currentValue" : "0.8",
|
||||||
|
"delta" : -0.19999999999999996,
|
||||||
|
"message" : "passRate changed"
|
||||||
|
}, {
|
||||||
|
"type" : "REGRESSION",
|
||||||
|
"scope" : "aggregate",
|
||||||
|
"caseId" : null,
|
||||||
|
"metric" : "averageToolCallCount",
|
||||||
|
"baselineValue" : "2.0",
|
||||||
|
"currentValue" : "3.0",
|
||||||
|
"delta" : 1.0,
|
||||||
|
"message" : "averageToolCallCount changed"
|
||||||
|
}, {
|
||||||
|
"type" : "CHANGED",
|
||||||
|
"scope" : "aggregate",
|
||||||
|
"caseId" : null,
|
||||||
|
"metric" : "verdictDistribution.LOW_CONFID",
|
||||||
|
"baselineValue" : "3",
|
||||||
|
"currentValue" : "2",
|
||||||
|
"delta" : -1.0,
|
||||||
|
"message" : "verdict count changed for LOW_CONFID"
|
||||||
|
}, {
|
||||||
|
"type" : "CHANGED",
|
||||||
|
"scope" : "aggregate",
|
||||||
|
"caseId" : null,
|
||||||
|
"metric" : "verdictDistribution.REJECT",
|
||||||
|
"baselineValue" : "0",
|
||||||
|
"currentValue" : "1",
|
||||||
|
"delta" : 1.0,
|
||||||
|
"message" : "verdict count changed for REJECT"
|
||||||
|
}, {
|
||||||
|
"type" : "REGRESSION",
|
||||||
|
"scope" : "case",
|
||||||
|
"caseId" : "redis-timeout",
|
||||||
|
"metric" : "passed",
|
||||||
|
"baselineValue" : "true",
|
||||||
|
"currentValue" : "false",
|
||||||
|
"delta" : null,
|
||||||
|
"message" : "redis-timeout pass state changed"
|
||||||
|
}, {
|
||||||
|
"type" : "REGRESSION",
|
||||||
|
"scope" : "case",
|
||||||
|
"caseId" : "redis-timeout",
|
||||||
|
"metric" : "verdict",
|
||||||
|
"baselineValue" : "LOW_CONFID",
|
||||||
|
"currentValue" : "REJECT",
|
||||||
|
"delta" : -1.0,
|
||||||
|
"message" : "redis-timeout verdict changed"
|
||||||
|
}, {
|
||||||
|
"type" : "REGRESSION",
|
||||||
|
"scope" : "case",
|
||||||
|
"caseId" : "redis-timeout",
|
||||||
|
"metric" : "matchedKeywordCount",
|
||||||
|
"baselineValue" : "2",
|
||||||
|
"currentValue" : "1",
|
||||||
|
"delta" : -1.0,
|
||||||
|
"message" : "redis-timeout matchedKeywordCount changed"
|
||||||
|
}, {
|
||||||
|
"type" : "REGRESSION",
|
||||||
|
"scope" : "case",
|
||||||
|
"caseId" : "redis-timeout",
|
||||||
|
"metric" : "evidenceCoverage.query_logs",
|
||||||
|
"baselineValue" : "true",
|
||||||
|
"currentValue" : "false",
|
||||||
|
"delta" : null,
|
||||||
|
"message" : "redis-timeout evidence coverage changed for query_logs"
|
||||||
|
} ]
|
||||||
|
}
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
# Diagnosis Eval Baseline Diff
|
||||||
|
|
||||||
|
- Baseline pass rate: 100.00%
|
||||||
|
- Current pass rate: 80.00%
|
||||||
|
- Baseline passed cases: 5/5
|
||||||
|
- Current passed cases: 4/5
|
||||||
|
- Regressions: 6
|
||||||
|
- Improvements: 0
|
||||||
|
- Other changes: 2
|
||||||
|
|
||||||
|
## Diff Items
|
||||||
|
|
||||||
|
| Type | Scope | Case | Metric | Baseline | Current | Delta | Message |
|
||||||
|
| --- | --- | --- | --- | --- | --- | ---: | --- |
|
||||||
|
| REGRESSION | aggregate | - | passRate | 1.0 | 0.8 | -0.200 | passRate changed |
|
||||||
|
| REGRESSION | aggregate | - | averageToolCallCount | 2.0 | 3.0 | 1.000 | averageToolCallCount changed |
|
||||||
|
| CHANGED | aggregate | - | verdictDistribution.LOW_CONFID | 3 | 2 | -1.000 | verdict count changed for LOW_CONFID |
|
||||||
|
| CHANGED | aggregate | - | verdictDistribution.REJECT | 0 | 1 | 1.000 | verdict count changed for REJECT |
|
||||||
|
| REGRESSION | case | redis-timeout | passed | true | false | - | redis-timeout pass state changed |
|
||||||
|
| REGRESSION | case | redis-timeout | verdict | LOW_CONFID | REJECT | -1.000 | redis-timeout verdict changed |
|
||||||
|
| REGRESSION | case | redis-timeout | matchedKeywordCount | 2 | 1 | -1.000 | redis-timeout matchedKeywordCount changed |
|
||||||
|
| REGRESSION | case | redis-timeout | evidenceCoverage.query_logs | true | false | - | redis-timeout evidence coverage changed for query_logs |
|
||||||
@@ -0,0 +1,82 @@
|
|||||||
|
{
|
||||||
|
"totalCases" : 5,
|
||||||
|
"passedCases" : 5,
|
||||||
|
"passRate" : 1.0,
|
||||||
|
"verdictDistribution" : {
|
||||||
|
"PASS" : 2,
|
||||||
|
"LOW_CONFID" : 3
|
||||||
|
},
|
||||||
|
"averageToolCallCount" : 2.0,
|
||||||
|
"averageDurationMs" : 45800.0,
|
||||||
|
"results" : [ {
|
||||||
|
"caseId" : "payment-timeout",
|
||||||
|
"title" : "Payment API timeout",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "PASS",
|
||||||
|
"matchedKeywordCount" : 3,
|
||||||
|
"requiredKeywordCount" : 3,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"lookup_knowledge" : true,
|
||||||
|
"query_logs" : true,
|
||||||
|
"query_metrics" : true
|
||||||
|
},
|
||||||
|
"toolCallCount" : 3,
|
||||||
|
"durationMs" : 42000
|
||||||
|
}, {
|
||||||
|
"caseId" : "mysql-pool-exhausted",
|
||||||
|
"title" : "MySQL connection pool exhausted",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "LOW_CONFID",
|
||||||
|
"matchedKeywordCount" : 3,
|
||||||
|
"requiredKeywordCount" : 3,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"lookup_knowledge" : true,
|
||||||
|
"query_logs" : true
|
||||||
|
},
|
||||||
|
"toolCallCount" : 2,
|
||||||
|
"durationMs" : 51000
|
||||||
|
}, {
|
||||||
|
"caseId" : "redis-timeout",
|
||||||
|
"title" : "Redis timeout",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "LOW_CONFID",
|
||||||
|
"matchedKeywordCount" : 2,
|
||||||
|
"requiredKeywordCount" : 2,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_logs" : true
|
||||||
|
},
|
||||||
|
"toolCallCount" : 1,
|
||||||
|
"durationMs" : 36000
|
||||||
|
}, {
|
||||||
|
"caseId" : "slow-response",
|
||||||
|
"title" : "Slow response",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "PASS",
|
||||||
|
"matchedKeywordCount" : 2,
|
||||||
|
"requiredKeywordCount" : 2,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_metrics" : true,
|
||||||
|
"query_logs" : true
|
||||||
|
},
|
||||||
|
"toolCallCount" : 2,
|
||||||
|
"durationMs" : 47000
|
||||||
|
}, {
|
||||||
|
"caseId" : "jvm-memory-risk",
|
||||||
|
"title" : "JVM memory risk",
|
||||||
|
"passed" : true,
|
||||||
|
"failedChecks" : [ ],
|
||||||
|
"verdict" : "LOW_CONFID",
|
||||||
|
"matchedKeywordCount" : 3,
|
||||||
|
"requiredKeywordCount" : 3,
|
||||||
|
"evidenceCoverage" : {
|
||||||
|
"query_metrics" : true,
|
||||||
|
"query_logs" : true
|
||||||
|
},
|
||||||
|
"toolCallCount" : 2,
|
||||||
|
"durationMs" : 53000
|
||||||
|
} ]
|
||||||
|
}
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
# Diagnosis Eval Report
|
||||||
|
|
||||||
|
- Total cases: 5
|
||||||
|
- Passed cases: 5
|
||||||
|
- Pass rate: 100.00%
|
||||||
|
- Average tool calls: 2.00
|
||||||
|
- Average duration ms: 45800.00
|
||||||
|
|
||||||
|
## Verdict Distribution
|
||||||
|
|
||||||
|
- PASS: 2
|
||||||
|
- LOW_CONFID: 3
|
||||||
|
|
||||||
|
## Cases
|
||||||
|
|
||||||
|
| Case | Result | Verdict | Keywords | Tool Calls | Duration ms | Failed Checks |
|
||||||
|
| --- | --- | --- | --- | ---: | ---: | --- |
|
||||||
|
| payment-timeout | PASS | PASS | 3/3 | 3 | 42000 | - |
|
||||||
|
| mysql-pool-exhausted | PASS | LOW_CONFID | 3/3 | 2 | 51000 | - |
|
||||||
|
| redis-timeout | PASS | LOW_CONFID | 2/2 | 1 | 36000 | - |
|
||||||
|
| slow-response | PASS | PASS | 2/2 | 2 | 47000 | - |
|
||||||
|
| jvm-memory-risk | PASS | LOW_CONFID | 3/3 | 2 | 53000 | - |
|
||||||
@@ -0,0 +1,202 @@
|
|||||||
|
# Diagnosis Eval Data Schema
|
||||||
|
|
||||||
|
这份文档记录评测基准里的数据结构。口语化理解就是:
|
||||||
|
|
||||||
|
```text
|
||||||
|
用例文件说“我要考什么”
|
||||||
|
trace 文件说“Agent 实际做了什么”
|
||||||
|
评测结果说“这次有没有跑偏”
|
||||||
|
汇总报告说“整体稳定性怎么样”
|
||||||
|
```
|
||||||
|
|
||||||
|
当前这套评测是代码规则判断,不是再调用一个 LLM 来打分。
|
||||||
|
|
||||||
|
## 1. 用例定义
|
||||||
|
|
||||||
|
文件:`mvp/eval/cases/diagnosis-cases.json`
|
||||||
|
|
||||||
|
每一条 case 是一个固定考题,告诉评测器“这个问题应该看哪些点、需要哪些证据、哪些结论可以接受”。
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"id": "payment-timeout",
|
||||||
|
"title": "Payment API timeout",
|
||||||
|
"question": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。",
|
||||||
|
"traceFixture": "payment-timeout-pass.json",
|
||||||
|
"expectedRootCauseKeywords": ["支付", "超时", "连接池"],
|
||||||
|
"minKeywordMatches": 2,
|
||||||
|
"requiredEvidenceTools": ["lookup_knowledge", "query_logs", "query_metrics"],
|
||||||
|
"allowedVerdicts": ["PASS", "LOW_CONFID"],
|
||||||
|
"forbiddenAnswerKeywords": ["无证据确定"]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
字段说明:
|
||||||
|
|
||||||
|
| 字段 | 意思 | 评测器怎么用 |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `id` | 这条用例的唯一名字 | 出现在报告里,方便定位是哪条 case 挂了 |
|
||||||
|
| `title` | 给人看的标题 | 出现在结果里,方便快速理解场景 |
|
||||||
|
| `question` | 要问 Agent 的问题 | fixture 模式下不会真的发送给 Agent,但它记录了这条 case 的原始输入 |
|
||||||
|
| `traceFixture` | 对应的 trace 文件名 | 评测器会去 `fixtures/` 目录加载这个文件 |
|
||||||
|
| `expectedRootCauseKeywords` | 最终回答里希望看到的关键点 | 评测器会在 `session.answer` 里做关键词命中检查 |
|
||||||
|
| `minKeywordMatches` | 至少要命中几个关键词 | 命中数低于这个值,就认为根因覆盖不够 |
|
||||||
|
| `requiredEvidenceTools` | 这条 case 至少应该用到哪些证据工具 | 评测器会检查 trace 里是否出现这些工具 |
|
||||||
|
| `allowedVerdicts` | Verifier 允许给出的结论 | 比如 `PASS` 或 `LOW_CONFID`,不在列表里就失败 |
|
||||||
|
| `forbiddenAnswerKeywords` | 回答里不应该出现的危险说法 | 命中这些词,说明回答可能过度自信或不符合降级策略 |
|
||||||
|
|
||||||
|
## 2. Trace Fixture
|
||||||
|
|
||||||
|
目录:`mvp/eval/fixtures/*.json`
|
||||||
|
|
||||||
|
trace fixture 是一次 Agent 运行后的“留痕快照”。评测器不会关心整个 trace 的所有字段,只读取当前能支撑基准判断的字段。
|
||||||
|
|
||||||
|
当前会读取这些字段:
|
||||||
|
|
||||||
|
| Trace 字段 | 意思 | 评测器怎么用 |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `session.answer` | Agent 最终给用户的回答 | 用来检查根因关键词和禁用词 |
|
||||||
|
| `session.totalDurationMs` | 这次运行耗时 | 进入报告,帮助观察性能是否明显变差 |
|
||||||
|
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 对最终回答的判断 | 必须存在,并且要落在 case 的 `allowedVerdicts` 里 |
|
||||||
|
| `session.selfEvaluation.verifier_evaluation.tool_trace_summary[*].tool_name` | Verifier 总结里看到的工具证据 | 用来补充判断证据工具是否出现 |
|
||||||
|
| `toolInvocations[*].toolName` | Agent 实际调用过的工具名 | 用来检查 `requiredEvidenceTools` 是否满足 |
|
||||||
|
| `toolInvocations[*].success` | 工具调用是否成功 | 当前主要保留在 trace 里,后续可以升级成更严格的成功率检查 |
|
||||||
|
|
||||||
|
简单说,trace 里最重要的是三类信息:
|
||||||
|
|
||||||
|
```text
|
||||||
|
最终回答:它说了什么
|
||||||
|
工具证据:它查了什么
|
||||||
|
Verifier:它自己有没有承认这个结论可靠
|
||||||
|
```
|
||||||
|
|
||||||
|
## 3. 单条评测结果
|
||||||
|
|
||||||
|
Java 类型:`DiagnosisEvalResult`
|
||||||
|
|
||||||
|
这是每条 case 跑完之后的判断结果。
|
||||||
|
|
||||||
|
| 字段 | 意思 |
|
||||||
|
| --- | --- |
|
||||||
|
| `caseId` | 对应的 case id |
|
||||||
|
| `title` | case 标题 |
|
||||||
|
| `passed` | 这条 case 是否通过 |
|
||||||
|
| `failedChecks` | 没通过的具体原因,比如缺工具、关键词不够、verdict 不允许 |
|
||||||
|
| `verdict` | 从 trace 里读出来的 Verifier verdict |
|
||||||
|
| `matchedKeywordCount` | 最终回答命中的关键词数量 |
|
||||||
|
| `requiredKeywordCount` | case 定义里一共有多少个关键词 |
|
||||||
|
| `evidenceCoverage` | 每个必需工具是否出现,例如 `{ "query_logs": true }` |
|
||||||
|
| `toolCallCount` | 本次 trace 里工具调用总数 |
|
||||||
|
| `durationMs` | 本次 trace 的耗时 |
|
||||||
|
|
||||||
|
判断通过的口语化规则:
|
||||||
|
|
||||||
|
```text
|
||||||
|
回答要说到关键点
|
||||||
|
该查的证据工具要查到
|
||||||
|
Verifier 的结论要在可接受范围内
|
||||||
|
回答不能出现危险的过度自信表达
|
||||||
|
如果是 REJECT,就必须走降级模板
|
||||||
|
```
|
||||||
|
|
||||||
|
## 4. 汇总报告
|
||||||
|
|
||||||
|
Java 类型:`DiagnosisEvalReport`
|
||||||
|
|
||||||
|
这是整个基准集跑完之后的总结果。
|
||||||
|
|
||||||
|
| 字段 | 意思 |
|
||||||
|
| --- | --- |
|
||||||
|
| `totalCases` | 总共评测了多少条 case |
|
||||||
|
| `passedCases` | 通过了多少条 |
|
||||||
|
| `passRate` | 通过率,范围是 `0.0` 到 `1.0` |
|
||||||
|
| `verdictDistribution` | Verifier verdict 的分布,比如有几个 `PASS`、几个 `LOW_CONFID` |
|
||||||
|
| `averageToolCallCount` | 平均每条 case 调用了多少次工具 |
|
||||||
|
| `averageDurationMs` | 平均耗时 |
|
||||||
|
| `results` | 每条 case 的详细结果列表 |
|
||||||
|
|
||||||
|
## 5. 怎么看这个基准
|
||||||
|
|
||||||
|
这套结构不是为了证明 Agent 永远正确,而是为了在每次改 prompt、工具、检索、Verifier 之后,有一个固定尺子能回答:
|
||||||
|
|
||||||
|
```text
|
||||||
|
以前能过的诊断题,现在还过不过?
|
||||||
|
它是不是少查了某些证据?
|
||||||
|
它是不是变得更自信但证据不足?
|
||||||
|
它是不是开始输出不该说的话?
|
||||||
|
它是不是明显变慢了?
|
||||||
|
```
|
||||||
|
|
||||||
|
所以面试里可以这样讲:
|
||||||
|
|
||||||
|
```text
|
||||||
|
我没有只看一次 demo 效果,而是把典型诊断场景固化成 case。
|
||||||
|
每条 case 都定义预期关键点、必需证据工具和可接受的 verifier 结论。
|
||||||
|
Agent 每次运行会留下 trace,评测器用代码规则读取 trace,输出结构化报告。
|
||||||
|
这样我改 Agent 的时候,可以用同一套基准判断有没有行为回退。
|
||||||
|
```
|
||||||
|
|
||||||
|
## 6. Baseline Diff
|
||||||
|
|
||||||
|
Baseline diff 是拿两份 report 做对比:
|
||||||
|
|
||||||
|
```text
|
||||||
|
baseline report:以前认可的基准结果
|
||||||
|
current report:这次改动后跑出来的新结果
|
||||||
|
diff report:告诉你哪里变好了、哪里变差了、哪里只是变了
|
||||||
|
```
|
||||||
|
|
||||||
|
Java 类型:
|
||||||
|
|
||||||
|
- `DiagnosisEvalDiffReport`
|
||||||
|
- `DiagnosisEvalDiffItem`
|
||||||
|
|
||||||
|
`DiagnosisEvalDiffReport` 字段:
|
||||||
|
|
||||||
|
| 字段 | 意思 |
|
||||||
|
| --- | --- |
|
||||||
|
| `baselineTotalCases` | baseline 里有多少条 case |
|
||||||
|
| `currentTotalCases` | current 里有多少条 case |
|
||||||
|
| `baselinePassedCases` | baseline 通过了多少条 |
|
||||||
|
| `currentPassedCases` | current 通过了多少条 |
|
||||||
|
| `baselinePassRate` | baseline 通过率 |
|
||||||
|
| `currentPassRate` | current 通过率 |
|
||||||
|
| `regressionCount` | 退化项数量 |
|
||||||
|
| `improvementCount` | 改善项数量 |
|
||||||
|
| `changedCount` | 普通变化项数量 |
|
||||||
|
| `hasRegression` | 是否存在退化 |
|
||||||
|
| `items` | 具体 diff 明细 |
|
||||||
|
|
||||||
|
`DiagnosisEvalDiffItem` 字段:
|
||||||
|
|
||||||
|
| 字段 | 意思 |
|
||||||
|
| --- | --- |
|
||||||
|
| `type` | `REGRESSION`、`IMPROVEMENT` 或 `CHANGED` |
|
||||||
|
| `scope` | `aggregate` 表示整体指标,`case` 表示单条 case |
|
||||||
|
| `caseId` | 如果是单条 case 变化,这里记录 case id |
|
||||||
|
| `metric` | 哪个指标变了,比如 `passRate` 或 `evidenceCoverage.query_logs` |
|
||||||
|
| `baselineValue` | baseline 里的值 |
|
||||||
|
| `currentValue` | current 里的值 |
|
||||||
|
| `delta` | 数值变化量;非数值变化为空 |
|
||||||
|
| `message` | 给人看的变化说明 |
|
||||||
|
|
||||||
|
口语化判断规则:
|
||||||
|
|
||||||
|
```text
|
||||||
|
pass rate 下降:退化
|
||||||
|
case 从通过变失败:退化
|
||||||
|
证据工具从有变没有:退化
|
||||||
|
关键词命中变少:退化
|
||||||
|
工具调用或耗时升高:成本上升,记为退化信号
|
||||||
|
verdict 分布变化:记录变化,供人工判断是否符合预期
|
||||||
|
```
|
||||||
|
|
||||||
|
面试里可以这样讲:
|
||||||
|
|
||||||
|
```text
|
||||||
|
我把 baseline report 和当前 report 做结构化 diff。
|
||||||
|
它不是再问 LLM,而是用代码比较固定字段。
|
||||||
|
如果某个 case 从 PASS 变 FAIL,或者 query_logs 证据没了,
|
||||||
|
diff 会直接标成 regression。
|
||||||
|
这样 Agent 改动可以用固定基准做回归判断。
|
||||||
|
```
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
# ISS-005 证据链补齐与降级契约收敛
|
||||||
|
|
||||||
|
**状态**:进行中(sm-flow)
|
||||||
|
**严重程度**:高
|
||||||
|
**发现时间**:2026-07-04
|
||||||
|
**来源**:P1-A 面试打磨项 / 基于 ISS-003 的当前实现复核
|
||||||
|
**关联**:ISS-003(Verifier 证据链、失败路径可验证性)、`chat-verifier-agent`、`mvp-demo-trace-acceptance`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 背景
|
||||||
|
|
||||||
|
当前 MVP 已具备:
|
||||||
|
|
||||||
|
- `lookup_knowledge`、`query_logs`、`query_metrics` 的工具调用落库
|
||||||
|
- Verifier 基于 `tool_trace_summary` 做事实核查
|
||||||
|
- `LOW_CONFID` / `REJECT` 的用户侧降级输出
|
||||||
|
- trace API 可回放 session、agent_step、tool_invocation 和 self_evaluation
|
||||||
|
|
||||||
|
但如果目标是拿这个项目去面试 Agent 工程师,当前实现仍有一个明显短板:
|
||||||
|
|
||||||
|
**证据链已经“有了”,但还没有被收敛成清晰、稳定、可测试的工程契约。**
|
||||||
|
|
||||||
|
这会直接影响三个面试问题的回答质量:
|
||||||
|
|
||||||
|
1. 工具失败时系统会怎样降级?
|
||||||
|
2. Verifier 看到的 evidence 到底是否一致、可审计?
|
||||||
|
3. 这些失败路径和降级行为有没有稳定测试,而不是只靠 runtime 演示?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 当前现状复核
|
||||||
|
|
||||||
|
### 1. 工具落库入口已经存在,但契约不统一
|
||||||
|
|
||||||
|
- `QueryLogsTools` 和 `QueryMetricsTools` 通过 `ToolInvocationRecorder.recordEvidenceTool(...)` 记录 evidence tool 调用。
|
||||||
|
- `LookupKnowledgeTool` 仍保留独立的 `saveToolInvocation(...)` 路径,自己构造 `ToolInvocation` 实体。
|
||||||
|
|
||||||
|
这意味着:
|
||||||
|
|
||||||
|
- evidence tool 的公共字段有一套约定
|
||||||
|
- knowledge retrieval 又有一套定制字段拼装
|
||||||
|
|
||||||
|
两者都能工作,但**没有形成统一的“证据调用记录契约”**。
|
||||||
|
|
||||||
|
### 2. 失败 / 无结果 / 去重命中的语义不够显式
|
||||||
|
|
||||||
|
当前实现里:
|
||||||
|
|
||||||
|
- `query_logs` 未命中时会返回 `success=false` + `"未找到匹配的日志"`
|
||||||
|
- `query_metrics` 失败时会返回 `success=false`
|
||||||
|
- `lookup_knowledge` 去重命中时会返回 `found=false`,但 `tool_invocation.success=true`
|
||||||
|
- `ToolTraceSummaryService` 通过 `success`、`relevanceLevel`、`dedupReason` 等字段做启发式摘要
|
||||||
|
|
||||||
|
这些行为在代码里是分散成立的,但**没有被定义成统一契约**,导致:
|
||||||
|
|
||||||
|
- Verifier 能看到的“失败”和“无证据”边界不够稳定
|
||||||
|
- 评测时难以明确统计哪些是“调用失败”、哪些是“无命中”、哪些是“已检索过”
|
||||||
|
|
||||||
|
### 3. ChatService 的降级路径有实现,但测试矩阵不完整
|
||||||
|
|
||||||
|
`ChatService` 已处理:
|
||||||
|
|
||||||
|
- `verifier_output` 缺失或无法解析 → fallback `LOW_CONFID`
|
||||||
|
- `REJECT` → degraded output
|
||||||
|
- `LOW_CONFID` → disclaimer output
|
||||||
|
|
||||||
|
但目前缺少成体系的专项验证,尤其是:
|
||||||
|
|
||||||
|
- Verifier 输出非法 JSON
|
||||||
|
- evidence tool 查询失败
|
||||||
|
- knowledge lookup 无有效证据
|
||||||
|
- fallback 文案是否只基于 verifier 缺口拼装
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 影响
|
||||||
|
|
||||||
|
- **面试表达弱化**:你能讲“我有 trace”,但还不能很硬地讲“我的失败路径是有契约和测试保护的”。
|
||||||
|
- **评测基础不稳**:后续 P1-B 做 case-based harness 时,统计口径会受 evidence 语义不一致影响。
|
||||||
|
- **Verifier 可审计性打折**:当前实现可用,但 still relies on code convention,而不是一份明确收敛后的工程协议。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 本 issue 目标
|
||||||
|
|
||||||
|
P1-A 只做三件事:
|
||||||
|
|
||||||
|
1. 收敛 evidence tool 的落库契约,让 `lookup_knowledge`、`query_logs`、`query_metrics` 的公共语义一致。
|
||||||
|
2. 明确失败 / 无证据 / 去重 / verifier 非法输出等降级契约,让 `ToolTraceSummaryService` 和 `ChatService` 面向统一状态工作。
|
||||||
|
3. 增加专项离线测试,覆盖证据摘要与关键降级路径。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 范围
|
||||||
|
|
||||||
|
### In scope
|
||||||
|
|
||||||
|
- `ToolInvocationRecorder` 契约增强
|
||||||
|
- `LookupKnowledgeTool` 入库路径收敛
|
||||||
|
- `QueryLogsTools` / `QueryMetricsTools` evidence 语义对齐
|
||||||
|
- `ToolTraceSummaryService` 对失败 / no-hit / mixed evidence 的摘要规则收敛
|
||||||
|
- `ChatService` 对 verifier 非法输出与降级输出的专项测试
|
||||||
|
- 与该 change 直接相关的文档、OpenSpec、devflow 记录
|
||||||
|
|
||||||
|
### Out of scope
|
||||||
|
|
||||||
|
- 不引入新的数据库表或 schema 变更
|
||||||
|
- 不扩展新的 evidence tool
|
||||||
|
- 不做 P1-B 评测集 / harness
|
||||||
|
- 不做前端 trace UI
|
||||||
|
- 不处理敏感配置和默认 `mvn test` 离线化
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 预期结果
|
||||||
|
|
||||||
|
完成后,项目在面试里应能更清楚地表述为:
|
||||||
|
|
||||||
|
```text
|
||||||
|
我不仅把 Agent 的工具调用落到了库里,
|
||||||
|
还把 evidence trace、失败语义和 verifier 降级路径收敛成了稳定契约,
|
||||||
|
并用离线测试覆盖了这些关键失败场景。
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 相关文件
|
||||||
|
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ToolInvocationRecorder.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/agent/tool/QueryLogsTools.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/agent/tool/QueryMetricsTools.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ChatService.java`
|
||||||
|
- `src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java`
|
||||||
|
- `src/test/java/com/superbiz/agent/tool/LookupKnowledgeToolTest.java`
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
# ISS-006 固定诊断评测集与回归 Harness
|
||||||
|
|
||||||
|
**状态**:进行中(sm-flow)
|
||||||
|
**严重程度**:高
|
||||||
|
**发现时间**:2026-07-04
|
||||||
|
**来源**:P1-B 面试打磨项
|
||||||
|
**依赖**:ISS-005 / `evidence-trace-hardening`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 背景
|
||||||
|
|
||||||
|
MVP 已经具备可追溯证据链、Verifier 质量门禁、trace API 和固定 demo 流程。上一阶段 `evidence-trace-hardening` 进一步统一了 evidence tool 的状态语义,让系统能稳定区分:
|
||||||
|
|
||||||
|
- `supported`
|
||||||
|
- `no_evidence`
|
||||||
|
- `deduped`
|
||||||
|
- `failed`
|
||||||
|
|
||||||
|
下一步需要证明 Agent 在一组固定诊断场景下的表现,而不是只依赖单次 demo。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 问题
|
||||||
|
|
||||||
|
当前项目能演示一次支付超时诊断,但还缺少稳定的评测基线:
|
||||||
|
|
||||||
|
- 每次改 prompt、工具、Verifier 或检索逻辑后,无法快速判断是否退化。
|
||||||
|
- 只能人工看 trace,缺少结构化通过 / 失败结果。
|
||||||
|
- 缺少面试时能展示的指标,如 evidence coverage、verdict 分布、工具调用数量和耗时。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 目标
|
||||||
|
|
||||||
|
建立一个轻量的固定 case 评测 harness,用于验证 MVP Agent 的诊断质量和证据链完整性。
|
||||||
|
|
||||||
|
第一版不做 LLM-as-judge,优先做规则化校验:
|
||||||
|
|
||||||
|
- 固定 5 个 MVP 诊断 case
|
||||||
|
- 每个 case 定义 expected root-cause keywords、required evidence tools、allowed verdicts
|
||||||
|
- 基于 trace 结果校验 evidence coverage、verifier evaluation、tool invocation、final answer shape
|
||||||
|
- 输出 JSON 和 Markdown 报告
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 范围
|
||||||
|
|
||||||
|
### In scope
|
||||||
|
|
||||||
|
- 评测 case 定义文件
|
||||||
|
- trace 规则校验器
|
||||||
|
- eval runner 或测试入口
|
||||||
|
- JSON / Markdown 报告输出
|
||||||
|
- demo 文档和 devflow 记录
|
||||||
|
|
||||||
|
### Out of scope
|
||||||
|
|
||||||
|
- 不引入 LLM-as-judge
|
||||||
|
- 不要求完整离线 LLM runtime
|
||||||
|
- 不新增生产 API
|
||||||
|
- 不修改 Chat 主链路
|
||||||
|
- 不修改 evidence trace 运行时语义
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 预期面试表达
|
||||||
|
|
||||||
|
完成后可以这样描述:
|
||||||
|
|
||||||
|
```text
|
||||||
|
我不仅有一个可演示的 Agent,还给它建立了固定 case 的回归评测。
|
||||||
|
每次修改 prompt、工具或 verifier 后,都可以跑同一批诊断 case,
|
||||||
|
检查证据覆盖、verdict 分布、工具调用成本和关键结论是否退化。
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 初始候选 case
|
||||||
|
|
||||||
|
| Case | 目标 |
|
||||||
|
| --- | --- |
|
||||||
|
| payment-timeout | 支付接口超时,验证知识库 + 日志 + 指标证据 |
|
||||||
|
| mysql-pool-exhausted | 数据库连接池耗尽,验证日志和知识库证据 |
|
||||||
|
| redis-timeout | Redis 连接超时,验证日志依赖证据 |
|
||||||
|
| slow-response | P99 响应时间过高,验证指标 + 慢请求日志 |
|
||||||
|
| jvm-memory-risk | JVM 内存 / OOM 风险,验证指标 + 系统事件日志 |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 相关文件
|
||||||
|
|
||||||
|
- `mvp/demo/README.md`
|
||||||
|
- `mvp/demo/payment-timeout-acceptance.md`
|
||||||
|
- `src/main/java/com/superbiz/agent/service/DiagnosisTraceService.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/domain/entity/DiagnosisSession.java`
|
||||||
|
- `src/main/java/com/superbiz/agent/domain/entity/ToolInvocation.java`
|
||||||
|
- `openspec/specs/evidence-trace-hardening/spec.md`
|
||||||
@@ -6,6 +6,11 @@
|
|||||||
| ISS-002 | Executor 无约束重复调用 lookup_knowledge | 中 | 已修复 | [ISS-002-executor-unconstrained-lookup.md](ISS-002-executor-unconstrained-lookup.md) |
|
| ISS-002 | Executor 无约束重复调用 lookup_knowledge | 中 | 已修复 | [ISS-002-executor-unconstrained-lookup.md](ISS-002-executor-unconstrained-lookup.md) |
|
||||||
| ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [ISS-003-mvp-design-implementation-review.md](ISS-003-mvp-design-implementation-review.md) |
|
| ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [ISS-003-mvp-design-implementation-review.md](ISS-003-mvp-design-implementation-review.md) |
|
||||||
| ISS-004 | Executor 域级检索水位控制(Phase 2) | 低 | 待规划 | [ISS-004-executor-domain-hard-limit.md](ISS-004-executor-domain-hard-limit.md) |
|
| ISS-004 | Executor 域级检索水位控制(Phase 2) | 低 | 待规划 | [ISS-004-executor-domain-hard-limit.md](ISS-004-executor-domain-hard-limit.md) |
|
||||||
|
| ISS-005 | 证据链补齐与降级契约收敛 | 高 | 已归档 | [ISS-005-evidence-trace-hardening.md](ISS-005-evidence-trace-hardening.md) |
|
||||||
|
| ISS-006 | 固定诊断评测集与回归 Harness | 高 | 已归档 | [ISS-006-diagnosis-eval-harness.md](ISS-006-diagnosis-eval-harness.md) |
|
||||||
|
| expand-diagnosis-eval-fixtures | 补齐固定诊断评测 fixture 与 baseline | 中 | 已归档 | [expand-diagnosis-eval-fixtures.md](expand-diagnosis-eval-fixtures.md) |
|
||||||
|
| diagnosis-eval-baseline-diff | 诊断评测 baseline diff 与回归判断 | 中 | 已归档 | [diagnosis-eval-baseline-diff.md](diagnosis-eval-baseline-diff.md) |
|
||||||
|
| mvp-demo-interview-runbook | Plan C 面试可复现 Demo 包 | 中 | 已归档 | [mvp-demo-interview-runbook.md](mvp-demo-interview-runbook.md) |
|
||||||
|
|
||||||
## RAG 重构计划
|
## RAG 重构计划
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,73 @@
|
|||||||
|
# Diagnosis Eval Baseline Diff
|
||||||
|
|
||||||
|
**状态**:已归档
|
||||||
|
**严重程度**:中
|
||||||
|
**发现时间**:2026-07-05
|
||||||
|
**来源**:P1-B follow-up
|
||||||
|
**依赖**:`diagnosis-eval-harness`, `expand-diagnosis-eval-fixtures`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 背景
|
||||||
|
|
||||||
|
现在项目已经有固定诊断 case、完整 fixture 和 baseline report。下一步需要把 baseline 真正用起来:每次改 Agent 后,把新的 report 和 baseline report 做对比。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 问题
|
||||||
|
|
||||||
|
当前 baseline 只能告诉我们“标准状态是什么”,但还不能自动告诉我们“这次改动有没有变差”。
|
||||||
|
|
||||||
|
典型问题包括:
|
||||||
|
|
||||||
|
- pass rate 是否下降。
|
||||||
|
- 某个 case 是否从通过变失败。
|
||||||
|
- 某个 evidence tool 是否从覆盖变成缺失。
|
||||||
|
- verifier verdict 分布是否异常变化。
|
||||||
|
- 平均工具调用数和耗时是否明显上升。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 目标
|
||||||
|
|
||||||
|
新增一个 deterministic baseline diff 能力,用代码比较两份 `DiagnosisEvalReport`。
|
||||||
|
|
||||||
|
完成后应该做到:
|
||||||
|
|
||||||
|
- 输入 baseline report 和 current report。
|
||||||
|
- 输出结构化 diff。
|
||||||
|
- 标出 regression、improvement 和普通 changed。
|
||||||
|
- 支持 JSON 和 Markdown 输出。
|
||||||
|
- 文档说明面试时怎么解释这套回归判断。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 范围
|
||||||
|
|
||||||
|
### In scope
|
||||||
|
|
||||||
|
- report-level diff 数据结构。
|
||||||
|
- aggregate 指标比较。
|
||||||
|
- case-level 指标比较。
|
||||||
|
- JSON / Markdown diff writer。
|
||||||
|
- focused tests 和 eval 文档。
|
||||||
|
|
||||||
|
### Out of scope
|
||||||
|
|
||||||
|
- 不运行真实 Agent。
|
||||||
|
- 不生成新 trace。
|
||||||
|
- 不引入 LLM-as-judge。
|
||||||
|
- 不改现有 evaluator 评分规则。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 面试表达
|
||||||
|
|
||||||
|
可以这样讲:
|
||||||
|
|
||||||
|
```text
|
||||||
|
我不是只保存了一份 baseline,而是加了 baseline diff。
|
||||||
|
每次改 prompt、tool、retrieval 或 verifier 后,
|
||||||
|
我都能把新 report 和 baseline 比较,
|
||||||
|
直接看到哪些 case 退化、哪些证据缺失、成本有没有上升。
|
||||||
|
```
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
# Expand Diagnosis Eval Fixtures
|
||||||
|
|
||||||
|
**状态**:已归档
|
||||||
|
**严重程度**:中
|
||||||
|
**发现时间**:2026-07-04
|
||||||
|
**来源**:P1-B follow-up
|
||||||
|
**依赖**:`diagnosis-eval-harness`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 背景
|
||||||
|
|
||||||
|
`diagnosis-eval-harness` 已经把固定 case、trace evaluator、JSON / Markdown report 和字段文档搭起来了。
|
||||||
|
|
||||||
|
现在还差一步:5 条固定诊断 case 里,只有 2 条有 fixture,另外 3 条还是 missing 状态。这个状态可以验证 evaluator 的错误报告能力,但还不能作为完整 baseline 展示。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 问题
|
||||||
|
|
||||||
|
当前 baseline 还不够完整:
|
||||||
|
|
||||||
|
- `redis-timeout` 没有对应 trace fixture。
|
||||||
|
- `slow-response` 没有对应 trace fixture。
|
||||||
|
- `jvm-memory-risk` 没有对应 trace fixture。
|
||||||
|
- 仓库里还没有一份固定的 baseline JSON / Markdown 报告可供对比。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 目标
|
||||||
|
|
||||||
|
补齐固定诊断评测集,让它从“框架可跑”变成“基准可用”。
|
||||||
|
|
||||||
|
完成后应该做到:
|
||||||
|
|
||||||
|
- 5 条固定 case 都能加载到对应 fixture。
|
||||||
|
- evaluator 能输出完整 baseline report。
|
||||||
|
- baseline report 被保存到仓库,后续 Agent 改动可以拿它做对比。
|
||||||
|
- 文档说明怎么重新生成和怎么看报告。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 范围
|
||||||
|
|
||||||
|
### In scope
|
||||||
|
|
||||||
|
- 补齐 3 个缺失 fixture。
|
||||||
|
- 保存 baseline JSON / Markdown 报告。
|
||||||
|
- 更新 eval 文档。
|
||||||
|
- 补充测试,确保 case 文件引用的 fixture 都存在。
|
||||||
|
|
||||||
|
### Out of scope
|
||||||
|
|
||||||
|
- 不新增 case 数量。
|
||||||
|
- 不改生产 Agent 主链路。
|
||||||
|
- 不引入 LLM-as-judge。
|
||||||
|
- 不启动真实 MySQL、Redis、Milvus 或 LLM。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 面试表达
|
||||||
|
|
||||||
|
可以这样讲:
|
||||||
|
|
||||||
|
```text
|
||||||
|
我先搭了评测 harness,然后把固定 case 的 trace fixture 补齐,
|
||||||
|
生成一份可复现的 baseline report。
|
||||||
|
这样以后每次改 prompt、tool 或 verifier,
|
||||||
|
都能看固定诊断集有没有行为回退,而不是只靠人工感觉。
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 相关文件
|
||||||
|
|
||||||
|
- `mvp/eval/cases/diagnosis-cases.json`
|
||||||
|
- `mvp/eval/fixtures/`
|
||||||
|
- `mvp/eval/reports/`
|
||||||
|
- `mvp/eval/README.md`
|
||||||
|
- `mvp/eval/schema.md`
|
||||||
|
- `src/main/java/com/superbiz/agent/eval/DiagnosisTraceEvaluator.java`
|
||||||
|
- `src/test/java/com/superbiz/agent/eval/DiagnosisTraceEvaluatorTest.java`
|
||||||
|
- `openspec/specs/diagnosis-eval-harness/spec.md`
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
# MVP Demo Interview Runbook
|
||||||
|
|
||||||
|
**状态**:已归档
|
||||||
|
**严重程度**:中
|
||||||
|
**发现时间**:2026-07-05
|
||||||
|
**来源**:Plan C
|
||||||
|
**依赖**:`mvp-demo-trace-acceptance`, `evidence-trace-hardening`, `diagnosis-eval-harness`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 背景
|
||||||
|
|
||||||
|
项目已经有 Agent 主链路、证据 trace、Verifier、反馈、eval baseline,但这些材料分散在不同目录。面试时真正需要的是一个能快速跑、快速讲清楚的 demo 入口。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 问题
|
||||||
|
|
||||||
|
当前 demo 还不够“面试友好”:
|
||||||
|
|
||||||
|
- 启动、请求、trace、反馈步骤分散在文档里。
|
||||||
|
- 没有固定请求 payload 文件。
|
||||||
|
- 没有一键跑 payment-timeout demo 的脚本。
|
||||||
|
- 没有把 trace 字段和面试讲法对应起来的 walkthrough。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 目标
|
||||||
|
|
||||||
|
把 Plan C 落地成 `mvp/demo` 下的可复现 demo 包:
|
||||||
|
|
||||||
|
- 固定支付超时请求。
|
||||||
|
- 一键执行 chat、trace、feedback。
|
||||||
|
- 保存 demo 输出,便于复盘。
|
||||||
|
- 提供面试讲解稿和 trace 检查清单。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 范围
|
||||||
|
|
||||||
|
### In scope
|
||||||
|
|
||||||
|
- `mvp/demo` 文档。
|
||||||
|
- `mvp/demo/requests` 请求文件。
|
||||||
|
- `mvp/demo/scripts` PowerShell 脚本。
|
||||||
|
- `mvp/demo/output` 目录说明。
|
||||||
|
|
||||||
|
### Out of scope
|
||||||
|
|
||||||
|
- 不新增后端 API。
|
||||||
|
- 不改 Agent prompt。
|
||||||
|
- 不扩 eval harness。
|
||||||
|
- 不处理密钥外置和完整离线化。
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
schema: spec-driven
|
||||||
|
created: 2026-07-04
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
## Context
|
||||||
|
|
||||||
|
The project now has the pieces needed for trace-based evaluation:
|
||||||
|
|
||||||
|
- `diagnosis_session` stores final answer, status, duration, counts, feedback, and `self_evaluation`
|
||||||
|
- `agent_step` stores ordered agent execution records
|
||||||
|
- `tool_invocation` stores evidence tool calls with normalized evidence semantics
|
||||||
|
- `GET /api/diagnosis/{sessionId}/trace` can aggregate one diagnosis trace for demo review
|
||||||
|
- `evidence-trace-hardening` defined stable `supported`, `no_evidence`, `deduped`, and `failed` semantics
|
||||||
|
|
||||||
|
P1-B should not add another runtime agent. It should create a repeatable evaluation surface that can be used after changing prompts, retrieval behavior, tools, or verifier logic.
|
||||||
|
|
||||||
|
## Goals / Non-Goals
|
||||||
|
|
||||||
|
**Goals:**
|
||||||
|
- Define fixed MVP diagnosis cases with expected evidence and verdict rules.
|
||||||
|
- Build a deterministic evaluator that can validate a diagnosis trace against a case definition.
|
||||||
|
- Produce JSON and Markdown reports with pass/fail status and key metrics.
|
||||||
|
- Keep the first version usable without a real LLM by allowing fixture trace inputs.
|
||||||
|
- Leave room for a later runtime mode that queries the trace API after a demo run.
|
||||||
|
|
||||||
|
**Non-Goals:**
|
||||||
|
- No LLM-as-judge in this slice.
|
||||||
|
- No automatic prompt optimization.
|
||||||
|
- No new production API.
|
||||||
|
- No change to chat, verifier, retrieval, upload, or feedback behavior.
|
||||||
|
- No requirement to start MySQL/Redis/Milvus/LLM for the first offline evaluator.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
| Decision | Choice | Alternative Considered | Rationale |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Evaluation source | Start with fixture / persisted trace JSON input | Always run live `/api/chat` first | Keeps the first harness deterministic and avoids mixing quality checks with external infrastructure availability. |
|
||||||
|
| Judging strategy | Rule-based trace validation | LLM-as-judge | The immediate goal is regression signal for evidence coverage and degraded behavior, not subjective answer scoring. |
|
||||||
|
| Case format | Static JSON/YAML case definitions | Hard-coded Java tests only | Case files are easier to inspect and explain in interviews. |
|
||||||
|
| Report format | JSON plus Markdown | Console-only output | JSON supports automation; Markdown supports quick human review. |
|
||||||
|
| Metrics | Evidence coverage, verdict distribution, tool-call count, duration, answer keyword coverage | Full semantic correctness | These metrics are available from existing trace data and align with the MVP's observable contract. |
|
||||||
|
|
||||||
|
## Risks / Trade-offs
|
||||||
|
|
||||||
|
- [Risk] Rule-based keyword checks can be brittle. -> Mitigation: keep checks focused on required evidence, verdicts, and high-signal root-cause terms rather than exact answer text.
|
||||||
|
- [Risk] Fixture-only evaluation may drift from runtime behavior. -> Mitigation: design the evaluator around the same trace response shape so runtime traces can be fed in later.
|
||||||
|
- [Risk] Metrics may encourage gaming tool counts. -> Mitigation: report tool counts as cost/efficiency signals, not the sole pass/fail criterion.
|
||||||
|
- [Risk] Too many cases can slow iteration. -> Mitigation: start with 5 MVP cases and keep each case small.
|
||||||
|
|
||||||
|
## Migration Plan
|
||||||
|
|
||||||
|
- No deployment migration is required.
|
||||||
|
- The harness is additive and can be run locally as a test or script.
|
||||||
|
- Rollback is deleting the eval case files, runner, and report docs.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- Should runtime trace API polling be included in the first implementation, or left as a follow-up after the fixture validator lands?
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
## Why
|
||||||
|
|
||||||
|
The MVP can now run a traceable diagnosis flow, but it still lacks a repeatable way to evaluate whether changes to prompts, tools, retrieval, or verifier behavior improve or regress agent quality. A fixed diagnosis evaluation harness gives the project an interview-ready quality baseline instead of relying on a single manual demo.
|
||||||
|
|
||||||
|
## What Changes
|
||||||
|
|
||||||
|
- Add a small fixed evaluation set for representative MVP diagnosis scenarios.
|
||||||
|
- Define expected assertions per case: root-cause keywords, required evidence tools, allowed verifier verdicts, and forbidden behavior.
|
||||||
|
- Add a trace-based evaluator that checks persisted diagnosis traces for evidence coverage, verifier output, final answer shape, tool-call count, and duration.
|
||||||
|
- Add JSON and Markdown report output for quick review after a run.
|
||||||
|
- Add documentation that explains how this evaluation harness should be used during prompt/tool/verifier iteration.
|
||||||
|
|
||||||
|
## Capabilities
|
||||||
|
|
||||||
|
### New Capabilities
|
||||||
|
- `diagnosis-eval-harness`: Defines fixed diagnosis cases, trace-based validation rules, and evaluation report output for MVP Agent regression checks.
|
||||||
|
|
||||||
|
### Modified Capabilities
|
||||||
|
- None.
|
||||||
|
|
||||||
|
## Impact
|
||||||
|
|
||||||
|
- Affected areas: evaluation resources/scripts/tests, MVP demo documentation, and devflow records.
|
||||||
|
- Affected runtime behavior: none. This change reads persisted trace data or fixture trace data and does not modify the chat execution path.
|
||||||
|
- Affected APIs: none.
|
||||||
|
- Dependencies: relies on the evidence semantics from `evidence-trace-hardening`, especially `tool_invocation`, `tool_trace_summary`, `verifier_evaluation`, and evidence status conventions.
|
||||||
|
- Non-goals: no LLM-as-judge, no full offline LLM runtime, no new production endpoint, no schema migration.
|
||||||
+59
@@ -0,0 +1,59 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL define fixed diagnosis cases
|
||||||
|
The system SHALL provide a small fixed set of MVP diagnosis evaluation cases with explicit expected trace and answer criteria.
|
||||||
|
|
||||||
|
#### Scenario: Case definition includes expected evidence
|
||||||
|
- **WHEN** an evaluation case is defined
|
||||||
|
- **THEN** it SHALL include a case id, user question, expected root-cause keywords, required evidence tools, and allowed verifier verdicts
|
||||||
|
|
||||||
|
#### Scenario: Case definition can express forbidden behavior
|
||||||
|
- **WHEN** a case has known unsafe behavior
|
||||||
|
- **THEN** the case definition SHALL be able to declare forbidden answer keywords or forbidden verdicts
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL validate diagnosis traces
|
||||||
|
The system SHALL validate a diagnosis trace against the corresponding case definition using deterministic rules.
|
||||||
|
|
||||||
|
#### Scenario: Evidence coverage validation
|
||||||
|
- **WHEN** a trace is evaluated
|
||||||
|
- **THEN** the evaluator SHALL verify that required evidence tools appear in `toolInvocations` or verifier trace summaries
|
||||||
|
|
||||||
|
#### Scenario: Verifier evaluation validation
|
||||||
|
- **WHEN** a trace is evaluated
|
||||||
|
- **THEN** the evaluator SHALL verify that `selfEvaluation.verifier_evaluation.verdict` exists
|
||||||
|
- **AND** the verdict SHALL be one of the case's allowed verdicts
|
||||||
|
|
||||||
|
#### Scenario: Answer keyword validation
|
||||||
|
- **WHEN** a trace is evaluated
|
||||||
|
- **THEN** the evaluator SHALL verify that the final answer includes at least the configured minimum root-cause keyword coverage
|
||||||
|
|
||||||
|
#### Scenario: Degraded output validation
|
||||||
|
- **WHEN** a trace verdict is `REJECT`
|
||||||
|
- **THEN** the evaluator SHALL verify that the final answer follows the degraded-output contract rather than leaking an unverified executor answer
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL report quality and cost signals
|
||||||
|
The system SHALL produce a report that summarizes pass/fail results and key trace metrics.
|
||||||
|
|
||||||
|
#### Scenario: JSON report output
|
||||||
|
- **WHEN** an evaluation run completes
|
||||||
|
- **THEN** the evaluator SHALL output a JSON report with per-case result, failed checks, verdict, evidence coverage, tool-call count, and duration
|
||||||
|
|
||||||
|
#### Scenario: Markdown report output
|
||||||
|
- **WHEN** an evaluation run completes
|
||||||
|
- **THEN** the evaluator SHALL output a Markdown report suitable for review in the repository
|
||||||
|
|
||||||
|
#### Scenario: Aggregate metrics
|
||||||
|
- **WHEN** multiple cases are evaluated
|
||||||
|
- **THEN** the report SHALL include aggregate pass rate, verdict distribution, average tool-call count, and average duration when available
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL support offline fixture mode
|
||||||
|
The first evaluator version SHALL be runnable without live MySQL, Redis, Milvus, or LLM services by evaluating saved trace fixtures.
|
||||||
|
|
||||||
|
#### Scenario: Fixture trace evaluation
|
||||||
|
- **WHEN** the evaluator is run against a directory of trace fixture files
|
||||||
|
- **THEN** it SHALL evaluate each trace file against its matching case definition
|
||||||
|
- **AND** it SHALL not require a running application service
|
||||||
|
|
||||||
|
#### Scenario: Missing fixture is reported clearly
|
||||||
|
- **WHEN** a case has no matching trace fixture
|
||||||
|
- **THEN** the evaluator SHALL mark the case as not run or failed with a clear reason
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
## 1. Case Definitions
|
||||||
|
|
||||||
|
- [x] 1.1 Add evaluation case definition format for fixed MVP diagnosis scenarios.
|
||||||
|
- [x] 1.2 Add the first 5 case definitions: payment timeout, MySQL pool exhausted, Redis timeout, slow response, and JVM memory risk.
|
||||||
|
- [x] 1.3 Document the meaning of expected keywords, required evidence tools, allowed verdicts, and forbidden behavior.
|
||||||
|
|
||||||
|
## 2. Trace Fixtures
|
||||||
|
|
||||||
|
- [x] 2.1 Add fixture trace schema or DTOs that match `DiagnosisTraceResponse` enough for offline evaluation.
|
||||||
|
- [x] 2.2 Add at least one representative trace fixture for a passing case.
|
||||||
|
- [x] 2.3 Add at least one fixture covering low-confidence or degraded behavior.
|
||||||
|
|
||||||
|
## 3. Evaluator
|
||||||
|
|
||||||
|
- [x] 3.1 Implement trace validation rules for evidence coverage, verifier verdict, answer keyword coverage, and degraded-output contract.
|
||||||
|
- [x] 3.2 Implement aggregate metrics: pass rate, verdict distribution, average tool-call count, and average duration.
|
||||||
|
- [x] 3.3 Implement JSON report output.
|
||||||
|
- [x] 3.4 Implement Markdown report output.
|
||||||
|
|
||||||
|
## 4. Tests And Documentation
|
||||||
|
|
||||||
|
- [x] 4.1 Add focused offline tests for the evaluator.
|
||||||
|
- [x] 4.2 Add run instructions under `mvp/demo` or `mvp/notes`.
|
||||||
|
- [x] 4.3 Run targeted tests for the evaluator.
|
||||||
|
- [x] 4.4 Run compile verification.
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
schema: spec-driven
|
||||||
|
created: 2026-07-04
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
## Context
|
||||||
|
|
||||||
|
The current MVP already has the core pieces required for traceable agent execution:
|
||||||
|
|
||||||
|
- `LookupKnowledgeTool` writes rich retrieval metadata into `tool_invocation`
|
||||||
|
- `QueryLogsTools` and `QueryMetricsTools` use `ToolInvocationRecorder`
|
||||||
|
- `ToolTraceSummaryService` turns persisted rows into verifier-facing evidence summaries
|
||||||
|
- `ChatService` already contains fallback behavior for missing or invalid `verifier_output`
|
||||||
|
|
||||||
|
The gap is no longer “there is no evidence trace”. The gap is that the evidence trace contract is split across two persistence paths and several implicit conventions:
|
||||||
|
|
||||||
|
- `LookupKnowledgeTool` builds `ToolInvocation` rows itself
|
||||||
|
- the other evidence tools use `ToolInvocationRecorder.recordEvidenceTool(...)`
|
||||||
|
- “failed”, “no evidence”, “deduped”, and “successful but weak” are inferred differently across tools
|
||||||
|
- degraded output behavior exists in code but is only lightly covered by tests
|
||||||
|
|
||||||
|
For interview-facing hardening, this slice should make those semantics explicit and testable without changing the database schema or the overall multi-agent workflow.
|
||||||
|
|
||||||
|
## Goals / Non-Goals
|
||||||
|
|
||||||
|
**Goals:**
|
||||||
|
- Centralize the common persistence contract for evidence-bearing tools.
|
||||||
|
- Preserve `lookup_knowledge`-specific retrieval fields while removing ad hoc duplication in how evidence rows are created.
|
||||||
|
- Define stable summarization semantics for:
|
||||||
|
- successful evidence
|
||||||
|
- no-hit / no-usable-evidence
|
||||||
|
- deduped retrievals
|
||||||
|
- failed evidence queries
|
||||||
|
- Make `ChatService` fallback and degraded-output paths testable as explicit product behavior.
|
||||||
|
- Keep the scope small enough to unblock the next P1-B evaluation harness.
|
||||||
|
|
||||||
|
**Non-Goals:**
|
||||||
|
- No new table, column, or Flyway migration.
|
||||||
|
- No new public API.
|
||||||
|
- No new verifier verdict type beyond `PASS` / `LOW_CONFID` / `REJECT`.
|
||||||
|
- No attempt to redesign planner/executor routing.
|
||||||
|
- No full offline runtime or end-to-end benchmark harness in this slice.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
| Decision | Choice | Alternative Considered | Rationale |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Evidence persistence ownership | Keep `ToolInvocationRecorder` as the single common entry point | Let each tool continue building `ToolInvocation` rows ad hoc | The recorder already exists and is the right seam for contract hardening. |
|
||||||
|
| `lookup_knowledge` integration style | Add a richer recorder entry path for retrieval-aware calls | Force `lookup_knowledge` into the same minimal method used by logs/metrics | `lookup_knowledge` carries domain-specific fields such as L0/L1 counts, relevance, dedup reason, and retrieval details that should stay structured. |
|
||||||
|
| No-evidence semantics | Distinguish failed calls from successful calls that yield no usable evidence | Collapse all non-successful evidence into one bucket | Verifier and future evaluation harnesses need to separate “tool broke” from “tool succeeded but found nothing useful”. |
|
||||||
|
| Degraded-path hardening | Add focused unit tests around verifier fallback and output shaping | Rely on runtime demo only | Interview value comes from proving the system fails predictably, not just that the happy path ran once. |
|
||||||
|
| Scope boundary | Keep changes additive and contract-oriented | Expand into P1-B evaluation harness in the same change | This keeps the slice reviewable and avoids mixing infrastructure hardening with evaluation product work. |
|
||||||
|
|
||||||
|
## Risks / Trade-offs
|
||||||
|
|
||||||
|
- [Risk] Tightening persistence semantics could subtly change existing trace summaries. -> Mitigation: keep field names stable and add regression tests around summary output.
|
||||||
|
- [Risk] Over-generalizing the recorder could make retrieval-specific rows less informative. -> Mitigation: keep a retrieval-aware recording path rather than flattening all tools to the same minimal payload.
|
||||||
|
- [Risk] Tests may lock in the current fallback copy too aggressively. -> Mitigation: assert protocol-level behavior and key phrases, not brittle full-string snapshots.
|
||||||
|
- [Risk] `lookup_knowledge` dedup semantics are product-specific and may not fit generic “success/failure” labels cleanly. -> Mitigation: preserve `dedupReason` and treat dedup as a first-class no-new-evidence case in summary logic.
|
||||||
|
|
||||||
|
## Migration Plan
|
||||||
|
|
||||||
|
- No deployment migration is required beyond shipping the code changes.
|
||||||
|
- Existing `tool_invocation` rows remain valid because this change reuses the same schema.
|
||||||
|
- Rollback is code-only: revert the recorder/summary/fallback hardening and keep the persisted rows as-is.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- Should P1-B metrics count deduped retrievals as “no-evidence”, or report them as a separate category? This change will preserve enough structure to decide later without another schema change.
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
## Why
|
||||||
|
|
||||||
|
The MVP already persists evidence tool invocations and uses a Verifier to judge answer quality, but the current evidence trace semantics are still only partially standardized. For interview-grade agent engineering, the system needs a tighter contract for evidence persistence, no-evidence/failure states, and degraded output behavior, plus focused tests that prove those paths work offline.
|
||||||
|
|
||||||
|
## What Changes
|
||||||
|
|
||||||
|
- Standardize the persisted evidence-tool contract across `lookup_knowledge`, `query_logs`, and `query_metrics`.
|
||||||
|
- Align how evidence tools represent success, no-hit, deduped, and failed calls so `ToolTraceSummaryService` can summarize them consistently.
|
||||||
|
- Harden `ChatService` fallback behavior for invalid or missing verifier output and make the degraded-output paths explicitly testable.
|
||||||
|
- Add focused offline tests for evidence recording, trace summarization, and verifier fallback / degraded output behavior.
|
||||||
|
- Record this slice as a dedicated P1-A change tied to the interview-focused MVP hardening track.
|
||||||
|
|
||||||
|
## Capabilities
|
||||||
|
|
||||||
|
### New Capabilities
|
||||||
|
- `evidence-trace-hardening`: Covers standardized evidence invocation persistence, verifier-facing evidence summary semantics, and explicit degraded-output contracts for evidence gaps and verifier failures.
|
||||||
|
|
||||||
|
### Modified Capabilities
|
||||||
|
- `chat-verifier-agent`: Tightens verifier input evidence semantics and fallback guarantees without changing the high-level planner/executor/verifier workflow.
|
||||||
|
|
||||||
|
## Impact
|
||||||
|
|
||||||
|
- Affected code: `ToolInvocationRecorder`, `LookupKnowledgeTool`, `QueryLogsTools`, `QueryMetricsTools`, `ToolTraceSummaryService`, `ChatService`, and focused test classes.
|
||||||
|
- Affected runtime behavior: evidence-bearing tools will persist more consistent invocation semantics; verifier fallback and degraded outputs remain additive hardening, not a product-flow rewrite.
|
||||||
|
- Affected APIs: none. No new endpoint or schema is introduced.
|
||||||
|
- Non-goals: no new evidence tools, no database migration, no evaluation harness, no trace UI, no security/config cleanup in this slice.
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Verifier inputs SHALL tolerate hardened no-evidence semantics
|
||||||
|
The verifier integration SHALL continue to work when evidence summaries distinguish failed calls, no-hit calls, and deduped retrievals more explicitly.
|
||||||
|
|
||||||
|
#### Scenario: Failed evidence remains a verifier-visible gap
|
||||||
|
- **WHEN** an evidence-bearing tool invocation fails
|
||||||
|
- **THEN** the verifier-facing trace summary SHALL preserve that failure as a gap
|
||||||
|
- **AND** the verifier flow SHALL continue without crashing
|
||||||
|
|
||||||
|
#### Scenario: Deduped retrievals do not count as fresh support
|
||||||
|
- **WHEN** the verifier-facing trace summary contains deduped `lookup_knowledge` entries
|
||||||
|
- **THEN** those entries SHALL be treated as no-new-evidence
|
||||||
|
- **AND** they SHALL NOT be interpreted as fresh direct support for the answer
|
||||||
+82
@@ -0,0 +1,82 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Evidence-bearing tools SHALL persist a unified invocation contract
|
||||||
|
The system SHALL persist evidence-bearing tool calls through a unified contract that guarantees the same baseline fields across `lookup_knowledge`, `query_logs`, and `query_metrics`.
|
||||||
|
|
||||||
|
#### Scenario: Common evidence fields are always persisted
|
||||||
|
- **WHEN** an evidence-bearing tool finishes a call
|
||||||
|
- **THEN** the persisted `tool_invocation` row SHALL include `session_id`, `tool_name`, `input_params`, `duration_ms`, `success`, and an output preview or explicit no-output state
|
||||||
|
|
||||||
|
#### Scenario: Retrieval-aware tools preserve structured retrieval fields
|
||||||
|
- **WHEN** `lookup_knowledge` persists a tool invocation
|
||||||
|
- **THEN** the row SHALL preserve retrieval-specific fields such as retrieval layer, L0/L1 counts, relevance level, dedup reason, and retrieval details
|
||||||
|
- **AND** those fields SHALL be written through the same recorder contract rather than by ad hoc row construction in unrelated code paths
|
||||||
|
|
||||||
|
### Requirement: Evidence trace semantics SHALL distinguish failure from no-evidence outcomes
|
||||||
|
The system SHALL keep failed calls separate from successful calls that return no usable evidence.
|
||||||
|
|
||||||
|
#### Scenario: Tool failure is preserved as failure
|
||||||
|
- **WHEN** an evidence-bearing tool throws, times out, or returns an execution error
|
||||||
|
- **THEN** the persisted row SHALL set `success=false`
|
||||||
|
- **AND** it SHALL preserve an `error_message` explaining the failure
|
||||||
|
|
||||||
|
#### Scenario: No usable evidence is preserved without pretending success
|
||||||
|
- **WHEN** an evidence-bearing tool completes normally but yields no usable evidence for the verifier
|
||||||
|
- **THEN** the persisted contract SHALL preserve that the call completed
|
||||||
|
- **AND** the verifier-facing summary SHALL describe it as a no-evidence outcome rather than direct support
|
||||||
|
|
||||||
|
#### Scenario: Deduped retrieval remains auditable
|
||||||
|
- **WHEN** `lookup_knowledge` is blocked by session-level deduplication
|
||||||
|
- **THEN** the persisted row SHALL preserve the dedup reason
|
||||||
|
- **AND** the verifier-facing summary SHALL treat that row as no-new-evidence rather than as a fresh supporting hit
|
||||||
|
|
||||||
|
### Requirement: Verifier-facing evidence summaries SHALL use stable no-evidence rules
|
||||||
|
The system SHALL summarize persisted tool rows into verifier-facing evidence entries using stable rules for success, failure, no-hit, and deduped outcomes.
|
||||||
|
|
||||||
|
#### Scenario: Failed evidence calls remain visible in the summary
|
||||||
|
- **WHEN** `ToolTraceSummaryService` processes failed evidence-bearing tool rows
|
||||||
|
- **THEN** the summary SHALL retain them
|
||||||
|
- **AND** it SHALL mark them as unsuccessful evidence with an output summary that explains the gap
|
||||||
|
|
||||||
|
#### Scenario: No-hit and deduped calls do not upgrade evidence level
|
||||||
|
- **WHEN** `ToolTraceSummaryService` processes rows that returned no usable evidence or were deduped
|
||||||
|
- **THEN** those rows SHALL NOT be promoted to direct or indirect evidence
|
||||||
|
- **AND** their counts SHALL still be reflected in the merged summary entry
|
||||||
|
|
||||||
|
#### Scenario: Successful evidence keeps the strongest available support
|
||||||
|
- **WHEN** multiple rows for the same tool and topic domain are merged
|
||||||
|
- **THEN** the summary SHALL preserve the strongest successful evidence level among them
|
||||||
|
- **AND** it SHALL also retain repeated-call, failed-call, and no-hit counts for auditability
|
||||||
|
|
||||||
|
### Requirement: ChatService SHALL degrade predictably on verifier output failures
|
||||||
|
The system SHALL treat missing or invalid verifier output as a bounded degraded path instead of an unstructured runtime error.
|
||||||
|
|
||||||
|
#### Scenario: Missing verifier output falls back to LOW_CONFID
|
||||||
|
- **WHEN** the verifier step completes without a usable `verifier_output`
|
||||||
|
- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision
|
||||||
|
- **AND** the final user-facing output SHALL use the fixed low-confidence protocol
|
||||||
|
|
||||||
|
#### Scenario: Invalid verifier JSON falls back to LOW_CONFID
|
||||||
|
- **WHEN** the verifier returns malformed or non-parseable JSON
|
||||||
|
- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision
|
||||||
|
- **AND** the fallback SHALL still persist a verifier-evaluation record
|
||||||
|
|
||||||
|
#### Scenario: REJECT output hides unverified raw answer text
|
||||||
|
- **WHEN** the final verifier decision is `REJECT`
|
||||||
|
- **THEN** the user-facing output SHALL use the degraded template
|
||||||
|
- **AND** it SHALL NOT pass through the raw executor answer
|
||||||
|
|
||||||
|
### Requirement: Evidence-trace hardening SHALL be covered by focused offline tests
|
||||||
|
The system SHALL provide offline tests for the hardened evidence contract and degraded-output behavior.
|
||||||
|
|
||||||
|
#### Scenario: Evidence recorder contract is tested offline
|
||||||
|
- **WHEN** the test suite runs the focused recorder tests
|
||||||
|
- **THEN** it SHALL verify the persisted semantics for successful, failed, and no-evidence evidence-tool calls without requiring external infrastructure
|
||||||
|
|
||||||
|
#### Scenario: Trace summary hardening is tested offline
|
||||||
|
- **WHEN** the test suite runs the focused trace-summary tests
|
||||||
|
- **THEN** it SHALL verify merged summary behavior for mixed success, failure, no-hit, and deduped tool rows
|
||||||
|
|
||||||
|
#### Scenario: Verifier fallback behavior is tested offline
|
||||||
|
- **WHEN** the test suite runs the focused `ChatService` fallback tests
|
||||||
|
- **THEN** it SHALL verify the missing-output, invalid-JSON, `LOW_CONFID`, and `REJECT` degraded paths without requiring a real LLM or database
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
## 1. Evidence Persistence Contract
|
||||||
|
|
||||||
|
- [x] 1.1 Extend `ToolInvocationRecorder` with a richer evidence-recording path that can preserve retrieval-aware fields as well as common evidence fields.
|
||||||
|
- [x] 1.2 Refactor `LookupKnowledgeTool` to persist `tool_invocation` rows through `ToolInvocationRecorder` instead of its own ad hoc row-construction path.
|
||||||
|
- [x] 1.3 Align `QueryLogsTools` and `QueryMetricsTools` no-hit / failure payloads with the hardened evidence contract.
|
||||||
|
|
||||||
|
## 2. Verifier-Facing Summary Semantics
|
||||||
|
|
||||||
|
- [x] 2.1 Harden `ToolTraceSummaryService` so failed, no-hit, and deduped evidence rows are summarized with stable no-evidence semantics.
|
||||||
|
- [x] 2.2 Preserve merged-call counts for repeated hits, failures, and no-new-evidence rows without overstating evidence strength.
|
||||||
|
|
||||||
|
## 3. Chat Degraded Paths
|
||||||
|
|
||||||
|
- [x] 3.1 Add focused `ChatService` tests for missing verifier output fallback to `LOW_CONFID`.
|
||||||
|
- [x] 3.2 Add focused `ChatService` tests for invalid verifier JSON fallback to `LOW_CONFID`.
|
||||||
|
- [x] 3.3 Add focused `ChatService` tests that `REJECT` output uses the degraded template and does not leak raw executor answer content.
|
||||||
|
|
||||||
|
## 4. Verification
|
||||||
|
|
||||||
|
- [x] 4.1 Add focused offline tests for the recorder contract and `ToolTraceSummaryService`.
|
||||||
|
- [x] 4.2 Run targeted test commands for the new/updated offline tests.
|
||||||
|
- [x] 4.3 Run compile verification.
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
schema: spec-driven
|
||||||
|
created: 2026-07-04
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
## Context
|
||||||
|
|
||||||
|
The eval harness now has a complete five-case fixture baseline and saved JSON / Markdown baseline reports. The missing piece is a deterministic comparison step that explains whether a new report is better, worse, or just different from the baseline.
|
||||||
|
|
||||||
|
## Goals / Non-Goals
|
||||||
|
|
||||||
|
**Goals:**
|
||||||
|
|
||||||
|
- Compare two `DiagnosisEvalReport` objects without requiring external services.
|
||||||
|
- Surface aggregate regressions such as pass-rate drops, verdict distribution shifts, and cost increases.
|
||||||
|
- Surface per-case regressions such as pass-to-fail changes, missing evidence coverage, verdict changes, keyword coverage loss, and missing cases.
|
||||||
|
- Write JSON and Markdown diff outputs for review.
|
||||||
|
|
||||||
|
**Non-Goals:**
|
||||||
|
|
||||||
|
- Do not run the Agent or regenerate traces.
|
||||||
|
- Do not introduce LLM-as-judge.
|
||||||
|
- Do not change evaluator scoring rules.
|
||||||
|
- Do not block on performance thresholds beyond simple numeric diff signals.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
- Decision: Compare report DTOs instead of raw traces.
|
||||||
|
- Reason: `DiagnosisEvalReport` is already the stable structured output of the evaluator and is cheaper to diff than trace internals.
|
||||||
|
- Alternative considered: compare raw trace fixtures. That would expose more detail but duplicate evaluator responsibilities.
|
||||||
|
|
||||||
|
- Decision: Classify each diff item as `REGRESSION`, `IMPROVEMENT`, or `CHANGED`.
|
||||||
|
- Reason: interview and CI usage both need a quick answer to "did this get worse?" while still preserving neutral changes.
|
||||||
|
- Alternative considered: only output numeric deltas. That is harder to scan and less actionable.
|
||||||
|
|
||||||
|
- Decision: Keep thresholds explicit and conservative.
|
||||||
|
- Reason: pass/fail and missing evidence are hard regressions; tool calls and duration are cost signals that should be visible even if not always blocking.
|
||||||
|
- Alternative considered: fail only on pass-rate drop. That misses cases where quality stays green but cost or confidence behavior changes.
|
||||||
|
|
||||||
|
## Risks / Trade-offs
|
||||||
|
|
||||||
|
- Report comparison can only see fields already captured by `DiagnosisEvalReport`. Mitigation: use this as the first regression layer and add richer report fields later if needed.
|
||||||
|
- Duration may fluctuate in live runs. Mitigation: fixture baseline uses stable durations; live-mode thresholds can be added later.
|
||||||
|
- Verdict distribution changes can be intentional. Mitigation: classify them as `CHANGED` unless they coincide with per-case regressions.
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
## Why
|
||||||
|
|
||||||
|
The evaluation baseline is now complete, but developers still need a repeatable way to decide whether a new Agent run regressed against that baseline. A deterministic baseline diff turns saved reports into an actionable regression signal instead of a static artifact.
|
||||||
|
|
||||||
|
## What Changes
|
||||||
|
|
||||||
|
- Add a baseline diff model that compares two `DiagnosisEvalReport` objects.
|
||||||
|
- Detect aggregate changes such as pass-rate drops, verdict distribution shifts, tool-call cost changes, and duration changes.
|
||||||
|
- Detect per-case changes such as pass/fail regression, verdict changes, keyword coverage changes, evidence coverage loss, and missing/new cases.
|
||||||
|
- Add JSON and Markdown diff output suitable for review.
|
||||||
|
- Document how to interpret the diff in the eval docs.
|
||||||
|
|
||||||
|
## Capabilities
|
||||||
|
|
||||||
|
### New Capabilities
|
||||||
|
|
||||||
|
- None.
|
||||||
|
|
||||||
|
### Modified Capabilities
|
||||||
|
|
||||||
|
- `diagnosis-eval-harness`: Extend the existing evaluation harness so a current report can be compared against the saved baseline report.
|
||||||
|
|
||||||
|
## Impact
|
||||||
|
|
||||||
|
- Affects eval-only Java code under `src/main/java/com/superbiz/agent/eval`.
|
||||||
|
- Adds focused tests under `src/test/java/com/superbiz/agent/eval`.
|
||||||
|
- Updates `mvp/eval` documentation and may add sample diff output.
|
||||||
|
- No production Agent runtime, API, database schema, or external dependency changes are expected.
|
||||||
+46
@@ -0,0 +1,46 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL compare reports against a baseline
|
||||||
|
The system SHALL compare a current diagnosis evaluation report against a saved baseline report using deterministic rules.
|
||||||
|
|
||||||
|
#### Scenario: Aggregate regression detection
|
||||||
|
- **WHEN** the current report has a lower pass rate than the baseline report
|
||||||
|
- **THEN** the diff SHALL record a regression with the old value, new value, and delta
|
||||||
|
|
||||||
|
#### Scenario: Cost signal detection
|
||||||
|
- **WHEN** average tool-call count or average duration changes between reports
|
||||||
|
- **THEN** the diff SHALL record the baseline value, current value, and delta
|
||||||
|
|
||||||
|
#### Scenario: Verdict distribution comparison
|
||||||
|
- **WHEN** verdict counts differ between reports
|
||||||
|
- **THEN** the diff SHALL record the verdict distribution changes
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL compare case-level report results
|
||||||
|
The system SHALL compare case results by case id and report actionable per-case changes.
|
||||||
|
|
||||||
|
#### Scenario: Case pass/fail regression
|
||||||
|
- **WHEN** a case changes from passing in the baseline to failing in the current report
|
||||||
|
- **THEN** the diff SHALL record a regression for that case
|
||||||
|
|
||||||
|
#### Scenario: Evidence coverage regression
|
||||||
|
- **WHEN** a required evidence tool changes from covered to uncovered for a case
|
||||||
|
- **THEN** the diff SHALL record a regression naming the case and tool
|
||||||
|
|
||||||
|
#### Scenario: Missing case detection
|
||||||
|
- **WHEN** a baseline case is absent from the current report
|
||||||
|
- **THEN** the diff SHALL record a regression for the missing case
|
||||||
|
|
||||||
|
#### Scenario: New case detection
|
||||||
|
- **WHEN** a current report contains a case absent from the baseline
|
||||||
|
- **THEN** the diff SHALL record the case as a non-regression change
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL report baseline diff results
|
||||||
|
The system SHALL expose baseline diff output in structured JSON and reviewable Markdown formats.
|
||||||
|
|
||||||
|
#### Scenario: JSON diff output
|
||||||
|
- **WHEN** a baseline diff is written as JSON
|
||||||
|
- **THEN** it SHALL include aggregate summary fields and detailed diff items
|
||||||
|
|
||||||
|
#### Scenario: Markdown diff output
|
||||||
|
- **WHEN** a baseline diff is written as Markdown
|
||||||
|
- **THEN** it SHALL include a readable summary and a table of diff items
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
## 1. OpenSpec And Issue Setup
|
||||||
|
|
||||||
|
- [x] 1.1 Create slug-based issue and devflow tracking files.
|
||||||
|
- [x] 1.2 Create OpenSpec proposal, design, delta spec, and tasks.
|
||||||
|
|
||||||
|
## 2. Baseline Diff Implementation
|
||||||
|
|
||||||
|
- [x] 2.1 Add diff result data structures for summary and per-item changes.
|
||||||
|
- [x] 2.2 Implement deterministic report comparison rules.
|
||||||
|
- [x] 2.3 Implement JSON and Markdown diff report writing.
|
||||||
|
|
||||||
|
## 3. Documentation
|
||||||
|
|
||||||
|
- [x] 3.1 Document baseline diff inputs, outputs, and interpretation in eval docs.
|
||||||
|
- [x] 3.2 Add sample diff output for a representative regression.
|
||||||
|
|
||||||
|
## 4. Tests And Validation
|
||||||
|
|
||||||
|
- [x] 4.1 Add focused tests for aggregate and case-level diff behavior.
|
||||||
|
- [x] 4.2 Add focused tests for JSON and Markdown diff output.
|
||||||
|
- [x] 4.3 Run evaluator/diff tests.
|
||||||
|
- [x] 4.4 Run compile verification.
|
||||||
|
- [x] 4.5 Run OpenSpec validation.
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
schema: spec-driven
|
||||||
|
created: 2026-07-04
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
## Context
|
||||||
|
|
||||||
|
`diagnosis-eval-harness` already provides fixed case definitions, fixture-mode evaluation, JSON / Markdown report writing, and focused evaluator tests. The current baseline is incomplete because three fixed cases intentionally point to missing fixtures.
|
||||||
|
|
||||||
|
## Goals / Non-Goals
|
||||||
|
|
||||||
|
**Goals:**
|
||||||
|
|
||||||
|
- Add representative trace fixtures for every fixed diagnosis case.
|
||||||
|
- Save a baseline report that can be reviewed and compared after future Agent changes.
|
||||||
|
- Keep the baseline reproducible in offline mode.
|
||||||
|
- Document how to regenerate the baseline.
|
||||||
|
|
||||||
|
**Non-Goals:**
|
||||||
|
|
||||||
|
- Do not change production Agent runtime behavior.
|
||||||
|
- Do not require live infrastructure or a real LLM.
|
||||||
|
- Do not introduce a new LLM-based grader.
|
||||||
|
- Do not expand the case set beyond the existing five fixed MVP diagnosis cases.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
- Use checked-in fixture traces instead of live service calls.
|
||||||
|
- Rationale: the goal is a stable regression baseline that can run in CI or interview environments without external dependencies.
|
||||||
|
- Alternative considered: start the application and call the trace API. That is useful later, but it introduces infrastructure noise before the baseline is complete.
|
||||||
|
|
||||||
|
- Save baseline reports under `mvp/eval/reports`.
|
||||||
|
- Rationale: reports are reviewable artifacts, not transient build output, and they show the expected current behavior of the baseline.
|
||||||
|
- Alternative considered: generate reports only in tests. That verifies behavior but does not give an easy artifact to show or diff.
|
||||||
|
|
||||||
|
- Keep fixture outcomes representative rather than forcing every case to pass.
|
||||||
|
- Rationale: a baseline should reflect expected behavior, including low-confidence or degraded cases, as long as the outcome is explicit and stable.
|
||||||
|
- Alternative considered: make every fixture pass. That looks cleaner but hides important degraded-path behavior.
|
||||||
|
|
||||||
|
## Risks / Trade-offs
|
||||||
|
|
||||||
|
- Fixture data can drift from real runtime traces. Mitigation: keep fixtures shaped like `DiagnosisTraceResponse` and add tests that load every referenced fixture.
|
||||||
|
- A saved baseline report can become stale after intentional rule changes. Mitigation: document regeneration steps and update the report in the same change as rule or fixture updates.
|
||||||
|
- Keyword-based checks are coarse. Mitigation: this change keeps the deterministic harness simple and leaves semantic scoring as a later improvement.
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
## Why
|
||||||
|
|
||||||
|
The diagnosis evaluation harness is implemented, but the baseline is still incomplete because only two of the five fixed cases have trace fixtures. Completing the fixture set and saving a baseline report makes the harness useful as a practical regression signal for interview demos and future Agent changes.
|
||||||
|
|
||||||
|
## What Changes
|
||||||
|
|
||||||
|
- Add trace fixtures for the remaining fixed diagnosis cases: Redis timeout, slow response, and JVM memory risk.
|
||||||
|
- Add a reproducible baseline report generated from the full fixture set.
|
||||||
|
- Document how to regenerate and interpret the baseline.
|
||||||
|
- Keep the evaluator deterministic and offline; no live MySQL, Redis, Milvus, or LLM service is required.
|
||||||
|
|
||||||
|
## Capabilities
|
||||||
|
|
||||||
|
### New Capabilities
|
||||||
|
|
||||||
|
- None.
|
||||||
|
|
||||||
|
### Modified Capabilities
|
||||||
|
|
||||||
|
- `diagnosis-eval-harness`: Extend the existing evaluation harness requirement so the fixed MVP case set has complete fixture coverage and a saved baseline report.
|
||||||
|
|
||||||
|
## Impact
|
||||||
|
|
||||||
|
- Affects `mvp/eval/cases`, `mvp/eval/fixtures`, and eval documentation.
|
||||||
|
- May add baseline output files under `mvp/eval/reports`.
|
||||||
|
- May add or update focused evaluator tests to assert full fixture coverage and report generation.
|
||||||
|
- No production runtime API or database schema changes are expected.
|
||||||
+27
@@ -0,0 +1,27 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL provide complete fixture coverage for fixed cases
|
||||||
|
The system SHALL include an offline trace fixture for every fixed MVP diagnosis evaluation case.
|
||||||
|
|
||||||
|
#### Scenario: Every case resolves to a fixture file
|
||||||
|
- **WHEN** the evaluator loads the fixed case definition file
|
||||||
|
- **THEN** every case's `traceFixture` value SHALL resolve to an existing JSON fixture file
|
||||||
|
|
||||||
|
#### Scenario: Fixture files are loadable as diagnosis traces
|
||||||
|
- **WHEN** each referenced fixture is loaded
|
||||||
|
- **THEN** it SHALL deserialize into the trace response shape used by the evaluator
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL preserve a reproducible baseline report
|
||||||
|
The system SHALL preserve a generated baseline report for the full fixed fixture set.
|
||||||
|
|
||||||
|
#### Scenario: Baseline report includes all fixed cases
|
||||||
|
- **WHEN** the baseline report is generated from the fixed case file and fixture directory
|
||||||
|
- **THEN** the report SHALL include one result for every fixed case
|
||||||
|
|
||||||
|
#### Scenario: Baseline report is reviewable
|
||||||
|
- **WHEN** the baseline report is written
|
||||||
|
- **THEN** it SHALL be available in JSON and Markdown formats under the eval documentation area
|
||||||
|
|
||||||
|
#### Scenario: Baseline regeneration is documented
|
||||||
|
- **WHEN** a developer changes fixtures or evaluator rules
|
||||||
|
- **THEN** the eval documentation SHALL explain how to regenerate the baseline report
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
## 1. Fixture Coverage
|
||||||
|
|
||||||
|
- [x] 1.1 Add Redis timeout trace fixture referenced by the fixed case file.
|
||||||
|
- [x] 1.2 Add slow response trace fixture referenced by the fixed case file.
|
||||||
|
- [x] 1.3 Add JVM memory risk trace fixture referenced by the fixed case file.
|
||||||
|
- [x] 1.4 Verify every `traceFixture` in `diagnosis-cases.json` resolves to an existing fixture file.
|
||||||
|
|
||||||
|
## 2. Baseline Reports
|
||||||
|
|
||||||
|
- [x] 2.1 Generate a full baseline JSON report for all fixed cases.
|
||||||
|
- [x] 2.2 Generate a full baseline Markdown report for review.
|
||||||
|
- [x] 2.3 Document how to regenerate and interpret the baseline reports.
|
||||||
|
|
||||||
|
## 3. Tests And Validation
|
||||||
|
|
||||||
|
- [x] 3.1 Add or update focused tests for full fixture coverage and baseline report generation.
|
||||||
|
- [x] 3.2 Run evaluator tests.
|
||||||
|
- [x] 3.3 Run compile verification.
|
||||||
|
- [x] 3.4 Run OpenSpec validation.
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
schema: spec-driven
|
||||||
|
created: 2026-07-04
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
## Context
|
||||||
|
|
||||||
|
The current `mvp/demo` folder documents the core flow, but the steps are embedded in prose. For an interview, the demo needs a sharper entry point: what to start, what to run, what files get produced, and what to point at when explaining Agent engineering quality.
|
||||||
|
|
||||||
|
## Goals / Non-Goals
|
||||||
|
|
||||||
|
**Goals:**
|
||||||
|
|
||||||
|
- Make the payment-timeout demo runnable through a small script.
|
||||||
|
- Save chat, trace, and feedback responses for review.
|
||||||
|
- Provide a short interview walkthrough that connects runtime evidence to the engineering story.
|
||||||
|
- Keep the demo focused on existing APIs and existing `mvp-demo` profile behavior.
|
||||||
|
|
||||||
|
**Non-Goals:**
|
||||||
|
|
||||||
|
- Do not add new backend endpoints.
|
||||||
|
- Do not modify Agent prompts or runtime orchestration.
|
||||||
|
- Do not solve secret cleanup or full offline test isolation in this change.
|
||||||
|
- Do not expand the eval harness.
|
||||||
|
|
||||||
|
## Decisions
|
||||||
|
|
||||||
|
- Decision: Use PowerShell scripts.
|
||||||
|
- Reason: the current runbook already uses PowerShell and the user environment is Windows.
|
||||||
|
|
||||||
|
- Decision: Save outputs under `mvp/demo/output`.
|
||||||
|
- Reason: interview review is easier when chat, trace, and feedback responses are persisted as files.
|
||||||
|
|
||||||
|
- Decision: Keep the walkthrough separate from the low-level runbook.
|
||||||
|
- Reason: `README.md` should tell how to run; `interview-walkthrough.md` should tell how to explain.
|
||||||
|
|
||||||
|
## Risks / Trade-offs
|
||||||
|
|
||||||
|
- The demo still depends on configured MySQL, Redis, Milvus, and model keys. Mitigation: document this explicitly and keep mock log/metric providers enabled through `mvp-demo`.
|
||||||
|
- Script assertions are intentionally lightweight. Mitigation: use the trace checklist for human review and keep automated regression in `mvp/eval`.
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
## Why
|
||||||
|
|
||||||
|
The MVP already has trace, evidence hardening, and evaluation artifacts, but the interview demo path is still too scattered. This change packages the existing capabilities into a repeatable demo runbook that can be executed and explained in a short interview window.
|
||||||
|
|
||||||
|
## What Changes
|
||||||
|
|
||||||
|
- Add a focused interview walkthrough for the payment-timeout MVP demo.
|
||||||
|
- Add reusable request payloads and PowerShell scripts under `mvp/demo`.
|
||||||
|
- Add a trace inspection checklist that maps runtime output to the engineering story.
|
||||||
|
- Keep the change documentation-only and script-only; no backend runtime behavior changes.
|
||||||
|
|
||||||
|
## Capabilities
|
||||||
|
|
||||||
|
### New Capabilities
|
||||||
|
|
||||||
|
- None.
|
||||||
|
|
||||||
|
### Modified Capabilities
|
||||||
|
|
||||||
|
- `mvp-demo-trace-acceptance`: Extend the demo acceptance surface with a repeatable interview runbook and executable local demo scripts.
|
||||||
|
|
||||||
|
## Impact
|
||||||
|
|
||||||
|
- Affects `mvp/demo` documentation and scripts.
|
||||||
|
- Adds issue and devflow tracking files.
|
||||||
|
- No Java production code, API contract, database schema, or dependency changes are expected.
|
||||||
+30
@@ -0,0 +1,30 @@
|
|||||||
|
## ADDED Requirements
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide an interview runbook
|
||||||
|
The MVP demo SHALL include a concise interview runbook that explains how to demonstrate the Agent flow and how to narrate the engineering value.
|
||||||
|
|
||||||
|
#### Scenario: Walkthrough explains the demo story
|
||||||
|
- **WHEN** a developer opens the interview walkthrough
|
||||||
|
- **THEN** it SHALL explain the user question, Agent flow, evidence tools, verifier judgment, trace API, feedback, and eval baseline connection
|
||||||
|
|
||||||
|
#### Scenario: Walkthrough stays scoped to existing capabilities
|
||||||
|
- **WHEN** the walkthrough describes the demo
|
||||||
|
- **THEN** it SHALL avoid claiming unsupported runtime behavior or new production features
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide executable local demo scripts
|
||||||
|
The MVP demo SHALL provide scripts and request payloads for running the payment-timeout case through existing local APIs.
|
||||||
|
|
||||||
|
#### Scenario: Demo script sends the fixed diagnosis request
|
||||||
|
- **WHEN** the demo script is executed against a running local service
|
||||||
|
- **THEN** it SHALL send the fixed payment-timeout chat request with a stable session id
|
||||||
|
|
||||||
|
#### Scenario: Demo script captures review artifacts
|
||||||
|
- **WHEN** the demo script finishes successfully
|
||||||
|
- **THEN** it SHALL write chat, trace, and feedback responses under a demo output directory
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide a trace inspection checklist
|
||||||
|
The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and session-level auditability.
|
||||||
|
|
||||||
|
#### Scenario: Checklist maps fields to interview claims
|
||||||
|
- **WHEN** a developer reviews a trace response
|
||||||
|
- **THEN** the checklist SHALL map concrete JSON paths to the claims made in the interview walkthrough
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
## 1. Demo Artifacts
|
||||||
|
|
||||||
|
- [x] 1.1 Add fixed payment-timeout request payload.
|
||||||
|
- [x] 1.2 Add PowerShell script to run chat, trace, and feedback steps.
|
||||||
|
- [x] 1.3 Add output directory documentation without committing generated outputs.
|
||||||
|
|
||||||
|
## 2. Interview Documentation
|
||||||
|
|
||||||
|
- [x] 2.1 Add interview walkthrough for the demo story.
|
||||||
|
- [x] 2.2 Add trace inspection checklist.
|
||||||
|
- [x] 2.3 Update `mvp/demo/README.md` to link the runnable demo package.
|
||||||
|
|
||||||
|
## 3. Tracking And Validation
|
||||||
|
|
||||||
|
- [x] 3.1 Add slug-based issue and devflow tracking files.
|
||||||
|
- [x] 3.2 Run OpenSpec validation.
|
||||||
@@ -190,3 +190,17 @@ Verifier facts SHALL be linkable to the evidence summaries used during verificat
|
|||||||
- **WHEN** the ChatService persists `verifier_evaluation`
|
- **WHEN** the ChatService persists `verifier_evaluation`
|
||||||
- **THEN** it SHALL include `traceability_version`
|
- **THEN** it SHALL include `traceability_version`
|
||||||
- **AND** it SHALL include the `tool_trace_summary` snapshot used by the Verifier
|
- **AND** it SHALL include the `tool_trace_summary` snapshot used by the Verifier
|
||||||
|
|
||||||
|
### Requirement: Verifier inputs SHALL tolerate hardened no-evidence semantics
|
||||||
|
The verifier integration SHALL continue to work when evidence summaries distinguish failed calls, no-hit calls, and deduped retrievals more explicitly.
|
||||||
|
|
||||||
|
#### Scenario: Failed evidence remains a verifier-visible gap
|
||||||
|
- **WHEN** an evidence-bearing tool invocation fails
|
||||||
|
- **THEN** the verifier-facing trace summary SHALL preserve that failure as a gap
|
||||||
|
- **AND** the verifier flow SHALL continue without crashing
|
||||||
|
|
||||||
|
#### Scenario: Deduped retrievals do not count as fresh support
|
||||||
|
- **WHEN** the verifier-facing trace summary contains deduped `lookup_knowledge` entries
|
||||||
|
- **THEN** those entries SHALL be treated as no-new-evidence
|
||||||
|
- **AND** they SHALL NOT be interpreted as fresh direct support for the answer
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# diagnosis-eval-harness Specification
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Provide a repeatable offline evaluation harness for MVP diagnosis Agent behavior, so prompt, tool, retrieval, and verifier changes can be checked against fixed trace-based regression cases.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL define fixed diagnosis cases
|
||||||
|
The system SHALL provide a small fixed set of MVP diagnosis evaluation cases with explicit expected trace and answer criteria.
|
||||||
|
|
||||||
|
#### Scenario: Case definition includes expected evidence
|
||||||
|
- **WHEN** an evaluation case is defined
|
||||||
|
- **THEN** it SHALL include a case id, user question, expected root-cause keywords, required evidence tools, and allowed verifier verdicts
|
||||||
|
|
||||||
|
#### Scenario: Case definition can express forbidden behavior
|
||||||
|
- **WHEN** a case has known unsafe behavior
|
||||||
|
- **THEN** the case definition SHALL be able to declare forbidden answer keywords or forbidden verdicts
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL validate diagnosis traces
|
||||||
|
The system SHALL validate a diagnosis trace against the corresponding case definition using deterministic rules.
|
||||||
|
|
||||||
|
#### Scenario: Evidence coverage validation
|
||||||
|
- **WHEN** a trace is evaluated
|
||||||
|
- **THEN** the evaluator SHALL verify that required evidence tools appear in `toolInvocations` or verifier trace summaries
|
||||||
|
|
||||||
|
#### Scenario: Verifier evaluation validation
|
||||||
|
- **WHEN** a trace is evaluated
|
||||||
|
- **THEN** the evaluator SHALL verify that `selfEvaluation.verifier_evaluation.verdict` exists
|
||||||
|
- **AND** the verdict SHALL be one of the case's allowed verdicts
|
||||||
|
|
||||||
|
#### Scenario: Answer keyword validation
|
||||||
|
- **WHEN** a trace is evaluated
|
||||||
|
- **THEN** the evaluator SHALL verify that the final answer includes at least the configured minimum root-cause keyword coverage
|
||||||
|
|
||||||
|
#### Scenario: Degraded output validation
|
||||||
|
- **WHEN** a trace verdict is `REJECT`
|
||||||
|
- **THEN** the evaluator SHALL verify that the final answer follows the degraded-output contract rather than leaking an unverified executor answer
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL report quality and cost signals
|
||||||
|
The system SHALL produce a report that summarizes pass/fail results and key trace metrics.
|
||||||
|
|
||||||
|
#### Scenario: JSON report output
|
||||||
|
- **WHEN** an evaluation run completes
|
||||||
|
- **THEN** the evaluator SHALL output a JSON report with per-case result, failed checks, verdict, evidence coverage, tool-call count, and duration
|
||||||
|
|
||||||
|
#### Scenario: Markdown report output
|
||||||
|
- **WHEN** an evaluation run completes
|
||||||
|
- **THEN** the evaluator SHALL output a Markdown report suitable for review in the repository
|
||||||
|
|
||||||
|
#### Scenario: Aggregate metrics
|
||||||
|
- **WHEN** multiple cases are evaluated
|
||||||
|
- **THEN** the report SHALL include aggregate pass rate, verdict distribution, average tool-call count, and average duration when available
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL support offline fixture mode
|
||||||
|
The first evaluator version SHALL be runnable without live MySQL, Redis, Milvus, or LLM services by evaluating saved trace fixtures.
|
||||||
|
|
||||||
|
#### Scenario: Fixture trace evaluation
|
||||||
|
- **WHEN** the evaluator is run against a directory of trace fixture files
|
||||||
|
- **THEN** it SHALL evaluate each trace file against its matching case definition
|
||||||
|
- **AND** it SHALL not require a running application service
|
||||||
|
|
||||||
|
#### Scenario: Missing fixture is reported clearly
|
||||||
|
- **WHEN** a case has no matching trace fixture
|
||||||
|
- **THEN** the evaluator SHALL mark the case as not run or failed with a clear reason
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL provide complete fixture coverage for fixed cases
|
||||||
|
The system SHALL include an offline trace fixture for every fixed MVP diagnosis evaluation case.
|
||||||
|
|
||||||
|
#### Scenario: Every case resolves to a fixture file
|
||||||
|
- **WHEN** the evaluator loads the fixed case definition file
|
||||||
|
- **THEN** every case's `traceFixture` value SHALL resolve to an existing JSON fixture file
|
||||||
|
|
||||||
|
#### Scenario: Fixture files are loadable as diagnosis traces
|
||||||
|
- **WHEN** each referenced fixture is loaded
|
||||||
|
- **THEN** it SHALL deserialize into the trace response shape used by the evaluator
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL preserve a reproducible baseline report
|
||||||
|
The system SHALL preserve a generated baseline report for the full fixed fixture set.
|
||||||
|
|
||||||
|
#### Scenario: Baseline report includes all fixed cases
|
||||||
|
- **WHEN** the baseline report is generated from the fixed case file and fixture directory
|
||||||
|
- **THEN** the report SHALL include one result for every fixed case
|
||||||
|
|
||||||
|
#### Scenario: Baseline report is reviewable
|
||||||
|
- **WHEN** the baseline report is written
|
||||||
|
- **THEN** it SHALL be available in JSON and Markdown formats under the eval documentation area
|
||||||
|
|
||||||
|
#### Scenario: Baseline regeneration is documented
|
||||||
|
- **WHEN** a developer changes fixtures or evaluator rules
|
||||||
|
- **THEN** the eval documentation SHALL explain how to regenerate the baseline report
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL compare reports against a baseline
|
||||||
|
The system SHALL compare a current diagnosis evaluation report against a saved baseline report using deterministic rules.
|
||||||
|
|
||||||
|
#### Scenario: Aggregate regression detection
|
||||||
|
- **WHEN** the current report has a lower pass rate than the baseline report
|
||||||
|
- **THEN** the diff SHALL record a regression with the old value, new value, and delta
|
||||||
|
|
||||||
|
#### Scenario: Cost signal detection
|
||||||
|
- **WHEN** average tool-call count or average duration changes between reports
|
||||||
|
- **THEN** the diff SHALL record the baseline value, current value, and delta
|
||||||
|
|
||||||
|
#### Scenario: Verdict distribution comparison
|
||||||
|
- **WHEN** verdict counts differ between reports
|
||||||
|
- **THEN** the diff SHALL record the verdict distribution changes
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL compare case-level report results
|
||||||
|
The system SHALL compare case results by case id and report actionable per-case changes.
|
||||||
|
|
||||||
|
#### Scenario: Case pass/fail regression
|
||||||
|
- **WHEN** a case changes from passing in the baseline to failing in the current report
|
||||||
|
- **THEN** the diff SHALL record a regression for that case
|
||||||
|
|
||||||
|
#### Scenario: Evidence coverage regression
|
||||||
|
- **WHEN** a required evidence tool changes from covered to uncovered for a case
|
||||||
|
- **THEN** the diff SHALL record a regression naming the case and tool
|
||||||
|
|
||||||
|
#### Scenario: Missing case detection
|
||||||
|
- **WHEN** a baseline case is absent from the current report
|
||||||
|
- **THEN** the diff SHALL record a regression for the missing case
|
||||||
|
|
||||||
|
#### Scenario: New case detection
|
||||||
|
- **WHEN** a current report contains a case absent from the baseline
|
||||||
|
- **THEN** the diff SHALL record the case as a non-regression change
|
||||||
|
|
||||||
|
### Requirement: Evaluation harness SHALL report baseline diff results
|
||||||
|
The system SHALL expose baseline diff output in structured JSON and reviewable Markdown formats.
|
||||||
|
|
||||||
|
#### Scenario: JSON diff output
|
||||||
|
- **WHEN** a baseline diff is written as JSON
|
||||||
|
- **THEN** it SHALL include aggregate summary fields and detailed diff items
|
||||||
|
|
||||||
|
#### Scenario: Markdown diff output
|
||||||
|
- **WHEN** a baseline diff is written as Markdown
|
||||||
|
- **THEN** it SHALL include a readable summary and a table of diff items
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
# evidence-trace-hardening Specification
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
TBD - created by archiving change evidence-trace-hardening. Update Purpose after archive.
|
||||||
|
## Requirements
|
||||||
|
### Requirement: Evidence-bearing tools SHALL persist a unified invocation contract
|
||||||
|
The system SHALL persist evidence-bearing tool calls through a unified contract that guarantees the same baseline fields across `lookup_knowledge`, `query_logs`, and `query_metrics`.
|
||||||
|
|
||||||
|
#### Scenario: Common evidence fields are always persisted
|
||||||
|
- **WHEN** an evidence-bearing tool finishes a call
|
||||||
|
- **THEN** the persisted `tool_invocation` row SHALL include `session_id`, `tool_name`, `input_params`, `duration_ms`, `success`, and an output preview or explicit no-output state
|
||||||
|
|
||||||
|
#### Scenario: Retrieval-aware tools preserve structured retrieval fields
|
||||||
|
- **WHEN** `lookup_knowledge` persists a tool invocation
|
||||||
|
- **THEN** the row SHALL preserve retrieval-specific fields such as retrieval layer, L0/L1 counts, relevance level, dedup reason, and retrieval details
|
||||||
|
- **AND** those fields SHALL be written through the same recorder contract rather than by ad hoc row construction in unrelated code paths
|
||||||
|
|
||||||
|
### Requirement: Evidence trace semantics SHALL distinguish failure from no-evidence outcomes
|
||||||
|
The system SHALL keep failed calls separate from successful calls that return no usable evidence.
|
||||||
|
|
||||||
|
#### Scenario: Tool failure is preserved as failure
|
||||||
|
- **WHEN** an evidence-bearing tool throws, times out, or returns an execution error
|
||||||
|
- **THEN** the persisted row SHALL set `success=false`
|
||||||
|
- **AND** it SHALL preserve an `error_message` explaining the failure
|
||||||
|
|
||||||
|
#### Scenario: No usable evidence is preserved without pretending success
|
||||||
|
- **WHEN** an evidence-bearing tool completes normally but yields no usable evidence for the verifier
|
||||||
|
- **THEN** the persisted contract SHALL preserve that the call completed
|
||||||
|
- **AND** the verifier-facing summary SHALL describe it as a no-evidence outcome rather than direct support
|
||||||
|
|
||||||
|
#### Scenario: Deduped retrieval remains auditable
|
||||||
|
- **WHEN** `lookup_knowledge` is blocked by session-level deduplication
|
||||||
|
- **THEN** the persisted row SHALL preserve the dedup reason
|
||||||
|
- **AND** the verifier-facing summary SHALL treat that row as no-new-evidence rather than as a fresh supporting hit
|
||||||
|
|
||||||
|
### Requirement: Verifier-facing evidence summaries SHALL use stable no-evidence rules
|
||||||
|
The system SHALL summarize persisted tool rows into verifier-facing evidence entries using stable rules for success, failure, no-hit, and deduped outcomes.
|
||||||
|
|
||||||
|
#### Scenario: Failed evidence calls remain visible in the summary
|
||||||
|
- **WHEN** `ToolTraceSummaryService` processes failed evidence-bearing tool rows
|
||||||
|
- **THEN** the summary SHALL retain them
|
||||||
|
- **AND** it SHALL mark them as unsuccessful evidence with an output summary that explains the gap
|
||||||
|
|
||||||
|
#### Scenario: No-hit and deduped calls do not upgrade evidence level
|
||||||
|
- **WHEN** `ToolTraceSummaryService` processes rows that returned no usable evidence or were deduped
|
||||||
|
- **THEN** those rows SHALL NOT be promoted to direct or indirect evidence
|
||||||
|
- **AND** their counts SHALL still be reflected in the merged summary entry
|
||||||
|
|
||||||
|
#### Scenario: Successful evidence keeps the strongest available support
|
||||||
|
- **WHEN** multiple rows for the same tool and topic domain are merged
|
||||||
|
- **THEN** the summary SHALL preserve the strongest successful evidence level among them
|
||||||
|
- **AND** it SHALL also retain repeated-call, failed-call, and no-hit counts for auditability
|
||||||
|
|
||||||
|
### Requirement: ChatService SHALL degrade predictably on verifier output failures
|
||||||
|
The system SHALL treat missing or invalid verifier output as a bounded degraded path instead of an unstructured runtime error.
|
||||||
|
|
||||||
|
#### Scenario: Missing verifier output falls back to LOW_CONFID
|
||||||
|
- **WHEN** the verifier step completes without a usable `verifier_output`
|
||||||
|
- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision
|
||||||
|
- **AND** the final user-facing output SHALL use the fixed low-confidence protocol
|
||||||
|
|
||||||
|
#### Scenario: Invalid verifier JSON falls back to LOW_CONFID
|
||||||
|
- **WHEN** the verifier returns malformed or non-parseable JSON
|
||||||
|
- **THEN** `ChatService` SHALL fall back to a `LOW_CONFID` decision
|
||||||
|
- **AND** the fallback SHALL still persist a verifier-evaluation record
|
||||||
|
|
||||||
|
#### Scenario: REJECT output hides unverified raw answer text
|
||||||
|
- **WHEN** the final verifier decision is `REJECT`
|
||||||
|
- **THEN** the user-facing output SHALL use the degraded template
|
||||||
|
- **AND** it SHALL NOT pass through the raw executor answer
|
||||||
|
|
||||||
|
### Requirement: Evidence-trace hardening SHALL be covered by focused offline tests
|
||||||
|
The system SHALL provide offline tests for the hardened evidence contract and degraded-output behavior.
|
||||||
|
|
||||||
|
#### Scenario: Evidence recorder contract is tested offline
|
||||||
|
- **WHEN** the test suite runs the focused recorder tests
|
||||||
|
- **THEN** it SHALL verify the persisted semantics for successful, failed, and no-evidence evidence-tool calls without requiring external infrastructure
|
||||||
|
|
||||||
|
#### Scenario: Trace summary hardening is tested offline
|
||||||
|
- **WHEN** the test suite runs the focused trace-summary tests
|
||||||
|
- **THEN** it SHALL verify merged summary behavior for mixed success, failure, no-hit, and deduped tool rows
|
||||||
|
|
||||||
|
#### Scenario: Verifier fallback behavior is tested offline
|
||||||
|
- **WHEN** the test suite runs the focused `ChatService` fallback tests
|
||||||
|
- **THEN** it SHALL verify the missing-output, invalid-JSON, `LOW_CONFID`, and `REJECT` degraded paths without requiring a real LLM or database
|
||||||
|
|
||||||
@@ -35,3 +35,32 @@ The project SHALL include an end-to-end acceptance case that demonstrates start-
|
|||||||
#### Scenario: Reviewer follows the acceptance case
|
#### Scenario: Reviewer follows the acceptance case
|
||||||
- **WHEN** a reviewer follows the documented MVP demo acceptance steps
|
- **WHEN** a reviewer follows the documented MVP demo acceptance steps
|
||||||
- **THEN** they can run the application, submit a diagnosis question, query the trace endpoint, and submit feedback for the same session id
|
- **THEN** they can run the application, submit a diagnosis question, query the trace endpoint, and submit feedback for the same session id
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide an interview runbook
|
||||||
|
The MVP demo SHALL include a concise interview runbook that explains how to demonstrate the Agent flow and how to narrate the engineering value.
|
||||||
|
|
||||||
|
#### Scenario: Walkthrough explains the demo story
|
||||||
|
- **WHEN** a developer opens the interview walkthrough
|
||||||
|
- **THEN** it SHALL explain the user question, Agent flow, evidence tools, verifier judgment, trace API, feedback, and eval baseline connection
|
||||||
|
|
||||||
|
#### Scenario: Walkthrough stays scoped to existing capabilities
|
||||||
|
- **WHEN** the walkthrough describes the demo
|
||||||
|
- **THEN** it SHALL avoid claiming unsupported runtime behavior or new production features
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide executable local demo scripts
|
||||||
|
The MVP demo SHALL provide scripts and request payloads for running the payment-timeout case through existing local APIs.
|
||||||
|
|
||||||
|
#### Scenario: Demo script sends the fixed diagnosis request
|
||||||
|
- **WHEN** the demo script is executed against a running local service
|
||||||
|
- **THEN** it SHALL send the fixed payment-timeout chat request with a stable session id
|
||||||
|
|
||||||
|
#### Scenario: Demo script captures review artifacts
|
||||||
|
- **WHEN** the demo script finishes successfully
|
||||||
|
- **THEN** it SHALL write chat, trace, and feedback responses under a demo output directory
|
||||||
|
|
||||||
|
### Requirement: MVP demo SHALL provide a trace inspection checklist
|
||||||
|
The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and session-level auditability.
|
||||||
|
|
||||||
|
#### Scenario: Checklist maps fields to interview claims
|
||||||
|
- **WHEN** a developer reviews a trace response
|
||||||
|
- **THEN** the checklist SHALL map concrete JSON paths to the claims made in the interview walkthrough
|
||||||
|
|||||||
@@ -131,13 +131,15 @@ public class QueryLogsTools {
|
|||||||
output.setMessage(String.format("共有 %d 个可用的日志主题。建议使用默认地域 'ap-guangzhou' 或省略 region 参数", topics.size()));
|
output.setMessage(String.format("共有 %d 个可用的日志主题。建议使用默认地域 'ap-guangzhou' 或省略 region 参数", topics.size()));
|
||||||
|
|
||||||
String response = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output);
|
String response = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output);
|
||||||
recordInvocation(startTime, "get_available_log_topics", null, null, null, response, true, null, "logs");
|
recordInvocation("get_available_log_topics", startTime, "get_available_log_topics", null, null, null,
|
||||||
|
response, true, null, "logs", ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED);
|
||||||
return response;
|
return response;
|
||||||
|
|
||||||
} catch (Exception e) {
|
} catch (Exception e) {
|
||||||
logger.error("获取日志主题列表失败", e);
|
logger.error("获取日志主题列表失败", e);
|
||||||
String response = "{\"success\":false,\"message\":\"获取日志主题列表失败: " + e.getMessage() + "\"}";
|
String response = "{\"success\":false,\"message\":\"获取日志主题列表失败: " + e.getMessage() + "\"}";
|
||||||
recordInvocation(startTime, "get_available_log_topics", null, null, null, response, false, e.getMessage(), "logs");
|
recordInvocation("get_available_log_topics", startTime, "get_available_log_topics", null, null, null,
|
||||||
|
response, false, e.getMessage(), "logs", ToolInvocationRecorder.EVIDENCE_STATUS_FAILED);
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -191,8 +193,9 @@ public class QueryLogsTools {
|
|||||||
} else {
|
} else {
|
||||||
// 真实模式:调用 CLS API(这里预留接口,后续实现)
|
// 真实模式:调用 CLS API(这里预留接口,后续实现)
|
||||||
String response = buildErrorResponse("CLS 真实查询尚未实现,请启用 mock 模式进行测试");
|
String response = buildErrorResponse("CLS 真实查询尚未实现,请启用 mock 模式进行测试");
|
||||||
recordInvocation(startTime, safeQuery, region, logTopic, actualLimit, response, false,
|
recordInvocation("query_logs", startTime, safeQuery, region, logTopic, actualLimit, response, false,
|
||||||
"CLS 真实查询尚未实现,请启用 mock 模式进行测试", normalizeTopicDomain(logTopic));
|
"CLS 真实查询尚未实现,请启用 mock 模式进行测试", normalizeTopicDomain(logTopic),
|
||||||
|
ToolInvocationRecorder.EVIDENCE_STATUS_FAILED);
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -208,23 +211,26 @@ public class QueryLogsTools {
|
|||||||
|
|
||||||
String jsonResult = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output);
|
String jsonResult = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output);
|
||||||
logger.info("日志查询完成: 找到 {} 条日志", logEntries.size());
|
logger.info("日志查询完成: 找到 {} 条日志", logEntries.size());
|
||||||
recordInvocation(startTime, safeQuery, region, logTopic, actualLimit, jsonResult,
|
recordInvocation("query_logs", startTime, safeQuery, region, logTopic, actualLimit, jsonResult,
|
||||||
!logEntries.isEmpty(), logEntries.isEmpty() ? "未找到匹配的日志" : null,
|
true, null, normalizeTopicDomain(logTopic),
|
||||||
normalizeTopicDomain(logTopic));
|
logEntries.isEmpty()
|
||||||
|
? ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE
|
||||||
|
: ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED);
|
||||||
|
|
||||||
return jsonResult;
|
return jsonResult;
|
||||||
|
|
||||||
} catch (Exception e) {
|
} catch (Exception e) {
|
||||||
logger.error("查询日志失败", e);
|
logger.error("查询日志失败", e);
|
||||||
String response = buildErrorResponse("查询失败: " + e.getMessage());
|
String response = buildErrorResponse("查询失败: " + e.getMessage());
|
||||||
recordInvocation(startTime, safeQuery, region, logTopic, actualLimit, response, false,
|
recordInvocation("query_logs", startTime, safeQuery, region, logTopic, actualLimit, response, false,
|
||||||
e.getMessage(), normalizeTopicDomain(logTopic));
|
e.getMessage(), normalizeTopicDomain(logTopic), ToolInvocationRecorder.EVIDENCE_STATUS_FAILED);
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private void recordInvocation(long startTime, String query, String region, String logTopic, Integer limit,
|
private void recordInvocation(String toolName, long startTime, String query, String region, String logTopic, Integer limit,
|
||||||
String output, boolean success, String errorMessage, String topicDomain) {
|
String output, boolean success, String errorMessage, String topicDomain,
|
||||||
|
String evidenceStatus) {
|
||||||
Map<String, Object> input = new HashMap<>();
|
Map<String, Object> input = new HashMap<>();
|
||||||
input.put("query", query == null || query.isBlank() ? "DEFAULT_QUERY" : query);
|
input.put("query", query == null || query.isBlank() ? "DEFAULT_QUERY" : query);
|
||||||
if (region != null) {
|
if (region != null) {
|
||||||
@@ -239,13 +245,15 @@ public class QueryLogsTools {
|
|||||||
input.put("mock_enabled", mockEnabled);
|
input.put("mock_enabled", mockEnabled);
|
||||||
|
|
||||||
toolInvocationRecorder.recordEvidenceTool(
|
toolInvocationRecorder.recordEvidenceTool(
|
||||||
"query_logs",
|
toolName,
|
||||||
input,
|
input,
|
||||||
output,
|
output,
|
||||||
success,
|
success,
|
||||||
startTime,
|
startTime,
|
||||||
errorMessage,
|
errorMessage,
|
||||||
topicDomain
|
topicDomain,
|
||||||
|
evidenceStatus,
|
||||||
|
Map.of("log_topic", logTopic == null ? "" : logTopic)
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -81,7 +81,7 @@ public class QueryMetricsTools {
|
|||||||
|
|
||||||
if (!"success".equals(result.getStatus())) {
|
if (!"success".equals(result.getStatus())) {
|
||||||
String response = buildErrorResponse("Prometheus API 返回非成功状态: " + result.getStatus(), result.getError());
|
String response = buildErrorResponse("Prometheus API 返回非成功状态: " + result.getStatus(), result.getError());
|
||||||
recordInvocation(startTime, response, false, result.getError());
|
recordInvocation(startTime, response, false, result.getError(), ToolInvocationRecorder.EVIDENCE_STATUS_FAILED);
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -119,19 +119,22 @@ public class QueryMetricsTools {
|
|||||||
|
|
||||||
String jsonResult = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output);
|
String jsonResult = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(output);
|
||||||
logger.info("Prometheus 告警查询完成: 找到 {} 个告警", simplifiedAlerts.size());
|
logger.info("Prometheus 告警查询完成: 找到 {} 个告警", simplifiedAlerts.size());
|
||||||
recordInvocation(startTime, jsonResult, true, null);
|
recordInvocation(startTime, jsonResult, true, null,
|
||||||
|
simplifiedAlerts.isEmpty()
|
||||||
|
? ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE
|
||||||
|
: ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED);
|
||||||
|
|
||||||
return jsonResult;
|
return jsonResult;
|
||||||
|
|
||||||
} catch (Exception e) {
|
} catch (Exception e) {
|
||||||
logger.error("查询 Prometheus 告警失败", e);
|
logger.error("查询 Prometheus 告警失败", e);
|
||||||
String response = buildErrorResponse("查询失败", e.getMessage());
|
String response = buildErrorResponse("查询失败", e.getMessage());
|
||||||
recordInvocation(startTime, response, false, e.getMessage());
|
recordInvocation(startTime, response, false, e.getMessage(), ToolInvocationRecorder.EVIDENCE_STATUS_FAILED);
|
||||||
return response;
|
return response;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private void recordInvocation(long startTime, String output, boolean success, String errorMessage) {
|
private void recordInvocation(long startTime, String output, boolean success, String errorMessage, String evidenceStatus) {
|
||||||
toolInvocationRecorder.recordEvidenceTool(
|
toolInvocationRecorder.recordEvidenceTool(
|
||||||
"query_metrics",
|
"query_metrics",
|
||||||
Map.of("query", "active_prometheus_alerts", "mock_enabled", mockEnabled),
|
Map.of("query", "active_prometheus_alerts", "mock_enabled", mockEnabled),
|
||||||
@@ -139,7 +142,9 @@ public class QueryMetricsTools {
|
|||||||
success,
|
success,
|
||||||
startTime,
|
startTime,
|
||||||
errorMessage,
|
errorMessage,
|
||||||
"prometheus_alerts"
|
"prometheus_alerts",
|
||||||
|
evidenceStatus,
|
||||||
|
Map.of("metric_family", "prometheus_alerts")
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,252 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.Comparator;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.LinkedHashSet;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Objects;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.function.Function;
|
||||||
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
|
public class DiagnosisEvalBaselineDiffer {
|
||||||
|
|
||||||
|
private static final String REGRESSION = "REGRESSION";
|
||||||
|
private static final String IMPROVEMENT = "IMPROVEMENT";
|
||||||
|
private static final String CHANGED = "CHANGED";
|
||||||
|
|
||||||
|
public DiagnosisEvalDiffReport compare(DiagnosisEvalReport baseline, DiagnosisEvalReport current) {
|
||||||
|
List<DiagnosisEvalDiffItem> items = new ArrayList<>();
|
||||||
|
|
||||||
|
compareDouble(items, "aggregate", null, "passRate",
|
||||||
|
baseline.getPassRate(), current.getPassRate(), true);
|
||||||
|
compareDouble(items, "aggregate", null, "averageToolCallCount",
|
||||||
|
baseline.getAverageToolCallCount(), current.getAverageToolCallCount(), false);
|
||||||
|
compareDouble(items, "aggregate", null, "averageDurationMs",
|
||||||
|
baseline.getAverageDurationMs(), current.getAverageDurationMs(), false);
|
||||||
|
compareVerdictDistribution(items, baseline.getVerdictDistribution(), current.getVerdictDistribution());
|
||||||
|
compareCases(items, safeResults(baseline), safeResults(current));
|
||||||
|
|
||||||
|
int regressionCount = countType(items, REGRESSION);
|
||||||
|
int improvementCount = countType(items, IMPROVEMENT);
|
||||||
|
int changedCount = countType(items, CHANGED);
|
||||||
|
|
||||||
|
return DiagnosisEvalDiffReport.builder()
|
||||||
|
.baselineTotalCases(baseline.getTotalCases())
|
||||||
|
.currentTotalCases(current.getTotalCases())
|
||||||
|
.baselinePassedCases(baseline.getPassedCases())
|
||||||
|
.currentPassedCases(current.getPassedCases())
|
||||||
|
.baselinePassRate(baseline.getPassRate())
|
||||||
|
.currentPassRate(current.getPassRate())
|
||||||
|
.regressionCount(regressionCount)
|
||||||
|
.improvementCount(improvementCount)
|
||||||
|
.changedCount(changedCount)
|
||||||
|
.hasRegression(regressionCount > 0)
|
||||||
|
.items(items)
|
||||||
|
.build();
|
||||||
|
}
|
||||||
|
|
||||||
|
private void compareVerdictDistribution(List<DiagnosisEvalDiffItem> items,
|
||||||
|
Map<String, Long> baseline,
|
||||||
|
Map<String, Long> current) {
|
||||||
|
Set<String> verdicts = new LinkedHashSet<>();
|
||||||
|
verdicts.addAll(safeMap(baseline).keySet());
|
||||||
|
verdicts.addAll(safeMap(current).keySet());
|
||||||
|
for (String verdict : verdicts) {
|
||||||
|
long baselineCount = safeMap(baseline).getOrDefault(verdict, 0L);
|
||||||
|
long currentCount = safeMap(current).getOrDefault(verdict, 0L);
|
||||||
|
if (baselineCount != currentCount) {
|
||||||
|
items.add(item(CHANGED, "aggregate", null, "verdictDistribution." + verdict,
|
||||||
|
String.valueOf(baselineCount), String.valueOf(currentCount),
|
||||||
|
(double) currentCount - baselineCount,
|
||||||
|
"verdict count changed for " + verdict));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private void compareCases(List<DiagnosisEvalDiffItem> items,
|
||||||
|
List<DiagnosisEvalResult> baselineResults,
|
||||||
|
List<DiagnosisEvalResult> currentResults) {
|
||||||
|
Map<String, DiagnosisEvalResult> baselineById = byCaseId(baselineResults);
|
||||||
|
Map<String, DiagnosisEvalResult> currentById = byCaseId(currentResults);
|
||||||
|
Set<String> caseIds = new LinkedHashSet<>();
|
||||||
|
caseIds.addAll(baselineById.keySet());
|
||||||
|
caseIds.addAll(currentById.keySet());
|
||||||
|
|
||||||
|
for (String caseId : caseIds) {
|
||||||
|
DiagnosisEvalResult baseline = baselineById.get(caseId);
|
||||||
|
DiagnosisEvalResult current = currentById.get(caseId);
|
||||||
|
if (baseline == null) {
|
||||||
|
items.add(item(CHANGED, "case", caseId, "casePresence",
|
||||||
|
"missing", "present", null, "new case appears in current report"));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (current == null) {
|
||||||
|
items.add(item(REGRESSION, "case", caseId, "casePresence",
|
||||||
|
"present", "missing", null, "baseline case is missing from current report"));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
comparePassState(items, baseline, current);
|
||||||
|
compareVerdict(items, baseline, current);
|
||||||
|
compareInteger(items, caseId, "matchedKeywordCount",
|
||||||
|
baseline.getMatchedKeywordCount(), current.getMatchedKeywordCount(), true);
|
||||||
|
compareInteger(items, caseId, "toolCallCount",
|
||||||
|
baseline.getToolCallCount(), current.getToolCallCount(), false);
|
||||||
|
compareInteger(items, caseId, "durationMs",
|
||||||
|
baseline.getDurationMs(), current.getDurationMs(), false);
|
||||||
|
compareEvidenceCoverage(items, baseline, current);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private void comparePassState(List<DiagnosisEvalDiffItem> items,
|
||||||
|
DiagnosisEvalResult baseline,
|
||||||
|
DiagnosisEvalResult current) {
|
||||||
|
if (baseline.isPassed() == current.isPassed()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
String type = baseline.isPassed() ? REGRESSION : IMPROVEMENT;
|
||||||
|
items.add(item(type, "case", baseline.getCaseId(), "passed",
|
||||||
|
String.valueOf(baseline.isPassed()), String.valueOf(current.isPassed()), null,
|
||||||
|
baseline.getCaseId() + " pass state changed"));
|
||||||
|
}
|
||||||
|
|
||||||
|
private void compareVerdict(List<DiagnosisEvalDiffItem> items,
|
||||||
|
DiagnosisEvalResult baseline,
|
||||||
|
DiagnosisEvalResult current) {
|
||||||
|
if (Objects.equals(baseline.getVerdict(), current.getVerdict())) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
int baselineRank = verdictRank(baseline.getVerdict());
|
||||||
|
int currentRank = verdictRank(current.getVerdict());
|
||||||
|
String type = currentRank < baselineRank ? REGRESSION : currentRank > baselineRank ? IMPROVEMENT : CHANGED;
|
||||||
|
items.add(item(type, "case", baseline.getCaseId(), "verdict",
|
||||||
|
value(baseline.getVerdict()), value(current.getVerdict()), (double) currentRank - baselineRank,
|
||||||
|
baseline.getCaseId() + " verdict changed"));
|
||||||
|
}
|
||||||
|
|
||||||
|
private void compareEvidenceCoverage(List<DiagnosisEvalDiffItem> items,
|
||||||
|
DiagnosisEvalResult baseline,
|
||||||
|
DiagnosisEvalResult current) {
|
||||||
|
Set<String> tools = new LinkedHashSet<>();
|
||||||
|
tools.addAll(safeMap(baseline.getEvidenceCoverage()).keySet());
|
||||||
|
tools.addAll(safeMap(current.getEvidenceCoverage()).keySet());
|
||||||
|
for (String tool : tools) {
|
||||||
|
boolean baselineCovered = Boolean.TRUE.equals(safeMap(baseline.getEvidenceCoverage()).get(tool));
|
||||||
|
boolean currentCovered = Boolean.TRUE.equals(safeMap(current.getEvidenceCoverage()).get(tool));
|
||||||
|
if (baselineCovered == currentCovered) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
String type = baselineCovered ? REGRESSION : IMPROVEMENT;
|
||||||
|
items.add(item(type, "case", baseline.getCaseId(), "evidenceCoverage." + tool,
|
||||||
|
String.valueOf(baselineCovered), String.valueOf(currentCovered), null,
|
||||||
|
baseline.getCaseId() + " evidence coverage changed for " + tool));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private void compareDouble(List<DiagnosisEvalDiffItem> items,
|
||||||
|
String scope,
|
||||||
|
String caseId,
|
||||||
|
String metric,
|
||||||
|
double baseline,
|
||||||
|
double current,
|
||||||
|
boolean higherIsBetter) {
|
||||||
|
if (Double.compare(baseline, current) == 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
double delta = current - baseline;
|
||||||
|
String type = classifyDelta(delta, higherIsBetter);
|
||||||
|
items.add(item(type, scope, caseId, metric,
|
||||||
|
String.valueOf(baseline), String.valueOf(current), delta,
|
||||||
|
metric + " changed"));
|
||||||
|
}
|
||||||
|
|
||||||
|
private void compareInteger(List<DiagnosisEvalDiffItem> items,
|
||||||
|
String caseId,
|
||||||
|
String metric,
|
||||||
|
Integer baseline,
|
||||||
|
Integer current,
|
||||||
|
boolean higherIsBetter) {
|
||||||
|
if (Objects.equals(baseline, current)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (baseline == null || current == null) {
|
||||||
|
items.add(item(CHANGED, "case", caseId, metric,
|
||||||
|
value(baseline), value(current), null, caseId + " " + metric + " changed"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
int delta = current - baseline;
|
||||||
|
items.add(item(classifyDelta(delta, higherIsBetter), "case", caseId, metric,
|
||||||
|
String.valueOf(baseline), String.valueOf(current), (double) delta,
|
||||||
|
caseId + " " + metric + " changed"));
|
||||||
|
}
|
||||||
|
|
||||||
|
private String classifyDelta(double delta, boolean higherIsBetter) {
|
||||||
|
if (delta == 0.0) {
|
||||||
|
return CHANGED;
|
||||||
|
}
|
||||||
|
boolean improved = higherIsBetter ? delta > 0 : delta < 0;
|
||||||
|
return improved ? IMPROVEMENT : REGRESSION;
|
||||||
|
}
|
||||||
|
|
||||||
|
private DiagnosisEvalDiffItem item(String type,
|
||||||
|
String scope,
|
||||||
|
String caseId,
|
||||||
|
String metric,
|
||||||
|
String baselineValue,
|
||||||
|
String currentValue,
|
||||||
|
Double delta,
|
||||||
|
String message) {
|
||||||
|
return DiagnosisEvalDiffItem.builder()
|
||||||
|
.type(type)
|
||||||
|
.scope(scope)
|
||||||
|
.caseId(caseId)
|
||||||
|
.metric(metric)
|
||||||
|
.baselineValue(baselineValue)
|
||||||
|
.currentValue(currentValue)
|
||||||
|
.delta(delta)
|
||||||
|
.message(message)
|
||||||
|
.build();
|
||||||
|
}
|
||||||
|
|
||||||
|
private Map<String, DiagnosisEvalResult> byCaseId(List<DiagnosisEvalResult> results) {
|
||||||
|
return results.stream()
|
||||||
|
.sorted(Comparator.comparing(DiagnosisEvalResult::getCaseId))
|
||||||
|
.collect(Collectors.toMap(
|
||||||
|
DiagnosisEvalResult::getCaseId,
|
||||||
|
Function.identity(),
|
||||||
|
(left, right) -> right,
|
||||||
|
LinkedHashMap::new));
|
||||||
|
}
|
||||||
|
|
||||||
|
private List<DiagnosisEvalResult> safeResults(DiagnosisEvalReport report) {
|
||||||
|
return report.getResults() == null ? List.of() : report.getResults();
|
||||||
|
}
|
||||||
|
|
||||||
|
private <T> Map<String, T> safeMap(Map<String, T> value) {
|
||||||
|
return value == null ? Map.of() : value;
|
||||||
|
}
|
||||||
|
|
||||||
|
private int countType(List<DiagnosisEvalDiffItem> items, String type) {
|
||||||
|
return (int) items.stream().filter(item -> type.equals(item.getType())).count();
|
||||||
|
}
|
||||||
|
|
||||||
|
private int verdictRank(String verdict) {
|
||||||
|
if ("PASS".equals(verdict)) {
|
||||||
|
return 3;
|
||||||
|
}
|
||||||
|
if ("LOW_CONFID".equals(verdict)) {
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
if ("REJECT".equals(verdict)) {
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
private String value(Object value) {
|
||||||
|
return value == null ? "-" : String.valueOf(value);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import lombok.AllArgsConstructor;
|
||||||
|
import lombok.Builder;
|
||||||
|
import lombok.Data;
|
||||||
|
import lombok.NoArgsConstructor;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
@Data
|
||||||
|
@Builder
|
||||||
|
@NoArgsConstructor
|
||||||
|
@AllArgsConstructor
|
||||||
|
public class DiagnosisEvalCase {
|
||||||
|
|
||||||
|
private String id;
|
||||||
|
private String title;
|
||||||
|
private String question;
|
||||||
|
private String traceFixture;
|
||||||
|
private List<String> expectedRootCauseKeywords;
|
||||||
|
private Integer minKeywordMatches;
|
||||||
|
private List<String> requiredEvidenceTools;
|
||||||
|
private List<String> allowedVerdicts;
|
||||||
|
private List<String> forbiddenAnswerKeywords;
|
||||||
|
}
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import lombok.AllArgsConstructor;
|
||||||
|
import lombok.Builder;
|
||||||
|
import lombok.Data;
|
||||||
|
import lombok.NoArgsConstructor;
|
||||||
|
|
||||||
|
@Data
|
||||||
|
@Builder
|
||||||
|
@NoArgsConstructor
|
||||||
|
@AllArgsConstructor
|
||||||
|
public class DiagnosisEvalDiffItem {
|
||||||
|
|
||||||
|
private String type;
|
||||||
|
private String scope;
|
||||||
|
private String caseId;
|
||||||
|
private String metric;
|
||||||
|
private String baselineValue;
|
||||||
|
private String currentValue;
|
||||||
|
private Double delta;
|
||||||
|
private String message;
|
||||||
|
}
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import lombok.AllArgsConstructor;
|
||||||
|
import lombok.Builder;
|
||||||
|
import lombok.Data;
|
||||||
|
import lombok.NoArgsConstructor;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
@Data
|
||||||
|
@Builder
|
||||||
|
@NoArgsConstructor
|
||||||
|
@AllArgsConstructor
|
||||||
|
public class DiagnosisEvalDiffReport {
|
||||||
|
|
||||||
|
private int baselineTotalCases;
|
||||||
|
private int currentTotalCases;
|
||||||
|
private int baselinePassedCases;
|
||||||
|
private int currentPassedCases;
|
||||||
|
private double baselinePassRate;
|
||||||
|
private double currentPassRate;
|
||||||
|
private int regressionCount;
|
||||||
|
private int improvementCount;
|
||||||
|
private int changedCount;
|
||||||
|
private boolean hasRegression;
|
||||||
|
private List<DiagnosisEvalDiffItem> items;
|
||||||
|
}
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
|
||||||
|
public class DiagnosisEvalDiffReportWriter {
|
||||||
|
|
||||||
|
private final ObjectMapper objectMapper;
|
||||||
|
|
||||||
|
public DiagnosisEvalDiffReportWriter(ObjectMapper objectMapper) {
|
||||||
|
this.objectMapper = objectMapper;
|
||||||
|
}
|
||||||
|
|
||||||
|
public void writeJson(DiagnosisEvalDiffReport report, Path outputFile) throws IOException {
|
||||||
|
Files.createDirectories(outputFile.getParent());
|
||||||
|
objectMapper.writerWithDefaultPrettyPrinter().writeValue(outputFile.toFile(), report);
|
||||||
|
}
|
||||||
|
|
||||||
|
public void writeMarkdown(DiagnosisEvalDiffReport report, Path outputFile) throws IOException {
|
||||||
|
Files.createDirectories(outputFile.getParent());
|
||||||
|
Files.writeString(outputFile, toMarkdown(report), StandardCharsets.UTF_8);
|
||||||
|
}
|
||||||
|
|
||||||
|
public String toMarkdown(DiagnosisEvalDiffReport report) {
|
||||||
|
StringBuilder builder = new StringBuilder();
|
||||||
|
builder.append("# Diagnosis Eval Baseline Diff\n\n");
|
||||||
|
builder.append("- Baseline pass rate: ").append(formatPercent(report.getBaselinePassRate())).append("\n");
|
||||||
|
builder.append("- Current pass rate: ").append(formatPercent(report.getCurrentPassRate())).append("\n");
|
||||||
|
builder.append("- Baseline passed cases: ").append(report.getBaselinePassedCases()).append("/")
|
||||||
|
.append(report.getBaselineTotalCases()).append("\n");
|
||||||
|
builder.append("- Current passed cases: ").append(report.getCurrentPassedCases()).append("/")
|
||||||
|
.append(report.getCurrentTotalCases()).append("\n");
|
||||||
|
builder.append("- Regressions: ").append(report.getRegressionCount()).append("\n");
|
||||||
|
builder.append("- Improvements: ").append(report.getImprovementCount()).append("\n");
|
||||||
|
builder.append("- Other changes: ").append(report.getChangedCount()).append("\n\n");
|
||||||
|
|
||||||
|
builder.append("## Diff Items\n\n");
|
||||||
|
if (report.getItems() == null || report.getItems().isEmpty()) {
|
||||||
|
builder.append("- No differences\n");
|
||||||
|
return builder.toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
builder.append("| Type | Scope | Case | Metric | Baseline | Current | Delta | Message |\n");
|
||||||
|
builder.append("| --- | --- | --- | --- | --- | --- | ---: | --- |\n");
|
||||||
|
for (DiagnosisEvalDiffItem item : report.getItems()) {
|
||||||
|
builder.append("| ")
|
||||||
|
.append(valueOrDash(item.getType()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(item.getScope()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(item.getCaseId()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(item.getMetric()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(item.getBaselineValue()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(item.getCurrentValue()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(item.getDelta() == null ? "-" : String.format("%.3f", item.getDelta()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(item.getMessage()))
|
||||||
|
.append(" |\n");
|
||||||
|
}
|
||||||
|
return builder.toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
private String formatPercent(double value) {
|
||||||
|
return String.format("%.2f%%", value * 100);
|
||||||
|
}
|
||||||
|
|
||||||
|
private String valueOrDash(String value) {
|
||||||
|
return value == null || value.isBlank() ? "-" : value;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,24 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import lombok.AllArgsConstructor;
|
||||||
|
import lombok.Builder;
|
||||||
|
import lombok.Data;
|
||||||
|
import lombok.NoArgsConstructor;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
@Data
|
||||||
|
@Builder
|
||||||
|
@NoArgsConstructor
|
||||||
|
@AllArgsConstructor
|
||||||
|
public class DiagnosisEvalReport {
|
||||||
|
|
||||||
|
private int totalCases;
|
||||||
|
private int passedCases;
|
||||||
|
private double passRate;
|
||||||
|
private Map<String, Long> verdictDistribution;
|
||||||
|
private double averageToolCallCount;
|
||||||
|
private double averageDurationMs;
|
||||||
|
private List<DiagnosisEvalResult> results;
|
||||||
|
}
|
||||||
@@ -0,0 +1,76 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
public class DiagnosisEvalReportWriter {
|
||||||
|
|
||||||
|
private final ObjectMapper objectMapper;
|
||||||
|
|
||||||
|
public DiagnosisEvalReportWriter(ObjectMapper objectMapper) {
|
||||||
|
this.objectMapper = objectMapper;
|
||||||
|
}
|
||||||
|
|
||||||
|
public void writeJson(DiagnosisEvalReport report, Path outputFile) throws IOException {
|
||||||
|
Files.createDirectories(outputFile.getParent());
|
||||||
|
objectMapper.writerWithDefaultPrettyPrinter().writeValue(outputFile.toFile(), report);
|
||||||
|
}
|
||||||
|
|
||||||
|
public void writeMarkdown(DiagnosisEvalReport report, Path outputFile) throws IOException {
|
||||||
|
Files.createDirectories(outputFile.getParent());
|
||||||
|
Files.writeString(outputFile, toMarkdown(report), StandardCharsets.UTF_8);
|
||||||
|
}
|
||||||
|
|
||||||
|
public String toMarkdown(DiagnosisEvalReport report) {
|
||||||
|
StringBuilder builder = new StringBuilder();
|
||||||
|
builder.append("# Diagnosis Eval Report\n\n");
|
||||||
|
builder.append("- Total cases: ").append(report.getTotalCases()).append("\n");
|
||||||
|
builder.append("- Passed cases: ").append(report.getPassedCases()).append("\n");
|
||||||
|
builder.append("- Pass rate: ").append(String.format("%.2f%%", report.getPassRate() * 100)).append("\n");
|
||||||
|
builder.append("- Average tool calls: ").append(String.format("%.2f", report.getAverageToolCallCount())).append("\n");
|
||||||
|
builder.append("- Average duration ms: ").append(String.format("%.2f", report.getAverageDurationMs())).append("\n\n");
|
||||||
|
|
||||||
|
builder.append("## Verdict Distribution\n\n");
|
||||||
|
if (report.getVerdictDistribution() == null || report.getVerdictDistribution().isEmpty()) {
|
||||||
|
builder.append("- None\n\n");
|
||||||
|
} else {
|
||||||
|
for (Map.Entry<String, Long> entry : report.getVerdictDistribution().entrySet()) {
|
||||||
|
builder.append("- ").append(entry.getKey()).append(": ").append(entry.getValue()).append("\n");
|
||||||
|
}
|
||||||
|
builder.append("\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
builder.append("## Cases\n\n");
|
||||||
|
builder.append("| Case | Result | Verdict | Keywords | Tool Calls | Duration ms | Failed Checks |\n");
|
||||||
|
builder.append("| --- | --- | --- | --- | ---: | ---: | --- |\n");
|
||||||
|
for (DiagnosisEvalResult result : report.getResults()) {
|
||||||
|
builder.append("| ")
|
||||||
|
.append(result.getCaseId())
|
||||||
|
.append(" | ")
|
||||||
|
.append(result.isPassed() ? "PASS" : "FAIL")
|
||||||
|
.append(" | ")
|
||||||
|
.append(valueOrDash(result.getVerdict()))
|
||||||
|
.append(" | ")
|
||||||
|
.append(result.getMatchedKeywordCount()).append("/").append(result.getRequiredKeywordCount())
|
||||||
|
.append(" | ")
|
||||||
|
.append(result.getToolCallCount() == null ? "-" : result.getToolCallCount())
|
||||||
|
.append(" | ")
|
||||||
|
.append(result.getDurationMs() == null ? "-" : result.getDurationMs())
|
||||||
|
.append(" | ")
|
||||||
|
.append(result.getFailedChecks() == null || result.getFailedChecks().isEmpty()
|
||||||
|
? "-"
|
||||||
|
: String.join("; ", result.getFailedChecks()))
|
||||||
|
.append(" |\n");
|
||||||
|
}
|
||||||
|
return builder.toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
private String valueOrDash(String value) {
|
||||||
|
return value == null || value.isBlank() ? "-" : value;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import lombok.AllArgsConstructor;
|
||||||
|
import lombok.Builder;
|
||||||
|
import lombok.Data;
|
||||||
|
import lombok.NoArgsConstructor;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
@Data
|
||||||
|
@Builder
|
||||||
|
@NoArgsConstructor
|
||||||
|
@AllArgsConstructor
|
||||||
|
public class DiagnosisEvalResult {
|
||||||
|
|
||||||
|
private String caseId;
|
||||||
|
private String title;
|
||||||
|
private boolean passed;
|
||||||
|
private List<String> failedChecks;
|
||||||
|
private String verdict;
|
||||||
|
private int matchedKeywordCount;
|
||||||
|
private int requiredKeywordCount;
|
||||||
|
private Map<String, Boolean> evidenceCoverage;
|
||||||
|
private Integer toolCallCount;
|
||||||
|
private Integer durationMs;
|
||||||
|
}
|
||||||
@@ -0,0 +1,217 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.core.type.TypeReference;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import com.superbiz.agent.dto.DiagnosisTraceResponse;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.LinkedHashSet;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Locale;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Objects;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
|
public class DiagnosisTraceEvaluator {
|
||||||
|
|
||||||
|
private static final TypeReference<List<DiagnosisEvalCase>> CASE_LIST_TYPE = new TypeReference<>() {};
|
||||||
|
private static final String REJECT_DEGRADED_PREFIX = "当前无法基于已获取证据生成可靠结论";
|
||||||
|
|
||||||
|
private final ObjectMapper objectMapper;
|
||||||
|
|
||||||
|
public DiagnosisTraceEvaluator(ObjectMapper objectMapper) {
|
||||||
|
this.objectMapper = objectMapper;
|
||||||
|
}
|
||||||
|
|
||||||
|
public List<DiagnosisEvalCase> loadCases(Path casesFile) throws IOException {
|
||||||
|
return objectMapper.readValue(casesFile.toFile(), CASE_LIST_TYPE);
|
||||||
|
}
|
||||||
|
|
||||||
|
public DiagnosisTraceResponse loadTrace(Path traceFile) throws IOException {
|
||||||
|
return objectMapper.readValue(traceFile.toFile(), DiagnosisTraceResponse.class);
|
||||||
|
}
|
||||||
|
|
||||||
|
public DiagnosisEvalReport evaluate(List<DiagnosisEvalCase> cases, Path fixtureDir) {
|
||||||
|
List<DiagnosisEvalResult> results = new ArrayList<>();
|
||||||
|
for (DiagnosisEvalCase evalCase : cases) {
|
||||||
|
try {
|
||||||
|
DiagnosisTraceResponse trace = loadTrace(fixtureDir.resolve(evalCase.getTraceFixture()));
|
||||||
|
results.add(evaluate(evalCase, trace));
|
||||||
|
} catch (Exception e) {
|
||||||
|
results.add(DiagnosisEvalResult.builder()
|
||||||
|
.caseId(evalCase.getId())
|
||||||
|
.title(evalCase.getTitle())
|
||||||
|
.passed(false)
|
||||||
|
.failedChecks(List.of("trace fixture unavailable: " + e.getMessage()))
|
||||||
|
.verdict(null)
|
||||||
|
.matchedKeywordCount(0)
|
||||||
|
.requiredKeywordCount(size(evalCase.getExpectedRootCauseKeywords()))
|
||||||
|
.evidenceCoverage(emptyCoverage(evalCase.getRequiredEvidenceTools()))
|
||||||
|
.toolCallCount(null)
|
||||||
|
.durationMs(null)
|
||||||
|
.build());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return toReport(results);
|
||||||
|
}
|
||||||
|
|
||||||
|
public DiagnosisEvalResult evaluate(DiagnosisEvalCase evalCase, DiagnosisTraceResponse trace) {
|
||||||
|
List<String> failedChecks = new ArrayList<>();
|
||||||
|
String answer = trace.getSession() == null ? "" : nullToEmpty(trace.getSession().getAnswer());
|
||||||
|
String normalizedAnswer = answer.toLowerCase(Locale.ROOT);
|
||||||
|
|
||||||
|
int requiredKeywordCount = size(evalCase.getExpectedRootCauseKeywords());
|
||||||
|
int matchedKeywordCount = countMatches(normalizedAnswer, evalCase.getExpectedRootCauseKeywords());
|
||||||
|
int minKeywordMatches = evalCase.getMinKeywordMatches() == null
|
||||||
|
? requiredKeywordCount
|
||||||
|
: evalCase.getMinKeywordMatches();
|
||||||
|
if (matchedKeywordCount < minKeywordMatches) {
|
||||||
|
failedChecks.add("answer keyword coverage too low: " + matchedKeywordCount + "/" + minKeywordMatches);
|
||||||
|
}
|
||||||
|
|
||||||
|
for (String forbidden : safeList(evalCase.getForbiddenAnswerKeywords())) {
|
||||||
|
if (normalizedAnswer.contains(forbidden.toLowerCase(Locale.ROOT))) {
|
||||||
|
failedChecks.add("answer contains forbidden keyword: " + forbidden);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Set<String> evidenceTools = collectEvidenceTools(trace);
|
||||||
|
Map<String, Boolean> evidenceCoverage = new LinkedHashMap<>();
|
||||||
|
for (String requiredTool : safeList(evalCase.getRequiredEvidenceTools())) {
|
||||||
|
boolean present = evidenceTools.contains(requiredTool);
|
||||||
|
evidenceCoverage.put(requiredTool, present);
|
||||||
|
if (!present) {
|
||||||
|
failedChecks.add("missing required evidence tool: " + requiredTool);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
String verdict = extractVerifierVerdict(trace);
|
||||||
|
if (verdict == null || verdict.isBlank()) {
|
||||||
|
failedChecks.add("missing verifier verdict");
|
||||||
|
} else if (!safeList(evalCase.getAllowedVerdicts()).isEmpty()
|
||||||
|
&& !safeList(evalCase.getAllowedVerdicts()).contains(verdict)) {
|
||||||
|
failedChecks.add("verdict not allowed: " + verdict);
|
||||||
|
}
|
||||||
|
|
||||||
|
if ("REJECT".equals(verdict) && !answer.startsWith(REJECT_DEGRADED_PREFIX)) {
|
||||||
|
failedChecks.add("reject output does not use degraded template");
|
||||||
|
}
|
||||||
|
|
||||||
|
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
|
||||||
|
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
|
||||||
|
|
||||||
|
return DiagnosisEvalResult.builder()
|
||||||
|
.caseId(evalCase.getId())
|
||||||
|
.title(evalCase.getTitle())
|
||||||
|
.passed(failedChecks.isEmpty())
|
||||||
|
.failedChecks(failedChecks)
|
||||||
|
.verdict(verdict)
|
||||||
|
.matchedKeywordCount(matchedKeywordCount)
|
||||||
|
.requiredKeywordCount(requiredKeywordCount)
|
||||||
|
.evidenceCoverage(evidenceCoverage)
|
||||||
|
.toolCallCount(toolCallCount)
|
||||||
|
.durationMs(durationMs)
|
||||||
|
.build();
|
||||||
|
}
|
||||||
|
|
||||||
|
private DiagnosisEvalReport toReport(List<DiagnosisEvalResult> results) {
|
||||||
|
int total = results.size();
|
||||||
|
int passed = (int) results.stream().filter(DiagnosisEvalResult::isPassed).count();
|
||||||
|
Map<String, Long> verdictDistribution = results.stream()
|
||||||
|
.map(DiagnosisEvalResult::getVerdict)
|
||||||
|
.filter(Objects::nonNull)
|
||||||
|
.collect(Collectors.groupingBy(value -> value, LinkedHashMap::new, Collectors.counting()));
|
||||||
|
double averageToolCallCount = results.stream()
|
||||||
|
.map(DiagnosisEvalResult::getToolCallCount)
|
||||||
|
.filter(Objects::nonNull)
|
||||||
|
.mapToInt(Integer::intValue)
|
||||||
|
.average()
|
||||||
|
.orElse(0.0);
|
||||||
|
double averageDurationMs = results.stream()
|
||||||
|
.map(DiagnosisEvalResult::getDurationMs)
|
||||||
|
.filter(Objects::nonNull)
|
||||||
|
.mapToInt(Integer::intValue)
|
||||||
|
.average()
|
||||||
|
.orElse(0.0);
|
||||||
|
|
||||||
|
return DiagnosisEvalReport.builder()
|
||||||
|
.totalCases(total)
|
||||||
|
.passedCases(passed)
|
||||||
|
.passRate(total == 0 ? 0.0 : (double) passed / total)
|
||||||
|
.verdictDistribution(verdictDistribution)
|
||||||
|
.averageToolCallCount(averageToolCallCount)
|
||||||
|
.averageDurationMs(averageDurationMs)
|
||||||
|
.results(results)
|
||||||
|
.build();
|
||||||
|
}
|
||||||
|
|
||||||
|
private Set<String> collectEvidenceTools(DiagnosisTraceResponse trace) {
|
||||||
|
Set<String> tools = new LinkedHashSet<>();
|
||||||
|
if (trace.getToolInvocations() != null) {
|
||||||
|
for (DiagnosisTraceResponse.ToolInvocationTrace invocation : trace.getToolInvocations()) {
|
||||||
|
if (invocation.getToolName() != null) {
|
||||||
|
tools.add(invocation.getToolName());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Object summaries = nestedValue(trace, "verifier_evaluation", "tool_trace_summary");
|
||||||
|
if (summaries instanceof List<?> list) {
|
||||||
|
for (Object item : list) {
|
||||||
|
if (item instanceof Map<?, ?> map && map.get("tool_name") != null) {
|
||||||
|
tools.add(String.valueOf(map.get("tool_name")));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return tools;
|
||||||
|
}
|
||||||
|
|
||||||
|
private String extractVerifierVerdict(DiagnosisTraceResponse trace) {
|
||||||
|
Object value = nestedValue(trace, "verifier_evaluation", "verdict");
|
||||||
|
return value == null ? null : String.valueOf(value);
|
||||||
|
}
|
||||||
|
|
||||||
|
private Object nestedValue(DiagnosisTraceResponse trace, String firstKey, String secondKey) {
|
||||||
|
if (trace.getSession() == null || trace.getSession().getSelfEvaluation() == null) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
Object first = trace.getSession().getSelfEvaluation().get(firstKey);
|
||||||
|
if (!(first instanceof Map<?, ?> map)) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
return map.get(secondKey);
|
||||||
|
}
|
||||||
|
|
||||||
|
private int countMatches(String normalizedAnswer, List<String> keywords) {
|
||||||
|
int count = 0;
|
||||||
|
for (String keyword : safeList(keywords)) {
|
||||||
|
if (normalizedAnswer.contains(keyword.toLowerCase(Locale.ROOT))) {
|
||||||
|
count++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
private Map<String, Boolean> emptyCoverage(List<String> tools) {
|
||||||
|
Map<String, Boolean> coverage = new LinkedHashMap<>();
|
||||||
|
for (String tool : safeList(tools)) {
|
||||||
|
coverage.put(tool, false);
|
||||||
|
}
|
||||||
|
return coverage;
|
||||||
|
}
|
||||||
|
|
||||||
|
private List<String> safeList(List<String> values) {
|
||||||
|
return values == null ? List.of() : values;
|
||||||
|
}
|
||||||
|
|
||||||
|
private int size(List<?> values) {
|
||||||
|
return values == null ? 0 : values.size();
|
||||||
|
}
|
||||||
|
|
||||||
|
private String nullToEmpty(String value) {
|
||||||
|
return value == null ? "" : value;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -3,11 +3,16 @@ package com.superbiz.agent.service;
|
|||||||
import com.fasterxml.jackson.core.JsonProcessingException;
|
import com.fasterxml.jackson.core.JsonProcessingException;
|
||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||||
|
import com.superbiz.agent.dto.LookupResult;
|
||||||
import com.superbiz.agent.repository.ToolInvocationRepository;
|
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||||||
|
import com.superbiz.agent.dto.KnowledgeEntry;
|
||||||
|
import com.superbiz.agent.service.VectorSearchService;
|
||||||
import com.superbiz.agent.util.SessionContextHolder;
|
import com.superbiz.agent.util.SessionContextHolder;
|
||||||
|
import lombok.Builder;
|
||||||
import lombok.extern.slf4j.Slf4j;
|
import lombok.extern.slf4j.Slf4j;
|
||||||
import org.springframework.stereotype.Service;
|
import org.springframework.stereotype.Service;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
import java.util.LinkedHashMap;
|
import java.util.LinkedHashMap;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
@@ -21,6 +26,10 @@ import java.util.UUID;
|
|||||||
public class ToolInvocationRecorder {
|
public class ToolInvocationRecorder {
|
||||||
|
|
||||||
private static final int OUTPUT_PREVIEW_LIMIT = 500;
|
private static final int OUTPUT_PREVIEW_LIMIT = 500;
|
||||||
|
public static final String EVIDENCE_STATUS_SUPPORTED = "supported";
|
||||||
|
public static final String EVIDENCE_STATUS_NO_EVIDENCE = "no_evidence";
|
||||||
|
public static final String EVIDENCE_STATUS_DEDUPED = "deduped";
|
||||||
|
public static final String EVIDENCE_STATUS_FAILED = "failed";
|
||||||
|
|
||||||
private final ToolInvocationRepository toolInvocationRepository;
|
private final ToolInvocationRepository toolInvocationRepository;
|
||||||
private final ObjectMapper objectMapper;
|
private final ObjectMapper objectMapper;
|
||||||
@@ -52,12 +61,29 @@ public class ToolInvocationRecorder {
|
|||||||
long startTimeMillis,
|
long startTimeMillis,
|
||||||
String errorMessage,
|
String errorMessage,
|
||||||
String topicDomain) {
|
String topicDomain) {
|
||||||
|
recordEvidenceTool(toolName, inputParams, output, success, startTimeMillis, errorMessage, topicDomain,
|
||||||
|
success ? EVIDENCE_STATUS_SUPPORTED : EVIDENCE_STATUS_FAILED, Map.of());
|
||||||
|
}
|
||||||
|
|
||||||
|
public void recordEvidenceTool(String toolName,
|
||||||
|
Map<String, Object> inputParams,
|
||||||
|
String output,
|
||||||
|
boolean success,
|
||||||
|
long startTimeMillis,
|
||||||
|
String errorMessage,
|
||||||
|
String topicDomain,
|
||||||
|
String evidenceStatus,
|
||||||
|
Map<String, Object> extraDetails) {
|
||||||
String outputPreview = preview(output);
|
String outputPreview = preview(output);
|
||||||
Map<String, Object> details = new LinkedHashMap<>();
|
Map<String, Object> details = new LinkedHashMap<>();
|
||||||
details.put("trace_id", UUID.randomUUID().toString());
|
details.put("trace_id", UUID.randomUUID().toString());
|
||||||
if (topicDomain != null && !topicDomain.isBlank()) {
|
if (topicDomain != null && !topicDomain.isBlank()) {
|
||||||
details.put("retrieved_domains", List.of(topicDomain));
|
details.put("retrieved_domains", List.of(topicDomain));
|
||||||
}
|
}
|
||||||
|
details.put("evidence_status", normalizeEvidenceStatus(success, evidenceStatus));
|
||||||
|
if (extraDetails != null && !extraDetails.isEmpty()) {
|
||||||
|
details.putAll(extraDetails);
|
||||||
|
}
|
||||||
|
|
||||||
ToolInvocation invocation = ToolInvocation.builder()
|
ToolInvocation invocation = ToolInvocation.builder()
|
||||||
.toolName(toolName)
|
.toolName(toolName)
|
||||||
@@ -73,6 +99,67 @@ public class ToolInvocationRecorder {
|
|||||||
save(invocation);
|
save(invocation);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public void recordLookupKnowledge(LookupKnowledgeRecord record) {
|
||||||
|
Map<String, Object> details = new LinkedHashMap<>();
|
||||||
|
details.put("trace_id", UUID.randomUUID().toString());
|
||||||
|
if (record.l0MatchCount() != null) {
|
||||||
|
details.put("l0_match_count", record.l0MatchCount());
|
||||||
|
}
|
||||||
|
if (record.l0Titles() != null && !record.l0Titles().isEmpty()) {
|
||||||
|
details.put("l0_titles", record.l0Titles());
|
||||||
|
}
|
||||||
|
if (record.l1TopScore() != null) {
|
||||||
|
details.put("l1_top_score", record.l1TopScore());
|
||||||
|
}
|
||||||
|
if (record.l1TopSimilarity() != null) {
|
||||||
|
details.put("l1_top_similarity", record.l1TopSimilarity());
|
||||||
|
}
|
||||||
|
if (record.l1MatchCount() != null) {
|
||||||
|
details.put("l1_match_count", record.l1MatchCount());
|
||||||
|
}
|
||||||
|
if (record.l1Scores() != null && !record.l1Scores().isEmpty()) {
|
||||||
|
details.put("l1_scores", record.l1Scores());
|
||||||
|
}
|
||||||
|
if (record.relevanceLevel() != null) {
|
||||||
|
details.put("relevance_level", record.relevanceLevel());
|
||||||
|
}
|
||||||
|
if (record.completenessHint() != null) {
|
||||||
|
details.put("completeness_hint", record.completenessHint());
|
||||||
|
}
|
||||||
|
if (record.domain() != null && !record.domain().isBlank()) {
|
||||||
|
details.put("retrieved_domains", List.of(record.domain()));
|
||||||
|
}
|
||||||
|
if (record.dedupReason() != null) {
|
||||||
|
details.put("dedup_reason", record.dedupReason());
|
||||||
|
}
|
||||||
|
details.put("evidence_status", normalizeEvidenceStatus(record.success(), record.evidenceStatus()));
|
||||||
|
|
||||||
|
ToolInvocation invocation = ToolInvocation.builder()
|
||||||
|
.toolName("lookup_knowledge")
|
||||||
|
.inputParams(toJson(Map.of("query", record.query())))
|
||||||
|
.outputPreview(preview(record.outputPreview()))
|
||||||
|
.outputLength(record.outputLength())
|
||||||
|
.retrievalLayer(record.retrievalLayer())
|
||||||
|
.l0MatchCount(record.l0MatchCount())
|
||||||
|
.l1MatchCount(record.l1MatchCount())
|
||||||
|
.isTruncated(Boolean.TRUE.equals(record.truncated()))
|
||||||
|
.retrievalDetails(toJson(details))
|
||||||
|
.relevanceLevel(record.relevanceLevel())
|
||||||
|
.dedupReason(record.dedupReason())
|
||||||
|
.durationMs(record.durationMs())
|
||||||
|
.success(record.success())
|
||||||
|
.errorMessage(record.errorMessage())
|
||||||
|
.build();
|
||||||
|
save(invocation);
|
||||||
|
}
|
||||||
|
|
||||||
|
private String normalizeEvidenceStatus(boolean success, String evidenceStatus) {
|
||||||
|
if (evidenceStatus != null && !evidenceStatus.isBlank()) {
|
||||||
|
return evidenceStatus;
|
||||||
|
}
|
||||||
|
return success ? EVIDENCE_STATUS_SUPPORTED : EVIDENCE_STATUS_FAILED;
|
||||||
|
}
|
||||||
|
|
||||||
private String preview(String output) {
|
private String preview(String output) {
|
||||||
if (output == null) {
|
if (output == null) {
|
||||||
return null;
|
return null;
|
||||||
@@ -90,4 +177,105 @@ public class ToolInvocationRecorder {
|
|||||||
return "{}";
|
return "{}";
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Builder
|
||||||
|
public record LookupKnowledgeRecord(
|
||||||
|
String query,
|
||||||
|
String outputPreview,
|
||||||
|
Integer outputLength,
|
||||||
|
String retrievalLayer,
|
||||||
|
Integer l0MatchCount,
|
||||||
|
Integer l1MatchCount,
|
||||||
|
Boolean truncated,
|
||||||
|
String relevanceLevel,
|
||||||
|
String completenessHint,
|
||||||
|
String domain,
|
||||||
|
String dedupReason,
|
||||||
|
Integer durationMs,
|
||||||
|
boolean success,
|
||||||
|
String evidenceStatus,
|
||||||
|
String errorMessage,
|
||||||
|
List<String> l0Titles,
|
||||||
|
Double l1TopScore,
|
||||||
|
Double l1TopSimilarity,
|
||||||
|
List<Double> l1Scores
|
||||||
|
) {
|
||||||
|
public static LookupKnowledgeRecord from(String query,
|
||||||
|
List<KnowledgeEntry> l0Matches,
|
||||||
|
List<VectorSearchService.SearchResult> l1Results,
|
||||||
|
boolean highConfidence,
|
||||||
|
LookupResult result,
|
||||||
|
String domain,
|
||||||
|
String dedupReason,
|
||||||
|
int durationMs,
|
||||||
|
double l1TopSimilarity) {
|
||||||
|
boolean hasL0 = l0Matches != null && !l0Matches.isEmpty();
|
||||||
|
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||||||
|
String layer;
|
||||||
|
if (hasL0 && !highConfidence) {
|
||||||
|
layer = "L0+L1";
|
||||||
|
} else if (hasL0) {
|
||||||
|
layer = "L0";
|
||||||
|
} else if (hasL1) {
|
||||||
|
layer = "L1";
|
||||||
|
} else {
|
||||||
|
layer = null;
|
||||||
|
}
|
||||||
|
|
||||||
|
String outputPreview = null;
|
||||||
|
int outputLength = 0;
|
||||||
|
boolean truncated = false;
|
||||||
|
if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) {
|
||||||
|
outputPreview = result.getPrimary().getContent();
|
||||||
|
outputLength = outputPreview.length();
|
||||||
|
truncated = outputLength > OUTPUT_PREVIEW_LIMIT;
|
||||||
|
} else if (hasL1 && l1Results.get(0).getContent() != null) {
|
||||||
|
outputPreview = l1Results.get(0).getContent();
|
||||||
|
outputLength = outputPreview.length();
|
||||||
|
truncated = outputLength > OUTPUT_PREVIEW_LIMIT;
|
||||||
|
}
|
||||||
|
|
||||||
|
String evidenceStatus = EVIDENCE_STATUS_SUPPORTED;
|
||||||
|
if (dedupReason != null) {
|
||||||
|
evidenceStatus = EVIDENCE_STATUS_DEDUPED;
|
||||||
|
} else if (result == null || !result.isFound()) {
|
||||||
|
evidenceStatus = EVIDENCE_STATUS_NO_EVIDENCE;
|
||||||
|
}
|
||||||
|
|
||||||
|
List<String> l0Titles = new ArrayList<>();
|
||||||
|
if (hasL0) {
|
||||||
|
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
|
||||||
|
l0Titles.add(l0Matches.get(i).getTitle());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
List<Double> l1Scores = new ArrayList<>();
|
||||||
|
if (hasL1) {
|
||||||
|
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
|
||||||
|
l1Scores.add((double) l1Results.get(i).getScore());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return LookupKnowledgeRecord.builder()
|
||||||
|
.query(query)
|
||||||
|
.outputPreview(outputPreview)
|
||||||
|
.outputLength(outputLength)
|
||||||
|
.retrievalLayer(layer)
|
||||||
|
.l0MatchCount(hasL0 ? l0Matches.size() : null)
|
||||||
|
.l1MatchCount(hasL1 ? l1Results.size() : null)
|
||||||
|
.truncated(truncated)
|
||||||
|
.relevanceLevel(result != null ? result.getRelevanceLevel() : null)
|
||||||
|
.completenessHint(result != null ? result.getCompletenessHint() : null)
|
||||||
|
.domain(domain)
|
||||||
|
.dedupReason(dedupReason)
|
||||||
|
.durationMs(durationMs)
|
||||||
|
.success(true)
|
||||||
|
.evidenceStatus(evidenceStatus)
|
||||||
|
.l0Titles(l0Titles)
|
||||||
|
.l1TopScore(hasL1 ? (double) l1Results.get(0).getScore() : null)
|
||||||
|
.l1TopSimilarity(hasL1 ? l1TopSimilarity : null)
|
||||||
|
.l1Scores(l1Scores)
|
||||||
|
.build();
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -107,6 +107,7 @@ public class ToolTraceSummaryService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private String extractOutputSummary(ToolInvocation invocation, String topicDomain) {
|
private String extractOutputSummary(ToolInvocation invocation, String topicDomain) {
|
||||||
|
String evidenceStatus = extractEvidenceStatus(invocation);
|
||||||
if (!Boolean.TRUE.equals(invocation.getSuccess())) {
|
if (!Boolean.TRUE.equals(invocation.getSuccess())) {
|
||||||
if (invocation.getErrorMessage() != null && !invocation.getErrorMessage().isBlank()) {
|
if (invocation.getErrorMessage() != null && !invocation.getErrorMessage().isBlank()) {
|
||||||
return "call failed: " + truncate(invocation.getErrorMessage(), 120);
|
return "call failed: " + truncate(invocation.getErrorMessage(), 120);
|
||||||
@@ -114,6 +115,17 @@ public class ToolTraceSummaryService {
|
|||||||
return "no usable evidence returned";
|
return "no usable evidence returned";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) {
|
||||||
|
return "retrieval skipped because the same document was already used in this session";
|
||||||
|
}
|
||||||
|
|
||||||
|
if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus)) {
|
||||||
|
if (invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank()) {
|
||||||
|
return "completed without usable evidence: " + truncate(invocation.getOutputPreview(), 120);
|
||||||
|
}
|
||||||
|
return "completed without usable evidence";
|
||||||
|
}
|
||||||
|
|
||||||
if ("lookup_knowledge".equals(invocation.getToolName())) {
|
if ("lookup_knowledge".equals(invocation.getToolName())) {
|
||||||
String relevance = invocation.getRelevanceLevel() != null ? invocation.getRelevanceLevel() : "UNKNOWN";
|
String relevance = invocation.getRelevanceLevel() != null ? invocation.getRelevanceLevel() : "UNKNOWN";
|
||||||
String preview = invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank()
|
String preview = invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank()
|
||||||
@@ -129,18 +141,46 @@ public class ToolTraceSummaryService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private String determineEvidenceLevel(ToolInvocation invocation) {
|
private String determineEvidenceLevel(ToolInvocation invocation) {
|
||||||
|
String evidenceStatus = extractEvidenceStatus(invocation);
|
||||||
if (!Boolean.TRUE.equals(invocation.getSuccess())) {
|
if (!Boolean.TRUE.equals(invocation.getSuccess())) {
|
||||||
return "none";
|
return "none";
|
||||||
}
|
}
|
||||||
|
if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus)
|
||||||
|
|| ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) {
|
||||||
|
return "none";
|
||||||
|
}
|
||||||
if ("PRECISE".equals(invocation.getRelevanceLevel()) || "HIGHLY_RELEVANT".equals(invocation.getRelevanceLevel())) {
|
if ("PRECISE".equals(invocation.getRelevanceLevel()) || "HIGHLY_RELEVANT".equals(invocation.getRelevanceLevel())) {
|
||||||
return "direct";
|
return "direct";
|
||||||
}
|
}
|
||||||
if ("REFERENCE".equals(invocation.getRelevanceLevel())) {
|
if ("REFERENCE".equals(invocation.getRelevanceLevel())) {
|
||||||
return "indirect";
|
return "indirect";
|
||||||
}
|
}
|
||||||
|
if (EVIDENCE_TOOLS.contains(invocation.getToolName())) {
|
||||||
|
return "direct";
|
||||||
|
}
|
||||||
return "none";
|
return "none";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private String extractEvidenceStatus(ToolInvocation invocation) {
|
||||||
|
if (invocation.getRetrievalDetails() == null || invocation.getRetrievalDetails().isBlank()) {
|
||||||
|
return Boolean.TRUE.equals(invocation.getSuccess())
|
||||||
|
? ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED
|
||||||
|
: ToolInvocationRecorder.EVIDENCE_STATUS_FAILED;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
Map<String, Object> details = objectMapper.readValue(invocation.getRetrievalDetails(), MAP_TYPE);
|
||||||
|
Object evidenceStatus = details.get("evidence_status");
|
||||||
|
if (evidenceStatus != null) {
|
||||||
|
return String.valueOf(evidenceStatus);
|
||||||
|
}
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.debug("Failed to parse evidence_status", e);
|
||||||
|
}
|
||||||
|
return Boolean.TRUE.equals(invocation.getSuccess())
|
||||||
|
? ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED
|
||||||
|
: ToolInvocationRecorder.EVIDENCE_STATUS_FAILED;
|
||||||
|
}
|
||||||
|
|
||||||
private List<String> extractStringList(Object value) {
|
private List<String> extractStringList(Object value) {
|
||||||
if (!(value instanceof List<?> list) || list.isEmpty()) {
|
if (!(value instanceof List<?> list) || list.isEmpty()) {
|
||||||
return List.of();
|
return List.of();
|
||||||
@@ -224,13 +264,22 @@ public class ToolTraceSummaryService {
|
|||||||
inputSummary = extractInputSummary(invocation);
|
inputSummary = extractInputSummary(invocation);
|
||||||
}
|
}
|
||||||
|
|
||||||
boolean invocationSuccess = Boolean.TRUE.equals(invocation.getSuccess());
|
String evidenceStatus = extractEvidenceStatus(invocation);
|
||||||
if (!invocationSuccess) {
|
if (!Boolean.TRUE.equals(invocation.getSuccess())) {
|
||||||
failedCount++;
|
failedCount++;
|
||||||
|
if (outputSummary == null || outputSummary.isBlank()) {
|
||||||
|
outputSummary = extractOutputSummary(invocation, topicDomain);
|
||||||
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (invocation.getDedupReason() != null) {
|
|
||||||
|
if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus)
|
||||||
|
|| ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) {
|
||||||
noHitCount++;
|
noHitCount++;
|
||||||
|
if (outputSummary == null || outputSummary.isBlank()) {
|
||||||
|
outputSummary = extractOutputSummary(invocation, topicDomain);
|
||||||
|
}
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
String invocationEvidenceLevel = determineEvidenceLevel(invocation);
|
String invocationEvidenceLevel = determineEvidenceLevel(invocation);
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ import org.springframework.beans.factory.annotation.Value;
|
|||||||
import org.springframework.stereotype.Component;
|
import org.springframework.stereotype.Component;
|
||||||
|
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
import java.util.Locale;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -309,131 +310,30 @@ public class LookupKnowledgeTool {
|
|||||||
String sessionId = SessionContextHolder.getSessionId();
|
String sessionId = SessionContextHolder.getSessionId();
|
||||||
if (sessionId == null) return;
|
if (sessionId == null) return;
|
||||||
|
|
||||||
boolean hasL0 = l0Matches != null && !l0Matches.isEmpty();
|
|
||||||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
|
||||||
long duration = System.currentTimeMillis() - startTime;
|
long duration = System.currentTimeMillis() - startTime;
|
||||||
|
double l1TopSimilarity = (l1Results != null && !l1Results.isEmpty())
|
||||||
|
? normalizeL2(l1Results.get(0).getScore())
|
||||||
|
: -1;
|
||||||
|
|
||||||
String layer;
|
ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.from(
|
||||||
String outputPreview = null;
|
query,
|
||||||
int outputLength = 0;
|
l0Matches,
|
||||||
int l0Count = 0;
|
l1Results,
|
||||||
int l1Count = 0;
|
highConfidence,
|
||||||
boolean truncated = false;
|
result,
|
||||||
|
domain,
|
||||||
if (hasL0 && !highConfidence) {
|
dedupReason,
|
||||||
layer = "L0+L1";
|
(int) duration,
|
||||||
l0Count = l0Matches.size();
|
l1TopSimilarity
|
||||||
l1Count = l1Results.size();
|
);
|
||||||
} else if (hasL0) {
|
toolInvocationRecorder.recordLookupKnowledge(record);
|
||||||
layer = "L0";
|
|
||||||
l0Count = l0Matches.size();
|
|
||||||
} else if (hasL1) {
|
|
||||||
layer = "L1";
|
|
||||||
l1Count = l1Results.size();
|
|
||||||
} else {
|
|
||||||
layer = null;
|
|
||||||
}
|
|
||||||
|
|
||||||
// output_preview
|
|
||||||
if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) {
|
|
||||||
String content = result.getPrimary().getContent();
|
|
||||||
outputLength = content.length();
|
|
||||||
if (content.length() > 500) {
|
|
||||||
outputPreview = content.substring(0, 500) + "...";
|
|
||||||
truncated = true;
|
|
||||||
} else {
|
|
||||||
outputPreview = content;
|
|
||||||
}
|
|
||||||
} else if (l1Results != null && !l1Results.isEmpty() && l1Results.get(0).getContent() != null) {
|
|
||||||
String content = l1Results.get(0).getContent();
|
|
||||||
outputLength = content.length();
|
|
||||||
if (content.length() > 500) {
|
|
||||||
outputPreview = content.substring(0, 500) + "...";
|
|
||||||
truncated = true;
|
|
||||||
} else {
|
|
||||||
outputPreview = content;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// L1 top score + similarity
|
|
||||||
float l1TopScore = (hasL1) ? l1Results.get(0).getScore() : -1;
|
|
||||||
double l1TopSimilarity = (hasL1) ? normalizeL2(l1TopScore) : -1;
|
|
||||||
|
|
||||||
// 构建检索明细 JSON(扩展版)
|
|
||||||
StringBuilder details = new StringBuilder("{");
|
|
||||||
if (hasL0) {
|
|
||||||
details.append("\"l0_match_count\":").append(l0Count).append(",");
|
|
||||||
details.append("\"l0_titles\":[");
|
|
||||||
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
|
|
||||||
if (i > 0) details.append(",");
|
|
||||||
details.append("\"").append(escapeJson(l0Matches.get(i).getTitle())).append("\"");
|
|
||||||
}
|
|
||||||
details.append("],");
|
|
||||||
}
|
|
||||||
if (hasL1) {
|
|
||||||
details.append("\"l1_top_score\":").append(String.format("%.4f", l1TopScore)).append(",");
|
|
||||||
details.append("\"l1_top_similarity\":").append(String.format("%.4f", l1TopSimilarity)).append(",");
|
|
||||||
details.append("\"l1_match_count\":").append(l1Count).append(",");
|
|
||||||
details.append("\"l1_scores\":[");
|
|
||||||
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
|
|
||||||
if (i > 0) details.append(",");
|
|
||||||
details.append(String.format("%.4f", l1Results.get(i).getScore()));
|
|
||||||
}
|
|
||||||
details.append("],");
|
|
||||||
}
|
|
||||||
// 归一化信息
|
|
||||||
if (result != null && result.getRelevanceLevel() != null) {
|
|
||||||
details.append("\"relevance_level\":\"").append(result.getRelevanceLevel()).append("\",");
|
|
||||||
details.append("\"completeness_hint\":\"").append(escapeJson(result.getCompletenessHint())).append("\",");
|
|
||||||
}
|
|
||||||
// 域信息
|
|
||||||
if (domain != null) {
|
|
||||||
details.append("\"retrieved_domains\":[\"").append(escapeJson(domain)).append("\"],");
|
|
||||||
}
|
|
||||||
// 去重原因
|
|
||||||
if (dedupReason != null) {
|
|
||||||
details.append("\"dedup_reason\":\"").append(dedupReason).append("\",");
|
|
||||||
}
|
|
||||||
// 移除末尾逗号
|
|
||||||
if (details.charAt(details.length() - 1) == ',') {
|
|
||||||
details.setLength(details.length() - 1);
|
|
||||||
}
|
|
||||||
details.append("}");
|
|
||||||
|
|
||||||
ToolInvocation inv = ToolInvocation.builder()
|
|
||||||
.sessionId(sessionId)
|
|
||||||
.toolName("lookup_knowledge")
|
|
||||||
.inputParams("{\"query\":\"" + escapeJson(query) + "\"}")
|
|
||||||
.outputPreview(outputPreview)
|
|
||||||
.outputLength(outputLength)
|
|
||||||
.retrievalLayer(layer)
|
|
||||||
.l0MatchCount(hasL0 ? l0Count : null)
|
|
||||||
.l1MatchCount(hasL1 ? l1Count : null)
|
|
||||||
.isTruncated(truncated)
|
|
||||||
.retrievalDetails(details.toString())
|
|
||||||
.relevanceLevel(result != null ? result.getRelevanceLevel() : null)
|
|
||||||
.dedupReason(dedupReason)
|
|
||||||
.durationMs((int) duration)
|
|
||||||
.success(true)
|
|
||||||
.build();
|
|
||||||
|
|
||||||
toolInvocationRecorder.save(inv);
|
|
||||||
log.debug("tool_invocation 已保存: sessionId={}, layer={}, relevanceLevel={}, duration={}ms",
|
log.debug("tool_invocation 已保存: sessionId={}, layer={}, relevanceLevel={}, duration={}ms",
|
||||||
sessionId, layer, result != null ? result.getRelevanceLevel() : null, duration);
|
sessionId, record.retrievalLayer(), record.relevanceLevel(), duration);
|
||||||
} catch (Exception e) {
|
} catch (Exception e) {
|
||||||
log.error("保存 tool_invocation 失败", e);
|
log.error("保存 tool_invocation 失败", e);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private String escapeJson(String s) {
|
|
||||||
if (s == null) return "";
|
|
||||||
return s.replace("\\", "\\\\")
|
|
||||||
.replace("\"", "\\\"")
|
|
||||||
.replace("\n", "\\n")
|
|
||||||
.replace("\r", "\\r")
|
|
||||||
.replace("\t", "\\t");
|
|
||||||
}
|
|
||||||
|
|
||||||
// ==================== 结果组装 ====================
|
// ==================== 结果组装 ====================
|
||||||
|
|
||||||
private LookupResult buildResult(
|
private LookupResult buildResult(
|
||||||
|
|||||||
@@ -0,0 +1,140 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
class DiagnosisEvalBaselineDiffTest {
|
||||||
|
|
||||||
|
private final ObjectMapper objectMapper = new ObjectMapper();
|
||||||
|
private final DiagnosisEvalBaselineDiffer differ = new DiagnosisEvalBaselineDiffer();
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void compareReportsDetectsAggregateAndCaseRegressions() throws Exception {
|
||||||
|
DiagnosisEvalReport baseline = readBaselineReport();
|
||||||
|
DiagnosisEvalReport current = readBaselineReport();
|
||||||
|
degradeRedisCase(current);
|
||||||
|
|
||||||
|
DiagnosisEvalDiffReport diff = differ.compare(baseline, current);
|
||||||
|
|
||||||
|
assertTrue(diff.isHasRegression());
|
||||||
|
assertEquals(6, diff.getRegressionCount());
|
||||||
|
assertEquals(2, diff.getChangedCount());
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "aggregate", null, "passRate"));
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "aggregate", null, "averageToolCallCount"));
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "case", "redis-timeout", "passed"));
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "case", "redis-timeout", "verdict"));
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "case", "redis-timeout", "matchedKeywordCount"));
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "case", "redis-timeout", "evidenceCoverage.query_logs"));
|
||||||
|
assertTrue(hasItem(diff, "CHANGED", "aggregate", null, "verdictDistribution.LOW_CONFID"));
|
||||||
|
assertTrue(hasItem(diff, "CHANGED", "aggregate", null, "verdictDistribution.REJECT"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void compareReportsDetectsMissingAndNewCases() throws Exception {
|
||||||
|
DiagnosisEvalReport baseline = readBaselineReport();
|
||||||
|
DiagnosisEvalReport current = readBaselineReport();
|
||||||
|
DiagnosisEvalResult removed = current.getResults().remove(0);
|
||||||
|
current.getResults().add(DiagnosisEvalResult.builder()
|
||||||
|
.caseId("new-case")
|
||||||
|
.title("New case")
|
||||||
|
.passed(true)
|
||||||
|
.failedChecks(List.of())
|
||||||
|
.verdict("PASS")
|
||||||
|
.matchedKeywordCount(1)
|
||||||
|
.requiredKeywordCount(1)
|
||||||
|
.evidenceCoverage(new LinkedHashMap<>())
|
||||||
|
.toolCallCount(1)
|
||||||
|
.durationMs(1000)
|
||||||
|
.build());
|
||||||
|
|
||||||
|
DiagnosisEvalDiffReport diff = differ.compare(baseline, current);
|
||||||
|
|
||||||
|
assertTrue(hasItem(diff, "REGRESSION", "case", removed.getCaseId(), "casePresence"));
|
||||||
|
assertTrue(hasItem(diff, "CHANGED", "case", "new-case", "casePresence"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void compareSameReportHasNoDiff() throws Exception {
|
||||||
|
DiagnosisEvalReport baseline = readBaselineReport();
|
||||||
|
|
||||||
|
DiagnosisEvalDiffReport diff = differ.compare(baseline, readBaselineReport());
|
||||||
|
|
||||||
|
assertFalse(diff.isHasRegression());
|
||||||
|
assertEquals(0, diff.getRegressionCount());
|
||||||
|
assertTrue(diff.getItems().isEmpty());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void writerOutputsJsonAndMarkdown(@TempDir Path tempDir) throws Exception {
|
||||||
|
DiagnosisEvalReport baseline = readBaselineReport();
|
||||||
|
DiagnosisEvalReport current = readBaselineReport();
|
||||||
|
degradeRedisCase(current);
|
||||||
|
DiagnosisEvalDiffReport diff = differ.compare(baseline, current);
|
||||||
|
DiagnosisEvalDiffReportWriter writer = new DiagnosisEvalDiffReportWriter(objectMapper);
|
||||||
|
|
||||||
|
Path json = tempDir.resolve("baseline-diff.json");
|
||||||
|
Path markdown = tempDir.resolve("baseline-diff.md");
|
||||||
|
writer.writeJson(diff, json);
|
||||||
|
writer.writeMarkdown(diff, markdown);
|
||||||
|
|
||||||
|
assertTrue(Files.exists(json));
|
||||||
|
assertTrue(Files.readString(json).contains("\"hasRegression\" : true"));
|
||||||
|
assertTrue(Files.readString(markdown).contains("# Diagnosis Eval Baseline Diff"));
|
||||||
|
assertTrue(Files.readString(markdown).contains("redis-timeout"));
|
||||||
|
}
|
||||||
|
|
||||||
|
private DiagnosisEvalReport readBaselineReport() throws Exception {
|
||||||
|
return objectMapper.readValue(Path.of("mvp/eval/reports/baseline-report.json").toFile(),
|
||||||
|
DiagnosisEvalReport.class);
|
||||||
|
}
|
||||||
|
|
||||||
|
private void degradeRedisCase(DiagnosisEvalReport report) {
|
||||||
|
report.setPassedCases(4);
|
||||||
|
report.setPassRate(0.8);
|
||||||
|
report.setAverageToolCallCount(3.0);
|
||||||
|
report.setAverageDurationMs(45800.0);
|
||||||
|
report.setVerdictDistribution(new LinkedHashMap<>());
|
||||||
|
report.getVerdictDistribution().put("PASS", 2L);
|
||||||
|
report.getVerdictDistribution().put("LOW_CONFID", 2L);
|
||||||
|
report.getVerdictDistribution().put("REJECT", 1L);
|
||||||
|
|
||||||
|
DiagnosisEvalResult redis = result(report, "redis-timeout");
|
||||||
|
redis.setPassed(false);
|
||||||
|
redis.setFailedChecks(new ArrayList<>(List.of("missing required evidence tool: query_logs")));
|
||||||
|
redis.setVerdict("REJECT");
|
||||||
|
redis.setMatchedKeywordCount(1);
|
||||||
|
redis.getEvidenceCoverage().put("query_logs", false);
|
||||||
|
redis.setToolCallCount(1);
|
||||||
|
redis.setDurationMs(36000);
|
||||||
|
}
|
||||||
|
|
||||||
|
private DiagnosisEvalResult result(DiagnosisEvalReport report, String caseId) {
|
||||||
|
return report.getResults().stream()
|
||||||
|
.filter(item -> caseId.equals(item.getCaseId()))
|
||||||
|
.findFirst()
|
||||||
|
.orElseThrow();
|
||||||
|
}
|
||||||
|
|
||||||
|
private boolean hasItem(DiagnosisEvalDiffReport diff,
|
||||||
|
String type,
|
||||||
|
String scope,
|
||||||
|
String caseId,
|
||||||
|
String metric) {
|
||||||
|
return diff.getItems().stream().anyMatch(item ->
|
||||||
|
type.equals(item.getType())
|
||||||
|
&& scope.equals(item.getScope())
|
||||||
|
&& java.util.Objects.equals(caseId, item.getCaseId())
|
||||||
|
&& metric.equals(item.getMetric()));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
package com.superbiz.agent.eval;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import com.superbiz.agent.dto.DiagnosisTraceResponse;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
class DiagnosisTraceEvaluatorTest {
|
||||||
|
|
||||||
|
private final ObjectMapper objectMapper = new ObjectMapper();
|
||||||
|
private final DiagnosisTraceEvaluator evaluator = new DiagnosisTraceEvaluator(objectMapper);
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void evaluateFixtureReportsFullBaseline() {
|
||||||
|
List<DiagnosisEvalCase> cases = readCases();
|
||||||
|
|
||||||
|
DiagnosisEvalReport report = evaluator.evaluate(cases, Path.of("mvp/eval/fixtures"));
|
||||||
|
|
||||||
|
assertEquals(5, report.getTotalCases());
|
||||||
|
assertEquals(5, report.getPassedCases());
|
||||||
|
assertEquals(1.0, report.getPassRate(), 0.001);
|
||||||
|
assertEquals(2L, report.getVerdictDistribution().get("PASS"));
|
||||||
|
assertEquals(3L, report.getVerdictDistribution().get("LOW_CONFID"));
|
||||||
|
|
||||||
|
DiagnosisEvalResult payment = result(report, "payment-timeout");
|
||||||
|
assertTrue(payment.isPassed());
|
||||||
|
assertTrue(payment.getEvidenceCoverage().get("lookup_knowledge"));
|
||||||
|
assertTrue(payment.getEvidenceCoverage().get("query_logs"));
|
||||||
|
assertTrue(payment.getEvidenceCoverage().get("query_metrics"));
|
||||||
|
|
||||||
|
DiagnosisEvalResult redis = result(report, "redis-timeout");
|
||||||
|
assertTrue(redis.isPassed());
|
||||||
|
assertTrue(redis.getEvidenceCoverage().get("query_logs"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void everyFixedCaseReferencesExistingFixture() {
|
||||||
|
for (DiagnosisEvalCase evalCase : readCases()) {
|
||||||
|
Path fixture = Path.of("mvp/eval/fixtures").resolve(evalCase.getTraceFixture());
|
||||||
|
assertTrue(Files.exists(fixture), "missing fixture: " + fixture);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void evaluateRejectRequiresDegradedOutput() {
|
||||||
|
DiagnosisEvalCase evalCase = DiagnosisEvalCase.builder()
|
||||||
|
.id("reject-case")
|
||||||
|
.title("Reject case")
|
||||||
|
.expectedRootCauseKeywords(List.of())
|
||||||
|
.requiredEvidenceTools(List.of())
|
||||||
|
.allowedVerdicts(List.of("REJECT"))
|
||||||
|
.build();
|
||||||
|
DiagnosisTraceResponse trace = DiagnosisTraceResponse.builder()
|
||||||
|
.session(DiagnosisTraceResponse.SessionTrace.builder()
|
||||||
|
.answer("EXECUTOR_FINAL_ANSWER")
|
||||||
|
.selfEvaluation(java.util.Map.of(
|
||||||
|
"verifier_evaluation", java.util.Map.of("verdict", "REJECT")))
|
||||||
|
.build())
|
||||||
|
.toolInvocations(List.of())
|
||||||
|
.build();
|
||||||
|
|
||||||
|
DiagnosisEvalResult result = evaluator.evaluate(evalCase, trace);
|
||||||
|
|
||||||
|
assertFalse(result.isPassed());
|
||||||
|
assertTrue(result.getFailedChecks().contains("reject output does not use degraded template"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void reportWriterOutputsJsonAndMarkdown(@TempDir Path tempDir) throws Exception {
|
||||||
|
DiagnosisEvalReport report = evaluator.evaluate(readCases(), Path.of("mvp/eval/fixtures"));
|
||||||
|
DiagnosisEvalReportWriter writer = new DiagnosisEvalReportWriter(objectMapper);
|
||||||
|
|
||||||
|
Path json = tempDir.resolve("eval-report.json");
|
||||||
|
Path markdown = tempDir.resolve("eval-report.md");
|
||||||
|
writer.writeJson(report, json);
|
||||||
|
writer.writeMarkdown(report, markdown);
|
||||||
|
|
||||||
|
assertTrue(Files.exists(json));
|
||||||
|
assertTrue(Files.readString(markdown).contains("# Diagnosis Eval Report"));
|
||||||
|
assertTrue(Files.readString(markdown).contains("payment-timeout"));
|
||||||
|
assertEquals(
|
||||||
|
comparableReportText(Files.readString(Path.of("mvp/eval/reports/baseline-report.json"))),
|
||||||
|
comparableReportText(Files.readString(json)));
|
||||||
|
assertEquals(
|
||||||
|
comparableReportText(Files.readString(Path.of("mvp/eval/reports/baseline-report.md"))),
|
||||||
|
comparableReportText(Files.readString(markdown)));
|
||||||
|
}
|
||||||
|
|
||||||
|
private List<DiagnosisEvalCase> readCases() {
|
||||||
|
try {
|
||||||
|
return evaluator.loadCases(Path.of("mvp/eval/cases/diagnosis-cases.json"));
|
||||||
|
} catch (Exception e) {
|
||||||
|
throw new AssertionError(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private DiagnosisEvalResult result(DiagnosisEvalReport report, String caseId) {
|
||||||
|
return report.getResults().stream()
|
||||||
|
.filter(item -> caseId.equals(item.getCaseId()))
|
||||||
|
.findFirst()
|
||||||
|
.orElseThrow();
|
||||||
|
}
|
||||||
|
|
||||||
|
private String comparableReportText(String value) {
|
||||||
|
return value.replace("\r\n", "\n").stripTrailing();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -25,6 +25,7 @@ import java.util.Optional;
|
|||||||
import java.util.concurrent.atomic.AtomicInteger;
|
import java.util.concurrent.atomic.AtomicInteger;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
import static org.mockito.ArgumentMatchers.any;
|
import static org.mockito.ArgumentMatchers.any;
|
||||||
@@ -86,6 +87,73 @@ class ChatServiceSequentialAgentTest {
|
|||||||
assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls);
|
assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void executeChatComplexFallsBackToLowConfidenceWhenVerifierOutputMissing() throws Exception {
|
||||||
|
ChatService chatService = createChatService();
|
||||||
|
ScriptedChatModel chatModel = new ScriptedChatModel("", "");
|
||||||
|
|
||||||
|
ChatService.ChatResult result = chatService.executeChatComplex(
|
||||||
|
chatModel,
|
||||||
|
new ToolCallback[0],
|
||||||
|
"请分析订单支付超时的原因,并给出修复建议",
|
||||||
|
List.of(),
|
||||||
|
"sequential-missing-verifier-session"
|
||||||
|
);
|
||||||
|
|
||||||
|
assertTrue(result.answer().startsWith("以下结论基于当前已获取证据"));
|
||||||
|
assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier", "chat_verifier"), chatModel.agentCalls);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void executeChatComplexFallsBackToLowConfidenceWhenVerifierJsonInvalid() throws Exception {
|
||||||
|
ChatService chatService = createChatService();
|
||||||
|
ScriptedChatModel chatModel = new ScriptedChatModel("not-json");
|
||||||
|
|
||||||
|
ChatService.ChatResult result = chatService.executeChatComplex(
|
||||||
|
chatModel,
|
||||||
|
new ToolCallback[0],
|
||||||
|
"请分析订单支付超时的原因,并给出修复建议",
|
||||||
|
List.of(),
|
||||||
|
"sequential-invalid-verifier-session"
|
||||||
|
);
|
||||||
|
|
||||||
|
assertTrue(result.answer().startsWith("以下结论基于当前已获取证据"));
|
||||||
|
assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void executeChatComplexRejectOutputDoesNotLeakExecutorAnswer() throws Exception {
|
||||||
|
ChatService chatService = createChatService();
|
||||||
|
ScriptedChatModel chatModel = new ScriptedChatModel("""
|
||||||
|
{
|
||||||
|
"verdict": "REJECT",
|
||||||
|
"groundedness_score": 0.0,
|
||||||
|
"critical_fact_count": 1,
|
||||||
|
"facts_checked": [
|
||||||
|
{
|
||||||
|
"fact": "payment timeout root cause",
|
||||||
|
"is_critical": true,
|
||||||
|
"verification": "contradicted",
|
||||||
|
"detail": "scripted contradiction",
|
||||||
|
"evidence_refs": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"rationale": "scripted reject"
|
||||||
|
}
|
||||||
|
""");
|
||||||
|
|
||||||
|
ChatService.ChatResult result = chatService.executeChatComplex(
|
||||||
|
chatModel,
|
||||||
|
new ToolCallback[0],
|
||||||
|
"请分析订单支付超时的原因,并给出修复建议",
|
||||||
|
List.of(),
|
||||||
|
"sequential-reject-session"
|
||||||
|
);
|
||||||
|
|
||||||
|
assertTrue(result.answer().startsWith("当前无法基于已获取证据生成可靠结论"));
|
||||||
|
assertFalse(result.answer().contains("EXECUTOR_FINAL_ANSWER"));
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void executeChatComplexRunsPlannerExecutorVerifierInFixedOrder() throws Exception {
|
void executeChatComplexRunsPlannerExecutorVerifierInFixedOrder() throws Exception {
|
||||||
ChatService chatService = createChatService();
|
ChatService chatService = createChatService();
|
||||||
@@ -178,7 +246,8 @@ class ChatServiceSequentialAgentTest {
|
|||||||
private final java.util.ArrayList<String> agentCalls = new java.util.ArrayList<>();
|
private final java.util.ArrayList<String> agentCalls = new java.util.ArrayList<>();
|
||||||
private String promptText = "";
|
private String promptText = "";
|
||||||
private boolean sawVerifierPrompt;
|
private boolean sawVerifierPrompt;
|
||||||
private final String verifierOutput;
|
private final java.util.List<String> verifierOutputs;
|
||||||
|
private int verifierOutputIndex;
|
||||||
|
|
||||||
private ScriptedChatModel() {
|
private ScriptedChatModel() {
|
||||||
this("""
|
this("""
|
||||||
@@ -201,7 +270,11 @@ class ChatServiceSequentialAgentTest {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private ScriptedChatModel(String verifierOutput) {
|
private ScriptedChatModel(String verifierOutput) {
|
||||||
this.verifierOutput = verifierOutput;
|
this.verifierOutputs = java.util.List.of(verifierOutput);
|
||||||
|
}
|
||||||
|
|
||||||
|
private ScriptedChatModel(String... verifierOutputs) {
|
||||||
|
this.verifierOutputs = java.util.List.of(verifierOutputs);
|
||||||
}
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
@@ -217,7 +290,9 @@ class ChatServiceSequentialAgentTest {
|
|||||||
} else if (promptText.contains("VERIFIER_TEST_PROMPT")) {
|
} else if (promptText.contains("VERIFIER_TEST_PROMPT")) {
|
||||||
agentCalls.add("chat_verifier");
|
agentCalls.add("chat_verifier");
|
||||||
sawVerifierPrompt = true;
|
sawVerifierPrompt = true;
|
||||||
text = verifierOutput;
|
int index = Math.min(verifierOutputIndex, verifierOutputs.size() - 1);
|
||||||
|
text = verifierOutputs.get(index);
|
||||||
|
verifierOutputIndex++;
|
||||||
} else {
|
} else {
|
||||||
text = "UNEXPECTED_PROMPT";
|
text = "UNEXPECTED_PROMPT";
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,96 @@
|
|||||||
|
package com.superbiz.agent.service;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||||
|
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||||||
|
import com.superbiz.agent.util.SessionContextHolder;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.mockito.ArgumentCaptor;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
import static org.mockito.ArgumentMatchers.any;
|
||||||
|
import static org.mockito.Mockito.mock;
|
||||||
|
import static org.mockito.Mockito.verify;
|
||||||
|
import static org.mockito.Mockito.when;
|
||||||
|
|
||||||
|
class ToolInvocationRecorderTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void recordEvidenceToolPreservesNoEvidenceSemantics() {
|
||||||
|
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
||||||
|
when(repository.save(any(ToolInvocation.class))).thenAnswer(invocation -> invocation.getArgument(0));
|
||||||
|
ToolInvocationRecorder recorder = new ToolInvocationRecorder(repository, new ObjectMapper());
|
||||||
|
SessionContextHolder.setSessionId("recorder-test-session");
|
||||||
|
|
||||||
|
try {
|
||||||
|
recorder.recordEvidenceTool(
|
||||||
|
"query_logs",
|
||||||
|
Map.of("query", "timeout"),
|
||||||
|
"{\"success\":false,\"message\":\"未找到匹配的日志\"}",
|
||||||
|
true,
|
||||||
|
System.currentTimeMillis() - 10,
|
||||||
|
null,
|
||||||
|
"application-logs",
|
||||||
|
ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE,
|
||||||
|
Map.of("log_topic", "application-logs")
|
||||||
|
);
|
||||||
|
} finally {
|
||||||
|
SessionContextHolder.clear();
|
||||||
|
}
|
||||||
|
|
||||||
|
ArgumentCaptor<ToolInvocation> captor = ArgumentCaptor.forClass(ToolInvocation.class);
|
||||||
|
verify(repository).save(captor.capture());
|
||||||
|
ToolInvocation saved = captor.getValue();
|
||||||
|
|
||||||
|
assertEquals("query_logs", saved.getToolName());
|
||||||
|
assertEquals(Boolean.TRUE, saved.getSuccess());
|
||||||
|
assertTrue(saved.getRetrievalDetails().contains("\"evidence_status\":\"no_evidence\""));
|
||||||
|
assertTrue(saved.getRetrievalDetails().contains("\"retrieved_domains\":[\"application-logs\"]"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void recordLookupKnowledgePreservesRetrievalSpecificFields() {
|
||||||
|
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
||||||
|
when(repository.save(any(ToolInvocation.class))).thenAnswer(invocation -> invocation.getArgument(0));
|
||||||
|
ToolInvocationRecorder recorder = new ToolInvocationRecorder(repository, new ObjectMapper());
|
||||||
|
SessionContextHolder.setSessionId("lookup-recorder-session");
|
||||||
|
|
||||||
|
ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.builder()
|
||||||
|
.query("ERR_TIMEOUT")
|
||||||
|
.outputPreview("matched payment doc")
|
||||||
|
.outputLength(18)
|
||||||
|
.retrievalLayer("L0")
|
||||||
|
.l0MatchCount(1)
|
||||||
|
.l1MatchCount(null)
|
||||||
|
.truncated(false)
|
||||||
|
.relevanceLevel("PRECISE")
|
||||||
|
.completenessHint("already precise")
|
||||||
|
.domain("payment")
|
||||||
|
.dedupReason("doc_retrieved")
|
||||||
|
.durationMs(42)
|
||||||
|
.success(true)
|
||||||
|
.evidenceStatus(ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED)
|
||||||
|
.l0Titles(List.of("payment/errors.md"))
|
||||||
|
.build();
|
||||||
|
|
||||||
|
try {
|
||||||
|
recorder.recordLookupKnowledge(record);
|
||||||
|
} finally {
|
||||||
|
SessionContextHolder.clear();
|
||||||
|
}
|
||||||
|
|
||||||
|
ArgumentCaptor<ToolInvocation> captor = ArgumentCaptor.forClass(ToolInvocation.class);
|
||||||
|
verify(repository).save(captor.capture());
|
||||||
|
ToolInvocation saved = captor.getValue();
|
||||||
|
|
||||||
|
assertEquals("lookup_knowledge", saved.getToolName());
|
||||||
|
assertEquals("PRECISE", saved.getRelevanceLevel());
|
||||||
|
assertEquals("doc_retrieved", saved.getDedupReason());
|
||||||
|
assertTrue(saved.getRetrievalDetails().contains("\"evidence_status\":\"deduped\""));
|
||||||
|
assertTrue(saved.getRetrievalDetails().contains("\"retrieved_domains\":[\"payment\"]"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
package com.superbiz.agent.service;
|
||||||
|
|
||||||
|
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||||
|
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
import static org.mockito.Mockito.mock;
|
||||||
|
import static org.mockito.Mockito.when;
|
||||||
|
|
||||||
|
class ToolTraceSummaryServiceTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void buildVerifierTraceSummaryTreatsNoEvidenceAsGapWithoutLosingSuccessfulEvidence() {
|
||||||
|
ToolInvocationRepository repository = mock(ToolInvocationRepository.class);
|
||||||
|
when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of(
|
||||||
|
ToolInvocation.builder()
|
||||||
|
.id(1L)
|
||||||
|
.sessionId("session-1")
|
||||||
|
.toolName("query_logs")
|
||||||
|
.inputParams("{\"query\":\"timeout\"}")
|
||||||
|
.outputPreview("payment timeout stack trace")
|
||||||
|
.retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"supported\"}")
|
||||||
|
.success(true)
|
||||||
|
.build(),
|
||||||
|
ToolInvocation.builder()
|
||||||
|
.id(2L)
|
||||||
|
.sessionId("session-1")
|
||||||
|
.toolName("query_logs")
|
||||||
|
.inputParams("{\"query\":\"timeout\"}")
|
||||||
|
.outputPreview("{\"success\":false,\"message\":\"未找到匹配的日志\"}")
|
||||||
|
.retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"no_evidence\"}")
|
||||||
|
.success(true)
|
||||||
|
.build(),
|
||||||
|
ToolInvocation.builder()
|
||||||
|
.id(3L)
|
||||||
|
.sessionId("session-1")
|
||||||
|
.toolName("query_metrics")
|
||||||
|
.inputParams("{\"query\":\"active_prometheus_alerts\"}")
|
||||||
|
.errorMessage("prometheus timeout")
|
||||||
|
.retrievalDetails("{\"retrieved_domains\":[\"prometheus_alerts\"],\"evidence_status\":\"failed\"}")
|
||||||
|
.success(false)
|
||||||
|
.build()
|
||||||
|
));
|
||||||
|
|
||||||
|
ToolTraceSummaryService service = new ToolTraceSummaryService(repository);
|
||||||
|
|
||||||
|
List<Map<String, Object>> summaries = service.buildVerifierTraceSummary("session-1", "application-logs point to timeout");
|
||||||
|
|
||||||
|
assertEquals(2, summaries.size());
|
||||||
|
|
||||||
|
Map<String, Object> logsSummary = summaries.stream()
|
||||||
|
.filter(item -> "query_logs".equals(item.get("tool_name")))
|
||||||
|
.findFirst()
|
||||||
|
.orElseThrow();
|
||||||
|
assertEquals(Boolean.TRUE, logsSummary.get("success"));
|
||||||
|
assertEquals("direct", logsSummary.get("evidence_level"));
|
||||||
|
assertEquals(2, logsSummary.get("invocation_count"));
|
||||||
|
assertEquals(1, logsSummary.get("no_hit_invocation_count"));
|
||||||
|
|
||||||
|
Map<String, Object> metricsSummary = summaries.stream()
|
||||||
|
.filter(item -> "query_metrics".equals(item.get("tool_name")))
|
||||||
|
.findFirst()
|
||||||
|
.orElseThrow();
|
||||||
|
assertEquals(Boolean.FALSE, metricsSummary.get("success"));
|
||||||
|
assertEquals("none", metricsSummary.get("evidence_level"));
|
||||||
|
assertEquals(1, metricsSummary.get("failed_invocation_count"));
|
||||||
|
assertTrue(String.valueOf(metricsSummary.get("output_summary")).contains("call failed"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -3,7 +3,9 @@ package com.superbiz.agent.tool;
|
|||||||
import com.superbiz.agent.dto.KnowledgeEntry;
|
import com.superbiz.agent.dto.KnowledgeEntry;
|
||||||
import com.superbiz.agent.dto.LookupResult;
|
import com.superbiz.agent.dto.LookupResult;
|
||||||
import com.superbiz.agent.service.KnowledgeIndexService;
|
import com.superbiz.agent.service.KnowledgeIndexService;
|
||||||
|
import com.superbiz.agent.service.ToolInvocationRecorder;
|
||||||
import com.superbiz.agent.service.VectorSearchService;
|
import com.superbiz.agent.service.VectorSearchService;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
import org.junit.jupiter.api.BeforeEach;
|
import org.junit.jupiter.api.BeforeEach;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.mockito.InjectMocks;
|
import org.mockito.InjectMocks;
|
||||||
@@ -28,6 +30,15 @@ class LookupKnowledgeToolTest {
|
|||||||
@Mock
|
@Mock
|
||||||
private VectorSearchService vectorSearchService;
|
private VectorSearchService vectorSearchService;
|
||||||
|
|
||||||
|
@Mock
|
||||||
|
private ToolInvocationRecorder toolInvocationRecorder;
|
||||||
|
|
||||||
|
@Mock
|
||||||
|
private RetrievedDocTracker retrievedDocTracker;
|
||||||
|
|
||||||
|
@Mock
|
||||||
|
private ObjectMapper objectMapper;
|
||||||
|
|
||||||
@InjectMocks
|
@InjectMocks
|
||||||
private LookupKnowledgeTool tool;
|
private LookupKnowledgeTool tool;
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user