Add diagnosis eval harness
This commit is contained in:
@@ -0,0 +1,35 @@
|
||||
# Diagnosis Eval Harness
|
||||
|
||||
This folder contains the first fixed-case evaluation set for the MVP diagnosis Agent.
|
||||
|
||||
## Scope
|
||||
|
||||
- Case definitions: `cases/diagnosis-cases.json`
|
||||
- Offline trace fixtures: `fixtures/*.json`
|
||||
- Field definitions: `schema.md`
|
||||
- Evaluator implementation: `DiagnosisTraceEvaluator`
|
||||
- Report writer: `DiagnosisEvalReportWriter`
|
||||
|
||||
## Current Mode
|
||||
|
||||
The first version evaluates saved trace fixtures. It does not start the application and does not require MySQL, Redis, Milvus, or a real LLM.
|
||||
|
||||
## Verification
|
||||
|
||||
Run the focused evaluator test:
|
||||
|
||||
```powershell
|
||||
mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test
|
||||
```
|
||||
|
||||
## Interview Story
|
||||
|
||||
The harness gives the MVP a repeatable baseline:
|
||||
|
||||
```text
|
||||
fixed diagnosis case
|
||||
-> saved or runtime trace
|
||||
-> rule-based trace validation
|
||||
-> JSON / Markdown report
|
||||
-> regression signal for prompts, tools, retrieval, and verifier behavior
|
||||
```
|
||||
@@ -0,0 +1,57 @@
|
||||
[
|
||||
{
|
||||
"id": "payment-timeout",
|
||||
"title": "Payment API timeout",
|
||||
"question": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。",
|
||||
"traceFixture": "payment-timeout-pass.json",
|
||||
"expectedRootCauseKeywords": ["支付", "超时", "连接池"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["lookup_knowledge", "query_logs", "query_metrics"],
|
||||
"allowedVerdicts": ["PASS", "LOW_CONFID"],
|
||||
"forbiddenAnswerKeywords": ["无证据确定"]
|
||||
},
|
||||
{
|
||||
"id": "mysql-pool-exhausted",
|
||||
"title": "MySQL connection pool exhausted",
|
||||
"question": "订单服务大量请求超时,请判断是否和 MySQL 连接池有关。",
|
||||
"traceFixture": "mysql-pool-low-confid.json",
|
||||
"expectedRootCauseKeywords": ["mysql", "连接池", "超时"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["lookup_knowledge", "query_logs"],
|
||||
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||
"forbiddenAnswerKeywords": ["已经完全确认"]
|
||||
},
|
||||
{
|
||||
"id": "redis-timeout",
|
||||
"title": "Redis timeout",
|
||||
"question": "支付服务出现 Redis 连接超时,请定位可能原因。",
|
||||
"traceFixture": "redis-timeout-missing.json",
|
||||
"expectedRootCauseKeywords": ["redis", "超时"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["query_logs"],
|
||||
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||
"forbiddenAnswerKeywords": ["无需进一步排查"]
|
||||
},
|
||||
{
|
||||
"id": "slow-response",
|
||||
"title": "Slow response",
|
||||
"question": "用户服务 P99 响应时间升高,请结合指标和日志分析。",
|
||||
"traceFixture": "slow-response-missing.json",
|
||||
"expectedRootCauseKeywords": ["p99", "慢响应"],
|
||||
"minKeywordMatches": 1,
|
||||
"requiredEvidenceTools": ["query_metrics", "query_logs"],
|
||||
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||
"forbiddenAnswerKeywords": ["没有风险"]
|
||||
},
|
||||
{
|
||||
"id": "jvm-memory-risk",
|
||||
"title": "JVM memory risk",
|
||||
"question": "订单服务内存使用率过高,请判断是否存在 OOM 风险。",
|
||||
"traceFixture": "jvm-memory-risk-missing.json",
|
||||
"expectedRootCauseKeywords": ["jvm", "内存", "oom"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["query_metrics", "query_logs"],
|
||||
"allowedVerdicts": ["LOW_CONFID", "PASS"],
|
||||
"forbiddenAnswerKeywords": ["可以忽略"]
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,53 @@
|
||||
{
|
||||
"session": {
|
||||
"sessionId": "eval-mysql-pool",
|
||||
"query": "订单服务大量请求超时,请判断是否和 MySQL 连接池有关。",
|
||||
"status": "SUCCESS",
|
||||
"agentFlow": "CHAT",
|
||||
"totalDurationMs": 51000,
|
||||
"toolCallCount": 2,
|
||||
"answer": "以下结论基于当前已获取证据,仍存在部分证据缺口,请谨慎参考。\n\nMySQL 连接池可能参与了本次超时问题。日志中出现 connection pool exhausted,但当前缺少完整指标证据,因此只能作为低置信结论处理。",
|
||||
"selfEvaluation": {
|
||||
"verifier_evaluation": {
|
||||
"verdict": "LOW_CONFID",
|
||||
"groundedness_score": 0.48,
|
||||
"tool_trace_summary": [
|
||||
{
|
||||
"tool_name": "lookup_knowledge",
|
||||
"success": true,
|
||||
"evidence_level": "direct"
|
||||
},
|
||||
{
|
||||
"tool_name": "query_logs",
|
||||
"success": true,
|
||||
"evidence_level": "direct"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"steps": [],
|
||||
"toolInvocations": [
|
||||
{
|
||||
"id": 1,
|
||||
"sessionId": "eval-mysql-pool",
|
||||
"toolName": "lookup_knowledge",
|
||||
"success": true,
|
||||
"relevanceLevel": "PRECISE"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"sessionId": "eval-mysql-pool",
|
||||
"toolName": "query_logs",
|
||||
"success": true
|
||||
}
|
||||
],
|
||||
"summary": {
|
||||
"persistedStepCount": 3,
|
||||
"returnedStepCount": 3,
|
||||
"persistedToolCallCount": 2,
|
||||
"returnedToolCallCount": 2,
|
||||
"hasVerifierEvaluation": true,
|
||||
"hasFeedback": false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
{
|
||||
"session": {
|
||||
"sessionId": "eval-payment-timeout",
|
||||
"query": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。",
|
||||
"status": "SUCCESS",
|
||||
"agentFlow": "CHAT",
|
||||
"totalDurationMs": 42000,
|
||||
"toolCallCount": 3,
|
||||
"answer": "支付接口超时与连接池等待有关。知识库说明支付超时需要同时检查连接池、日志和指标;日志出现 connection pool exhausted;指标显示支付服务延迟升高。",
|
||||
"selfEvaluation": {
|
||||
"verifier_evaluation": {
|
||||
"verdict": "PASS",
|
||||
"groundedness_score": 0.86,
|
||||
"tool_trace_summary": [
|
||||
{
|
||||
"tool_name": "lookup_knowledge",
|
||||
"success": true,
|
||||
"evidence_level": "direct"
|
||||
},
|
||||
{
|
||||
"tool_name": "query_logs",
|
||||
"success": true,
|
||||
"evidence_level": "direct"
|
||||
},
|
||||
{
|
||||
"tool_name": "query_metrics",
|
||||
"success": true,
|
||||
"evidence_level": "direct"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"steps": [],
|
||||
"toolInvocations": [
|
||||
{
|
||||
"id": 1,
|
||||
"sessionId": "eval-payment-timeout",
|
||||
"toolName": "lookup_knowledge",
|
||||
"success": true,
|
||||
"relevanceLevel": "PRECISE"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"sessionId": "eval-payment-timeout",
|
||||
"toolName": "query_logs",
|
||||
"success": true
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"sessionId": "eval-payment-timeout",
|
||||
"toolName": "query_metrics",
|
||||
"success": true
|
||||
}
|
||||
],
|
||||
"summary": {
|
||||
"persistedStepCount": 3,
|
||||
"returnedStepCount": 3,
|
||||
"persistedToolCallCount": 3,
|
||||
"returnedToolCallCount": 3,
|
||||
"hasVerifierEvaluation": true,
|
||||
"hasFeedback": false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
# Diagnosis Eval Data Schema
|
||||
|
||||
这份文档记录评测基准里的数据结构。口语化理解就是:
|
||||
|
||||
```text
|
||||
用例文件说“我要考什么”
|
||||
trace 文件说“Agent 实际做了什么”
|
||||
评测结果说“这次有没有跑偏”
|
||||
汇总报告说“整体稳定性怎么样”
|
||||
```
|
||||
|
||||
当前这套评测是代码规则判断,不是再调用一个 LLM 来打分。
|
||||
|
||||
## 1. 用例定义
|
||||
|
||||
文件:`mvp/eval/cases/diagnosis-cases.json`
|
||||
|
||||
每一条 case 是一个固定考题,告诉评测器“这个问题应该看哪些点、需要哪些证据、哪些结论可以接受”。
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "payment-timeout",
|
||||
"title": "Payment API timeout",
|
||||
"question": "支付接口最近出现超时,请结合知识库、日志和指标判断可能原因,并给出修复建议。",
|
||||
"traceFixture": "payment-timeout-pass.json",
|
||||
"expectedRootCauseKeywords": ["支付", "超时", "连接池"],
|
||||
"minKeywordMatches": 2,
|
||||
"requiredEvidenceTools": ["lookup_knowledge", "query_logs", "query_metrics"],
|
||||
"allowedVerdicts": ["PASS", "LOW_CONFID"],
|
||||
"forbiddenAnswerKeywords": ["无证据确定"]
|
||||
}
|
||||
```
|
||||
|
||||
字段说明:
|
||||
|
||||
| 字段 | 意思 | 评测器怎么用 |
|
||||
| --- | --- | --- |
|
||||
| `id` | 这条用例的唯一名字 | 出现在报告里,方便定位是哪条 case 挂了 |
|
||||
| `title` | 给人看的标题 | 出现在结果里,方便快速理解场景 |
|
||||
| `question` | 要问 Agent 的问题 | fixture 模式下不会真的发送给 Agent,但它记录了这条 case 的原始输入 |
|
||||
| `traceFixture` | 对应的 trace 文件名 | 评测器会去 `fixtures/` 目录加载这个文件 |
|
||||
| `expectedRootCauseKeywords` | 最终回答里希望看到的关键点 | 评测器会在 `session.answer` 里做关键词命中检查 |
|
||||
| `minKeywordMatches` | 至少要命中几个关键词 | 命中数低于这个值,就认为根因覆盖不够 |
|
||||
| `requiredEvidenceTools` | 这条 case 至少应该用到哪些证据工具 | 评测器会检查 trace 里是否出现这些工具 |
|
||||
| `allowedVerdicts` | Verifier 允许给出的结论 | 比如 `PASS` 或 `LOW_CONFID`,不在列表里就失败 |
|
||||
| `forbiddenAnswerKeywords` | 回答里不应该出现的危险说法 | 命中这些词,说明回答可能过度自信或不符合降级策略 |
|
||||
|
||||
## 2. Trace Fixture
|
||||
|
||||
目录:`mvp/eval/fixtures/*.json`
|
||||
|
||||
trace fixture 是一次 Agent 运行后的“留痕快照”。评测器不会关心整个 trace 的所有字段,只读取当前能支撑基准判断的字段。
|
||||
|
||||
当前会读取这些字段:
|
||||
|
||||
| Trace 字段 | 意思 | 评测器怎么用 |
|
||||
| --- | --- | --- |
|
||||
| `session.answer` | Agent 最终给用户的回答 | 用来检查根因关键词和禁用词 |
|
||||
| `session.totalDurationMs` | 这次运行耗时 | 进入报告,帮助观察性能是否明显变差 |
|
||||
| `session.selfEvaluation.verifier_evaluation.verdict` | Verifier 对最终回答的判断 | 必须存在,并且要落在 case 的 `allowedVerdicts` 里 |
|
||||
| `session.selfEvaluation.verifier_evaluation.tool_trace_summary[*].tool_name` | Verifier 总结里看到的工具证据 | 用来补充判断证据工具是否出现 |
|
||||
| `toolInvocations[*].toolName` | Agent 实际调用过的工具名 | 用来检查 `requiredEvidenceTools` 是否满足 |
|
||||
| `toolInvocations[*].success` | 工具调用是否成功 | 当前主要保留在 trace 里,后续可以升级成更严格的成功率检查 |
|
||||
|
||||
简单说,trace 里最重要的是三类信息:
|
||||
|
||||
```text
|
||||
最终回答:它说了什么
|
||||
工具证据:它查了什么
|
||||
Verifier:它自己有没有承认这个结论可靠
|
||||
```
|
||||
|
||||
## 3. 单条评测结果
|
||||
|
||||
Java 类型:`DiagnosisEvalResult`
|
||||
|
||||
这是每条 case 跑完之后的判断结果。
|
||||
|
||||
| 字段 | 意思 |
|
||||
| --- | --- |
|
||||
| `caseId` | 对应的 case id |
|
||||
| `title` | case 标题 |
|
||||
| `passed` | 这条 case 是否通过 |
|
||||
| `failedChecks` | 没通过的具体原因,比如缺工具、关键词不够、verdict 不允许 |
|
||||
| `verdict` | 从 trace 里读出来的 Verifier verdict |
|
||||
| `matchedKeywordCount` | 最终回答命中的关键词数量 |
|
||||
| `requiredKeywordCount` | case 定义里一共有多少个关键词 |
|
||||
| `evidenceCoverage` | 每个必需工具是否出现,例如 `{ "query_logs": true }` |
|
||||
| `toolCallCount` | 本次 trace 里工具调用总数 |
|
||||
| `durationMs` | 本次 trace 的耗时 |
|
||||
|
||||
判断通过的口语化规则:
|
||||
|
||||
```text
|
||||
回答要说到关键点
|
||||
该查的证据工具要查到
|
||||
Verifier 的结论要在可接受范围内
|
||||
回答不能出现危险的过度自信表达
|
||||
如果是 REJECT,就必须走降级模板
|
||||
```
|
||||
|
||||
## 4. 汇总报告
|
||||
|
||||
Java 类型:`DiagnosisEvalReport`
|
||||
|
||||
这是整个基准集跑完之后的总结果。
|
||||
|
||||
| 字段 | 意思 |
|
||||
| --- | --- |
|
||||
| `totalCases` | 总共评测了多少条 case |
|
||||
| `passedCases` | 通过了多少条 |
|
||||
| `passRate` | 通过率,范围是 `0.0` 到 `1.0` |
|
||||
| `verdictDistribution` | Verifier verdict 的分布,比如有几个 `PASS`、几个 `LOW_CONFID` |
|
||||
| `averageToolCallCount` | 平均每条 case 调用了多少次工具 |
|
||||
| `averageDurationMs` | 平均耗时 |
|
||||
| `results` | 每条 case 的详细结果列表 |
|
||||
|
||||
## 5. 怎么看这个基准
|
||||
|
||||
这套结构不是为了证明 Agent 永远正确,而是为了在每次改 prompt、工具、检索、Verifier 之后,有一个固定尺子能回答:
|
||||
|
||||
```text
|
||||
以前能过的诊断题,现在还过不过?
|
||||
它是不是少查了某些证据?
|
||||
它是不是变得更自信但证据不足?
|
||||
它是不是开始输出不该说的话?
|
||||
它是不是明显变慢了?
|
||||
```
|
||||
|
||||
所以面试里可以这样讲:
|
||||
|
||||
```text
|
||||
我没有只看一次 demo 效果,而是把典型诊断场景固化成 case。
|
||||
每条 case 都定义预期关键点、必需证据工具和可接受的 verifier 结论。
|
||||
Agent 每次运行会留下 trace,评测器用代码规则读取 trace,输出结构化报告。
|
||||
这样我改 Agent 的时候,可以用同一套基准判断有没有行为回退。
|
||||
```
|
||||
Reference in New Issue
Block a user