132 lines
3.5 KiB
JSON
132 lines
3.5 KiB
JSON
{
|
|
"generatedAt": "2026-07-04T17:59:52.172759+00:00",
|
|
"caseFile": "eval/rag-retrieval/cases/golden-cases.json",
|
|
"fixtureDir": "eval/rag-retrieval/fixtures",
|
|
"aggregate": {
|
|
"caseCount": 6,
|
|
"topK": 5,
|
|
"strongHitCount": 6,
|
|
"mediumHitCount": 0,
|
|
"weakHitCount": 0,
|
|
"missCount": 0,
|
|
"recallAtK": 1.0,
|
|
"strongHitRate": 1.0,
|
|
"averageFirstHitRank": 1.0
|
|
},
|
|
"results": [
|
|
{
|
|
"caseId": "chat-mysql-connection-pool",
|
|
"scenario": "chat",
|
|
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
|
"hitLevel": "strong",
|
|
"passed": true,
|
|
"firstExpectedRank": 1,
|
|
"topCandidates": [
|
|
"1:mysql-connection-pool",
|
|
"2:incident-diagnosis-flow"
|
|
],
|
|
"matchedKeywords": [
|
|
"connection pool",
|
|
"max_connections",
|
|
"hikaricp"
|
|
],
|
|
"breadcrumbMatched": true,
|
|
"failedChecks": []
|
|
},
|
|
{
|
|
"caseId": "chat-diagnosis-flow",
|
|
"scenario": "chat",
|
|
"query": "What is the standard troubleshooting flow for an application incident?",
|
|
"hitLevel": "strong",
|
|
"passed": true,
|
|
"firstExpectedRank": 1,
|
|
"topCandidates": [
|
|
"1:incident-diagnosis-flow",
|
|
"2:rag-chunk-context-reconstruction"
|
|
],
|
|
"matchedKeywords": [
|
|
"collect evidence",
|
|
"verify",
|
|
"remediation"
|
|
],
|
|
"breadcrumbMatched": true,
|
|
"failedChecks": []
|
|
},
|
|
{
|
|
"caseId": "aiops-payment-latency-alert",
|
|
"scenario": "aiops",
|
|
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
|
"hitLevel": "strong",
|
|
"passed": true,
|
|
"firstExpectedRank": 1,
|
|
"topCandidates": [
|
|
"1:payment-service-latency",
|
|
"2:mysql-connection-pool"
|
|
],
|
|
"matchedKeywords": [
|
|
"p95 latency",
|
|
"payment-service",
|
|
"downstream dependency"
|
|
],
|
|
"breadcrumbMatched": true,
|
|
"failedChecks": []
|
|
},
|
|
{
|
|
"caseId": "aiops-prometheus-alert-scope",
|
|
"scenario": "aiops",
|
|
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
|
"hitLevel": "strong",
|
|
"passed": true,
|
|
"firstExpectedRank": 1,
|
|
"topCandidates": [
|
|
"1:aiops-alert-scope-control"
|
|
],
|
|
"matchedKeywords": [
|
|
"payload",
|
|
"unrelated active alerts",
|
|
"scope"
|
|
],
|
|
"breadcrumbMatched": true,
|
|
"failedChecks": []
|
|
},
|
|
{
|
|
"caseId": "chat-rag-chunk-context",
|
|
"scenario": "chat",
|
|
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
|
"hitLevel": "strong",
|
|
"passed": true,
|
|
"firstExpectedRank": 1,
|
|
"topCandidates": [
|
|
"1:rag-chunk-context-reconstruction",
|
|
"2:rag-breadcrumb-embedding-gap"
|
|
],
|
|
"matchedKeywords": [
|
|
"neighbor chunk",
|
|
"same section",
|
|
"breadcrumb"
|
|
],
|
|
"breadcrumbMatched": true,
|
|
"failedChecks": []
|
|
},
|
|
{
|
|
"caseId": "chat-l0-domain-hint",
|
|
"scenario": "chat",
|
|
"query": "Should L0 keyword matching decide the final retrieval result?",
|
|
"hitLevel": "strong",
|
|
"passed": true,
|
|
"firstExpectedRank": 1,
|
|
"topCandidates": [
|
|
"1:rag-l0-domain-entity-hint",
|
|
"2:rag-l0-l1-fusion-ranking"
|
|
],
|
|
"matchedKeywords": [
|
|
"domain detector",
|
|
"entity extractor",
|
|
"metadata filter"
|
|
],
|
|
"breadcrumbMatched": true,
|
|
"failedChecks": []
|
|
}
|
|
]
|
|
}
|