Files

224 lines
6.5 KiB
JSON

{
"generatedAt": "2026-07-06T13:37:59.726351+00:00",
"caseFile": "eval/rag-retrieval/cases/golden-cases.json",
"fixtureDir": "eval/rag-retrieval/fixtures",
"aggregate": {
"caseCount": 7,
"topK": 5,
"passedCount": 7,
"failedCount": 0,
"passRate": 1.0,
"lookupResultCaseCount": 7,
"strongHitCount": 7,
"mediumHitCount": 0,
"weakHitCount": 0,
"missCount": 0,
"recallAtK": 1.0,
"strongHitRate": 1.0,
"averageFirstHitRank": 1.0
},
"results": [
{
"caseId": "chat-mysql-connection-pool",
"scenario": "chat",
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:mysql-connection-pool",
"2:incident-diagnosis-flow"
],
"matchedKeywords": [
"connection pool",
"max_connections",
"hikaricp"
],
"breadcrumbMatched": true,
"selectedAttempt": "FILTERED_VECTOR",
"fallbackReason": null,
"evidenceStatus": "supported",
"includedSources": [
"mysql-connection-pool",
"incident-diagnosis-flow"
],
"omittedSources": [],
"rerankTopSource": "mysql-connection-pool",
"failedChecks": []
},
{
"caseId": "chat-diagnosis-flow",
"scenario": "chat",
"query": "What is the standard troubleshooting flow for an application incident?",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:incident-diagnosis-flow",
"2:rag-chunk-context-reconstruction"
],
"matchedKeywords": [
"collect evidence",
"verify",
"remediation"
],
"breadcrumbMatched": true,
"selectedAttempt": "FILTERED_VECTOR",
"fallbackReason": null,
"evidenceStatus": "supported",
"includedSources": [
"incident-diagnosis-flow",
"rag-chunk-context-reconstruction"
],
"omittedSources": [],
"rerankTopSource": "incident-diagnosis-flow",
"failedChecks": []
},
{
"caseId": "aiops-payment-latency-alert",
"scenario": "aiops",
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:payment-service-latency",
"2:mysql-connection-pool"
],
"matchedKeywords": [
"p95 latency",
"payment-service",
"downstream dependency"
],
"breadcrumbMatched": true,
"selectedAttempt": "FILTERED_VECTOR",
"fallbackReason": null,
"evidenceStatus": "supported",
"includedSources": [
"payment-service-latency",
"mysql-connection-pool"
],
"omittedSources": [],
"rerankTopSource": "payment-service-latency",
"failedChecks": []
},
{
"caseId": "aiops-prometheus-alert-scope",
"scenario": "aiops",
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:aiops-alert-scope-control"
],
"matchedKeywords": [
"payload",
"unrelated active alerts",
"scope"
],
"breadcrumbMatched": true,
"selectedAttempt": "FILTERED_VECTOR",
"fallbackReason": null,
"evidenceStatus": "supported",
"includedSources": [
"aiops-alert-scope-control"
],
"omittedSources": [],
"rerankTopSource": "aiops-alert-scope-control",
"failedChecks": []
},
{
"caseId": "chat-rag-chunk-context",
"scenario": "chat",
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:rag-chunk-context-reconstruction",
"2:rag-breadcrumb-embedding-gap"
],
"matchedKeywords": [
"neighbor chunk",
"same section",
"breadcrumb"
],
"breadcrumbMatched": true,
"selectedAttempt": "FILTERED_VECTOR",
"fallbackReason": null,
"evidenceStatus": "supported",
"includedSources": [
"rag-chunk-context-reconstruction",
"rag-breadcrumb-embedding-gap"
],
"omittedSources": [],
"rerankTopSource": "rag-chunk-context-reconstruction",
"failedChecks": []
},
{
"caseId": "chat-l0-domain-hint",
"scenario": "chat",
"query": "Should L0 keyword matching decide the final retrieval result?",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:rag-l0-domain-entity-hint",
"2:rag-l0-l1-fusion-ranking"
],
"matchedKeywords": [
"domain detector",
"entity extractor",
"metadata filter"
],
"breadcrumbMatched": true,
"selectedAttempt": "FILTERED_VECTOR",
"fallbackReason": null,
"evidenceStatus": "supported",
"includedSources": [
"rag-l0-domain-entity-hint",
"rag-l0-l1-fusion-ranking"
],
"omittedSources": [],
"rerankTopSource": "rag-l0-domain-entity-hint",
"failedChecks": []
},
{
"caseId": "chat-l0-filter-fallback",
"scenario": "chat",
"query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?",
"dataShape": "lookupResult",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:rag-l0-filter-fallback",
"2:rag-l0-domain-entity-hint"
],
"matchedKeywords": [
"skip the l0 filter",
"unfiltered vector retry",
"low quality"
],
"breadcrumbMatched": true,
"selectedAttempt": "UNFILTERED_VECTOR_RETRY",
"fallbackReason": "filtered_vector_low_quality",
"evidenceStatus": "supported",
"includedSources": [
"rag-l0-filter-fallback",
"rag-l0-domain-entity-hint"
],
"omittedSources": [],
"rerankTopSource": "rag-l0-filter-fallback",
"failedChecks": []
}
]
}