{ "generatedAt": "2026-07-04T17:59:52.172759+00:00", "caseFile": "eval/rag-retrieval/cases/golden-cases.json", "fixtureDir": "eval/rag-retrieval/fixtures", "aggregate": { "caseCount": 6, "topK": 5, "strongHitCount": 6, "mediumHitCount": 0, "weakHitCount": 0, "missCount": 0, "recallAtK": 1.0, "strongHitRate": 1.0, "averageFirstHitRank": 1.0 }, "results": [ { "caseId": "chat-mysql-connection-pool", "scenario": "chat", "query": "MySQL connection pool is exhausted. How should I diagnose it?", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, "topCandidates": [ "1:mysql-connection-pool", "2:incident-diagnosis-flow" ], "matchedKeywords": [ "connection pool", "max_connections", "hikaricp" ], "breadcrumbMatched": true, "failedChecks": [] }, { "caseId": "chat-diagnosis-flow", "scenario": "chat", "query": "What is the standard troubleshooting flow for an application incident?", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, "topCandidates": [ "1:incident-diagnosis-flow", "2:rag-chunk-context-reconstruction" ], "matchedKeywords": [ "collect evidence", "verify", "remediation" ], "breadcrumbMatched": true, "failedChecks": [] }, { "caseId": "aiops-payment-latency-alert", "scenario": "aiops", "query": "Alert HighLatency on payment-service with p95 latency above threshold", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, "topCandidates": [ "1:payment-service-latency", "2:mysql-connection-pool" ], "matchedKeywords": [ "p95 latency", "payment-service", "downstream dependency" ], "breadcrumbMatched": true, "failedChecks": [] }, { "caseId": "aiops-prometheus-alert-scope", "scenario": "aiops", "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, "topCandidates": [ "1:aiops-alert-scope-control" ], "matchedKeywords": [ "payload", "unrelated active alerts", "scope" ], "breadcrumbMatched": true, "failedChecks": [] }, { "caseId": "chat-rag-chunk-context", "scenario": "chat", "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, "topCandidates": [ "1:rag-chunk-context-reconstruction", "2:rag-breadcrumb-embedding-gap" ], "matchedKeywords": [ "neighbor chunk", "same section", "breadcrumb" ], "breadcrumbMatched": true, "failedChecks": [] }, { "caseId": "chat-l0-domain-hint", "scenario": "chat", "query": "Should L0 keyword matching decide the final retrieval result?", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, "topCandidates": [ "1:rag-l0-domain-entity-hint", "2:rag-l0-l1-fusion-ranking" ], "matchedKeywords": [ "domain detector", "entity extractor", "metadata filter" ], "breadcrumbMatched": true, "failedChecks": [] } ] }