Files
SuperBizAgent-java/eval/rag-retrieval/cases/golden-cases.json
T

113 lines
5.5 KiB
JSON

{
"version": 1,
"description": "Offline golden retrieval cases for RAG refactor baseline.",
"topK": 5,
"cases": [
{
"caseId": "chat-mysql-connection-pool",
"scenario": "chat",
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
"expectedDocIds": ["mysql-connection-pool"],
"expectedSources": ["mysql-connection-pool"],
"expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
"expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
"expectedSelectedAttempt": "FILTERED_VECTOR",
"expectedFallbackReason": null,
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["mysql-connection-pool"],
"expectedRerankTopSource": "mysql-connection-pool",
"notes": "Covers precise database troubleshooting retrieval."
},
{
"caseId": "chat-diagnosis-flow",
"scenario": "chat",
"query": "What is the standard troubleshooting flow for an application incident?",
"expectedDocIds": ["incident-diagnosis-flow"],
"expectedSources": ["incident-diagnosis-flow"],
"expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
"expectedKeywords": ["collect evidence", "verify", "remediation"],
"expectedSelectedAttempt": "FILTERED_VECTOR",
"expectedFallbackReason": null,
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["incident-diagnosis-flow"],
"expectedRerankTopSource": "incident-diagnosis-flow",
"notes": "Covers process-style knowledge where breadcrumb matters."
},
{
"caseId": "aiops-payment-latency-alert",
"scenario": "aiops",
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
"expectedDocIds": ["payment-service-latency"],
"expectedSources": ["payment-service-latency"],
"expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
"expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
"expectedSelectedAttempt": "FILTERED_VECTOR",
"expectedFallbackReason": null,
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["payment-service-latency"],
"expectedRerankTopSource": "payment-service-latency",
"notes": "Covers alert payload terms that should become retrieval hints."
},
{
"caseId": "aiops-prometheus-alert-scope",
"scenario": "aiops",
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
"expectedDocIds": ["aiops-alert-scope-control"],
"expectedSources": ["aiops-alert-scope-control"],
"expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
"expectedKeywords": ["payload", "unrelated active alerts", "scope"],
"expectedSelectedAttempt": "FILTERED_VECTOR",
"expectedFallbackReason": null,
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["aiops-alert-scope-control"],
"expectedRerankTopSource": "aiops-alert-scope-control",
"notes": "Covers scoped alert diagnosis behavior."
},
{
"caseId": "chat-rag-chunk-context",
"scenario": "chat",
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
"expectedDocIds": ["rag-chunk-context-reconstruction"],
"expectedSources": ["rag-chunk-context-reconstruction"],
"expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
"expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
"expectedSelectedAttempt": "FILTERED_VECTOR",
"expectedFallbackReason": null,
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["rag-chunk-context-reconstruction"],
"expectedRerankTopSource": "rag-chunk-context-reconstruction",
"notes": "Covers the known RAG refactor issue around context reconstruction."
},
{
"caseId": "chat-l0-domain-hint",
"scenario": "chat",
"query": "Should L0 keyword matching decide the final retrieval result?",
"expectedDocIds": ["rag-l0-domain-entity-hint"],
"expectedSources": ["rag-l0-domain-entity-hint"],
"expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
"expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
"expectedSelectedAttempt": "FILTERED_VECTOR",
"expectedFallbackReason": null,
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["rag-l0-domain-entity-hint"],
"expectedRerankTopSource": "rag-l0-domain-entity-hint",
"notes": "Covers the target L0 role after refactor."
},
{
"caseId": "chat-l0-filter-fallback",
"scenario": "chat",
"query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?",
"expectedDocIds": ["rag-l0-filter-fallback"],
"expectedSources": ["rag-l0-filter-fallback"],
"expectedBreadcrumbs": ["RAG > Fallback > Unfiltered Retry"],
"expectedKeywords": ["skip the L0 filter", "unfiltered vector retry", "low quality"],
"expectedSelectedAttempt": "UNFILTERED_VECTOR_RETRY",
"expectedFallbackReasons": ["filtered_vector_low_quality", "filtered_vector_no_evidence"],
"expectedEvidenceStatus": "supported",
"expectedContextSources": ["rag-l0-filter-fallback"],
"expectedRerankTopSource": "rag-l0-filter-fallback",
"notes": "Covers the MVP fallback rule: if filtered L1 is low quality, retry raw query without L0 filter."
}
]
}