113 lines
5.5 KiB
JSON
113 lines
5.5 KiB
JSON
{
|
|
"version": 1,
|
|
"description": "Offline golden retrieval cases for RAG refactor baseline.",
|
|
"topK": 5,
|
|
"cases": [
|
|
{
|
|
"caseId": "chat-mysql-connection-pool",
|
|
"scenario": "chat",
|
|
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
|
"expectedDocIds": ["mysql-connection-pool"],
|
|
"expectedSources": ["mysql-connection-pool"],
|
|
"expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
|
|
"expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
|
|
"expectedSelectedAttempt": "FILTERED_VECTOR",
|
|
"expectedFallbackReason": null,
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["mysql-connection-pool"],
|
|
"expectedRerankTopSource": "mysql-connection-pool",
|
|
"notes": "Covers precise database troubleshooting retrieval."
|
|
},
|
|
{
|
|
"caseId": "chat-diagnosis-flow",
|
|
"scenario": "chat",
|
|
"query": "What is the standard troubleshooting flow for an application incident?",
|
|
"expectedDocIds": ["incident-diagnosis-flow"],
|
|
"expectedSources": ["incident-diagnosis-flow"],
|
|
"expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
|
|
"expectedKeywords": ["collect evidence", "verify", "remediation"],
|
|
"expectedSelectedAttempt": "FILTERED_VECTOR",
|
|
"expectedFallbackReason": null,
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["incident-diagnosis-flow"],
|
|
"expectedRerankTopSource": "incident-diagnosis-flow",
|
|
"notes": "Covers process-style knowledge where breadcrumb matters."
|
|
},
|
|
{
|
|
"caseId": "aiops-payment-latency-alert",
|
|
"scenario": "aiops",
|
|
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
|
"expectedDocIds": ["payment-service-latency"],
|
|
"expectedSources": ["payment-service-latency"],
|
|
"expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
|
|
"expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
|
|
"expectedSelectedAttempt": "FILTERED_VECTOR",
|
|
"expectedFallbackReason": null,
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["payment-service-latency"],
|
|
"expectedRerankTopSource": "payment-service-latency",
|
|
"notes": "Covers alert payload terms that should become retrieval hints."
|
|
},
|
|
{
|
|
"caseId": "aiops-prometheus-alert-scope",
|
|
"scenario": "aiops",
|
|
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
|
"expectedDocIds": ["aiops-alert-scope-control"],
|
|
"expectedSources": ["aiops-alert-scope-control"],
|
|
"expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
|
|
"expectedKeywords": ["payload", "unrelated active alerts", "scope"],
|
|
"expectedSelectedAttempt": "FILTERED_VECTOR",
|
|
"expectedFallbackReason": null,
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["aiops-alert-scope-control"],
|
|
"expectedRerankTopSource": "aiops-alert-scope-control",
|
|
"notes": "Covers scoped alert diagnosis behavior."
|
|
},
|
|
{
|
|
"caseId": "chat-rag-chunk-context",
|
|
"scenario": "chat",
|
|
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
|
"expectedDocIds": ["rag-chunk-context-reconstruction"],
|
|
"expectedSources": ["rag-chunk-context-reconstruction"],
|
|
"expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
|
|
"expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
|
|
"expectedSelectedAttempt": "FILTERED_VECTOR",
|
|
"expectedFallbackReason": null,
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["rag-chunk-context-reconstruction"],
|
|
"expectedRerankTopSource": "rag-chunk-context-reconstruction",
|
|
"notes": "Covers the known RAG refactor issue around context reconstruction."
|
|
},
|
|
{
|
|
"caseId": "chat-l0-domain-hint",
|
|
"scenario": "chat",
|
|
"query": "Should L0 keyword matching decide the final retrieval result?",
|
|
"expectedDocIds": ["rag-l0-domain-entity-hint"],
|
|
"expectedSources": ["rag-l0-domain-entity-hint"],
|
|
"expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
|
|
"expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
|
|
"expectedSelectedAttempt": "FILTERED_VECTOR",
|
|
"expectedFallbackReason": null,
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["rag-l0-domain-entity-hint"],
|
|
"expectedRerankTopSource": "rag-l0-domain-entity-hint",
|
|
"notes": "Covers the target L0 role after refactor."
|
|
},
|
|
{
|
|
"caseId": "chat-l0-filter-fallback",
|
|
"scenario": "chat",
|
|
"query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?",
|
|
"expectedDocIds": ["rag-l0-filter-fallback"],
|
|
"expectedSources": ["rag-l0-filter-fallback"],
|
|
"expectedBreadcrumbs": ["RAG > Fallback > Unfiltered Retry"],
|
|
"expectedKeywords": ["skip the L0 filter", "unfiltered vector retry", "low quality"],
|
|
"expectedSelectedAttempt": "UNFILTERED_VECTOR_RETRY",
|
|
"expectedFallbackReasons": ["filtered_vector_low_quality", "filtered_vector_no_evidence"],
|
|
"expectedEvidenceStatus": "supported",
|
|
"expectedContextSources": ["rag-l0-filter-fallback"],
|
|
"expectedRerankTopSource": "rag-l0-filter-fallback",
|
|
"notes": "Covers the MVP fallback rule: if filtered L1 is low quality, retry raw query without L0 filter."
|
|
}
|
|
]
|
|
}
|