62 lines
2.7 KiB
JSON
62 lines
2.7 KiB
JSON
{
|
|
"version": 1,
|
|
"description": "Offline golden retrieval cases for RAG refactor baseline.",
|
|
"topK": 5,
|
|
"cases": [
|
|
{
|
|
"caseId": "chat-mysql-connection-pool",
|
|
"scenario": "chat",
|
|
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
|
"expectedDocIds": ["mysql-connection-pool"],
|
|
"expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
|
|
"expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
|
|
"notes": "Covers precise database troubleshooting retrieval."
|
|
},
|
|
{
|
|
"caseId": "chat-diagnosis-flow",
|
|
"scenario": "chat",
|
|
"query": "What is the standard troubleshooting flow for an application incident?",
|
|
"expectedDocIds": ["incident-diagnosis-flow"],
|
|
"expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
|
|
"expectedKeywords": ["collect evidence", "verify", "remediation"],
|
|
"notes": "Covers process-style knowledge where breadcrumb matters."
|
|
},
|
|
{
|
|
"caseId": "aiops-payment-latency-alert",
|
|
"scenario": "aiops",
|
|
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
|
"expectedDocIds": ["payment-service-latency"],
|
|
"expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
|
|
"expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
|
|
"notes": "Covers alert payload terms that should become retrieval hints."
|
|
},
|
|
{
|
|
"caseId": "aiops-prometheus-alert-scope",
|
|
"scenario": "aiops",
|
|
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
|
"expectedDocIds": ["aiops-alert-scope-control"],
|
|
"expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
|
|
"expectedKeywords": ["payload", "unrelated active alerts", "scope"],
|
|
"notes": "Covers scoped alert diagnosis behavior."
|
|
},
|
|
{
|
|
"caseId": "chat-rag-chunk-context",
|
|
"scenario": "chat",
|
|
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
|
"expectedDocIds": ["rag-chunk-context-reconstruction"],
|
|
"expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
|
|
"expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
|
|
"notes": "Covers the known RAG refactor issue around context reconstruction."
|
|
},
|
|
{
|
|
"caseId": "chat-l0-domain-hint",
|
|
"scenario": "chat",
|
|
"query": "Should L0 keyword matching decide the final retrieval result?",
|
|
"expectedDocIds": ["rag-l0-domain-entity-hint"],
|
|
"expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
|
|
"expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
|
|
"notes": "Covers the target L0 role after refactor."
|
|
}
|
|
]
|
|
}
|