test: add rag retrieval baseline
This commit is contained in:
@@ -0,0 +1,51 @@
|
||||
# RAG Retrieval Baseline
|
||||
|
||||
This directory contains the offline retrieval baseline for the RAG refactor.
|
||||
|
||||
The baseline is intentionally narrower than full diagnosis evaluation. It checks
|
||||
whether fixed retrieval queries can recover expected documents, breadcrumbs, and
|
||||
evidence keywords before changing L0 behavior, query augmentation, evidence
|
||||
post-processing, or Spring AI VectorStore integration.
|
||||
|
||||
## Layout
|
||||
|
||||
```text
|
||||
eval/rag-retrieval/
|
||||
cases/golden-cases.json Fixed retrieval golden cases
|
||||
fixtures/*.json Saved retrieval candidates for each case
|
||||
reports/baseline.json Machine-readable baseline report
|
||||
reports/baseline.md Human-readable baseline report
|
||||
```
|
||||
|
||||
## Run
|
||||
|
||||
From the repository root:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_retrieval.py
|
||||
```
|
||||
|
||||
Custom paths are also supported:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_retrieval.py \
|
||||
--cases eval/rag-retrieval/cases/golden-cases.json \
|
||||
--fixtures eval/rag-retrieval/fixtures \
|
||||
--json-report eval/rag-retrieval/reports/baseline.json \
|
||||
--markdown-report eval/rag-retrieval/reports/baseline.md
|
||||
```
|
||||
|
||||
## Hit Levels
|
||||
|
||||
- `strong`: expected document is found and breadcrumb or evidence keyword coverage is satisfied.
|
||||
- `medium`: expected document is found, but breadcrumb or keyword coverage is incomplete.
|
||||
- `weak`: expected evidence keyword is found, but expected document is missing.
|
||||
- `miss`: expected document and expected evidence are not found.
|
||||
|
||||
`Recall@K` counts `strong` and `medium` as retrieved.
|
||||
|
||||
## Scope
|
||||
|
||||
This baseline runs fully offline and does not call MySQL, Redis, Milvus, an LLM,
|
||||
or the Spring Boot application. It is a regression harness for retrieval behavior,
|
||||
not a claim that live production retrieval accuracy is complete.
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"version": 1,
|
||||
"description": "Offline golden retrieval cases for RAG refactor baseline.",
|
||||
"topK": 5,
|
||||
"cases": [
|
||||
{
|
||||
"caseId": "chat-mysql-connection-pool",
|
||||
"scenario": "chat",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"expectedDocIds": ["mysql-connection-pool"],
|
||||
"expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
|
||||
"expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
|
||||
"notes": "Covers precise database troubleshooting retrieval."
|
||||
},
|
||||
{
|
||||
"caseId": "chat-diagnosis-flow",
|
||||
"scenario": "chat",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"expectedDocIds": ["incident-diagnosis-flow"],
|
||||
"expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
|
||||
"expectedKeywords": ["collect evidence", "verify", "remediation"],
|
||||
"notes": "Covers process-style knowledge where breadcrumb matters."
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-payment-latency-alert",
|
||||
"scenario": "aiops",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"expectedDocIds": ["payment-service-latency"],
|
||||
"expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
|
||||
"expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
|
||||
"notes": "Covers alert payload terms that should become retrieval hints."
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-prometheus-alert-scope",
|
||||
"scenario": "aiops",
|
||||
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
||||
"expectedDocIds": ["aiops-alert-scope-control"],
|
||||
"expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
|
||||
"expectedKeywords": ["payload", "unrelated active alerts", "scope"],
|
||||
"notes": "Covers scoped alert diagnosis behavior."
|
||||
},
|
||||
{
|
||||
"caseId": "chat-rag-chunk-context",
|
||||
"scenario": "chat",
|
||||
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"expectedDocIds": ["rag-chunk-context-reconstruction"],
|
||||
"expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
|
||||
"expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
|
||||
"notes": "Covers the known RAG refactor issue around context reconstruction."
|
||||
},
|
||||
{
|
||||
"caseId": "chat-l0-domain-hint",
|
||||
"scenario": "chat",
|
||||
"query": "Should L0 keyword matching decide the final retrieval result?",
|
||||
"expectedDocIds": ["rag-l0-domain-entity-hint"],
|
||||
"expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
|
||||
"expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
|
||||
"notes": "Covers the target L0 role after refactor."
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "aiops-payment-latency-alert",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "payment-service-latency",
|
||||
"title": "Payment Service Latency Alert Playbook",
|
||||
"breadcrumb": "AIOps > Service Alerts > Payment Latency",
|
||||
"content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.",
|
||||
"score": 0.84,
|
||||
"retrievalLayer": "L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "mysql-connection-pool",
|
||||
"title": "MySQL Connection Pool Troubleshooting",
|
||||
"breadcrumb": "Database > MySQL > Connection Pool",
|
||||
"content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.",
|
||||
"score": 0.68,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
{
|
||||
"caseId": "aiops-prometheus-alert-scope",
|
||||
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "aiops-alert-scope-control",
|
||||
"title": "AIOps Alert Scope Control",
|
||||
"breadcrumb": "AIOps > Alert Scope Control",
|
||||
"content": "When payload mode is active, queryPrometheusAlerts can verify the supplied alert, but unrelated active alerts must remain scoped context and should not become full diagnoses.",
|
||||
"score": 0.9,
|
||||
"retrievalLayer": "L0+L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-diagnosis-flow",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "incident-diagnosis-flow",
|
||||
"title": "Incident Diagnosis Flow",
|
||||
"breadcrumb": "AIOps > Diagnosis Flow",
|
||||
"content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.",
|
||||
"score": 0.82,
|
||||
"retrievalLayer": "L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "rag-chunk-context-reconstruction",
|
||||
"title": "RAG Chunk Context Reconstruction",
|
||||
"breadcrumb": "RAG > Chunking > Context Reconstruction",
|
||||
"content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.",
|
||||
"score": 0.55,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-l0-domain-hint",
|
||||
"query": "Should L0 keyword matching decide the final retrieval result?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "rag-l0-domain-entity-hint",
|
||||
"title": "RAG L0 Domain Entity Hint",
|
||||
"breadcrumb": "RAG > L0 > Domain Entity Hint",
|
||||
"content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.",
|
||||
"score": 0.88,
|
||||
"retrievalLayer": "L0"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "rag-l0-l1-fusion-ranking",
|
||||
"title": "RAG L0 L1 Fusion Ranking",
|
||||
"breadcrumb": "RAG > Ranking > Fusion",
|
||||
"content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.",
|
||||
"score": 0.75,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-mysql-connection-pool",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "mysql-connection-pool",
|
||||
"title": "MySQL Connection Pool Troubleshooting",
|
||||
"breadcrumb": "Database > MySQL > Connection Pool",
|
||||
"content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.",
|
||||
"score": 0.86,
|
||||
"retrievalLayer": "L0+L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "incident-diagnosis-flow",
|
||||
"title": "Incident Diagnosis Flow",
|
||||
"breadcrumb": "AIOps > Diagnosis Flow",
|
||||
"content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.",
|
||||
"score": 0.61,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-rag-chunk-context",
|
||||
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "rag-chunk-context-reconstruction",
|
||||
"title": "RAG Chunk Context Reconstruction",
|
||||
"breadcrumb": "RAG > Chunking > Context Reconstruction",
|
||||
"content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.",
|
||||
"score": 0.79,
|
||||
"retrievalLayer": "L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "rag-breadcrumb-embedding-gap",
|
||||
"title": "RAG Breadcrumb Embedding Gap",
|
||||
"breadcrumb": "RAG > Embedding > Breadcrumb",
|
||||
"content": "Embedding title and breadcrumb with content helps recover section semantics.",
|
||||
"score": 0.72,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1 @@
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
{
|
||||
"generatedAt": "2026-07-04T17:59:52.172759+00:00",
|
||||
"caseFile": "eval/rag-retrieval/cases/golden-cases.json",
|
||||
"fixtureDir": "eval/rag-retrieval/fixtures",
|
||||
"aggregate": {
|
||||
"caseCount": 6,
|
||||
"topK": 5,
|
||||
"strongHitCount": 6,
|
||||
"mediumHitCount": 0,
|
||||
"weakHitCount": 0,
|
||||
"missCount": 0,
|
||||
"recallAtK": 1.0,
|
||||
"strongHitRate": 1.0,
|
||||
"averageFirstHitRank": 1.0
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"caseId": "chat-mysql-connection-pool",
|
||||
"scenario": "chat",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:mysql-connection-pool",
|
||||
"2:incident-diagnosis-flow"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"connection pool",
|
||||
"max_connections",
|
||||
"hikaricp"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "chat-diagnosis-flow",
|
||||
"scenario": "chat",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:incident-diagnosis-flow",
|
||||
"2:rag-chunk-context-reconstruction"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"collect evidence",
|
||||
"verify",
|
||||
"remediation"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-payment-latency-alert",
|
||||
"scenario": "aiops",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:payment-service-latency",
|
||||
"2:mysql-connection-pool"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"p95 latency",
|
||||
"payment-service",
|
||||
"downstream dependency"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-prometheus-alert-scope",
|
||||
"scenario": "aiops",
|
||||
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:aiops-alert-scope-control"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"payload",
|
||||
"unrelated active alerts",
|
||||
"scope"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "chat-rag-chunk-context",
|
||||
"scenario": "chat",
|
||||
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:rag-chunk-context-reconstruction",
|
||||
"2:rag-breadcrumb-embedding-gap"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"neighbor chunk",
|
||||
"same section",
|
||||
"breadcrumb"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "chat-l0-domain-hint",
|
||||
"scenario": "chat",
|
||||
"query": "Should L0 keyword matching decide the final retrieval result?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:rag-l0-domain-entity-hint",
|
||||
"2:rag-l0-l1-fusion-ranking"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"domain detector",
|
||||
"entity extractor",
|
||||
"metadata filter"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
# RAG Retrieval Baseline
|
||||
|
||||
Generated at: `2026-07-04T17:59:52.172759+00:00`
|
||||
|
||||
## Aggregate
|
||||
|
||||
| Metric | Value |
|
||||
|---|---:|
|
||||
| Cases | 6 |
|
||||
| Top K | 5 |
|
||||
| Recall@K | 1.0 |
|
||||
| Strong hit rate | 1.0 |
|
||||
| Strong hits | 6 |
|
||||
| Medium hits | 0 |
|
||||
| Weak hits | 0 |
|
||||
| Misses | 0 |
|
||||
| Average first hit rank | 1.0 |
|
||||
|
||||
## Cases
|
||||
|
||||
| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |
|
||||
|---|---|---|---:|---|---|
|
||||
| chat-mysql-connection-pool | chat | strong | 1 | 1:mysql-connection-pool<br>2:incident-diagnosis-flow | |
|
||||
| chat-diagnosis-flow | chat | strong | 1 | 1:incident-diagnosis-flow<br>2:rag-chunk-context-reconstruction | |
|
||||
| aiops-payment-latency-alert | aiops | strong | 1 | 1:payment-service-latency<br>2:mysql-connection-pool | |
|
||||
| aiops-prometheus-alert-scope | aiops | strong | 1 | 1:aiops-alert-scope-control | |
|
||||
| chat-rag-chunk-context | chat | strong | 1 | 1:rag-chunk-context-reconstruction<br>2:rag-breadcrumb-embedding-gap | |
|
||||
| chat-l0-domain-hint | chat | strong | 1 | 1:rag-l0-domain-entity-hint<br>2:rag-l0-l1-fusion-ranking | |
|
||||
@@ -202,6 +202,7 @@ relevanceLevel
|
||||
- 覆盖 Chat 和 AIOps 场景。
|
||||
- 每条 query 标注 expected doc、breadcrumb、关键 chunk 或 evidence。
|
||||
- 用当前链路跑一遍,记录 baseline。
|
||||
- 初始离线基线落在 `eval/rag-retrieval/`,用于后续 change 对比。
|
||||
|
||||
验收:
|
||||
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-04
|
||||
@@ -0,0 +1,62 @@
|
||||
## Context
|
||||
|
||||
The repository already has diagnosis-level evaluation specs and trace observability, but the RAG refactor plan needs a narrower retrieval baseline. The upcoming changes will alter L0 responsibilities, query augmentation, evidence post-processing, and eventually the vector store implementation. Those changes need a fixed set of retrieval cases and deterministic scoring before production retrieval behavior changes.
|
||||
|
||||
The first baseline must be offline. It should not require MySQL, Milvus, Redis, LLM calls, or a running Spring Boot application. It can evaluate saved retrieval result fixtures that represent the current behavior and produce reports that future changes can compare against.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Add a small fixed golden query set for RAG retrieval.
|
||||
- Evaluate retrieval fixtures against expected documents, breadcrumbs, evidence keywords, and hit levels.
|
||||
- Produce JSON and Markdown baseline reports.
|
||||
- Document how to regenerate the reports.
|
||||
- Keep the evaluator simple enough to run from the repository with Python.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not change `lookup_knowledge`, `VectorSearchService`, Milvus schema, L0 matching, or Agent prompts.
|
||||
- Do not require live services.
|
||||
- Do not implement Spring AI VectorStore migration in this change.
|
||||
- Do not implement RRF, BM25, rerank, or evidence packing in this change.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision: Use offline retrieval fixtures first
|
||||
|
||||
The evaluator will read saved retrieval fixtures rather than calling the live application.
|
||||
|
||||
Rationale: the first change should establish a stable measurement surface before the RAG internals change. Live retrieval depends on embeddings, Milvus state, and service configuration, which makes it a poor first baseline.
|
||||
|
||||
Alternative considered: call `SearchController` or `lookup_knowledge` directly. That is useful later, but it would require a running app and seeded knowledge base.
|
||||
|
||||
### Decision: Score by hit level, not exact chunk id only
|
||||
|
||||
The evaluator will classify each case as:
|
||||
|
||||
- `strong`: expected document plus expected breadcrumb or key evidence coverage.
|
||||
- `medium`: expected document found, but breadcrumb or evidence coverage is incomplete.
|
||||
- `weak`: related evidence is present but the expected document is missing.
|
||||
- `miss`: no expected document or expected evidence is found.
|
||||
|
||||
Rationale: chunk indexes can change after splitter changes, so exact chunk-only scoring would make later refactors look worse even when evidence quality is preserved.
|
||||
|
||||
### Decision: Keep case format explicit and reviewable
|
||||
|
||||
Golden cases will be stored as JSON with fields such as `caseId`, `query`, `expectedDocIds`, `expectedBreadcrumbs`, `expectedKeywords`, and optional `notes`.
|
||||
|
||||
Rationale: the case file should be easy to inspect in code review and easy to extend during interviews or later refactors.
|
||||
|
||||
### Decision: Preserve both machine and human reports
|
||||
|
||||
The evaluator will write JSON for automation and Markdown for review.
|
||||
|
||||
Rationale: future changes can compare JSON, while the Markdown report is easier to use during design review and interview preparation.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- Offline fixtures can drift from real runtime behavior -> add documentation that this is a baseline harness, not a live retrieval accuracy claim.
|
||||
- Keyword-based evidence checks are approximate -> use them only as deterministic guardrails, not as a replacement for human review.
|
||||
- Small golden set may underrepresent production queries -> start with 10-20 cases and expand as new RAG issues are found.
|
||||
- Fixture schema may not match future retrieval outputs -> normalize fixtures into a simple candidate shape and keep raw fields optional.
|
||||
@@ -0,0 +1,27 @@
|
||||
## Why
|
||||
|
||||
The RAG refactor needs a repeatable baseline before changing L0, metadata filtering, post-processing, or Spring AI retriever integration. Without fixed retrieval cases and measurable output, later changes can look cleaner architecturally while silently degrading recall or evidence quality.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add a retrieval evaluation baseline for RAG queries, separate from full diagnosis evaluation.
|
||||
- Define golden retrieval cases covering Chat-style knowledge lookup and AIOps-style alert diagnosis retrieval.
|
||||
- Add a lightweight offline evaluator that compares retrieved candidates against expected documents, breadcrumbs, and evidence keywords.
|
||||
- Preserve baseline JSON and Markdown reports so future changes can compare retrieval behavior.
|
||||
- No production retrieval behavior changes in this change.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
- `rag-retrieval-evaluation`: Defines fixed retrieval golden cases, deterministic retrieval evaluation, and baseline report preservation.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- None.
|
||||
|
||||
## Impact
|
||||
|
||||
- Adds retrieval evaluation fixtures, documentation, and scripts.
|
||||
- May read existing retrieval/tool trace output or saved fixtures, but does not require live LLM calls.
|
||||
- Does not change the `lookup_knowledge` runtime behavior, Milvus schema, document upload API, or Agent flow.
|
||||
+61
@@ -0,0 +1,61 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL define fixed golden cases
|
||||
The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
|
||||
|
||||
#### Scenario: Golden case includes expected retrieval evidence
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
|
||||
|
||||
#### Scenario: Golden case distinguishes scenario type
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL run offline against fixtures
|
||||
The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
|
||||
|
||||
#### Scenario: Fixture evaluation
|
||||
- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
|
||||
- **THEN** it SHALL evaluate each case against its matching fixture file
|
||||
- **AND** it SHALL not call external services
|
||||
|
||||
#### Scenario: Missing fixture is reported
|
||||
- **WHEN** a golden case has no matching retrieval fixture
|
||||
- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL classify hit quality
|
||||
The evaluator SHALL classify each case into a deterministic hit level.
|
||||
|
||||
#### Scenario: Strong hit classification
|
||||
- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `strong`
|
||||
|
||||
#### Scenario: Medium hit classification
|
||||
- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `medium`
|
||||
|
||||
#### Scenario: Miss classification
|
||||
- **WHEN** retrieved candidates do not include expected documents or expected evidence
|
||||
- **THEN** the evaluator SHALL classify the case as `miss`
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL report ranking signals
|
||||
The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
|
||||
|
||||
#### Scenario: Per-case ranking output
|
||||
- **WHEN** a case is evaluated
|
||||
- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
|
||||
|
||||
#### Scenario: Aggregate metrics output
|
||||
- **WHEN** multiple cases are evaluated
|
||||
- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL preserve baseline reports
|
||||
The system SHALL preserve generated baseline reports in JSON and Markdown formats.
|
||||
|
||||
#### Scenario: Baseline report generation
|
||||
- **WHEN** the baseline evaluator is run for the fixed golden case set
|
||||
- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
|
||||
|
||||
#### Scenario: Baseline regeneration is documented
|
||||
- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
|
||||
- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
|
||||
@@ -0,0 +1,26 @@
|
||||
## 1. Golden Cases
|
||||
|
||||
- [x] 1.1 Create retrieval evaluation directory structure.
|
||||
- [x] 1.2 Add fixed golden retrieval cases covering Chat and AIOps retrieval scenarios.
|
||||
- [x] 1.3 Add matching offline retrieval fixtures for every golden case.
|
||||
|
||||
## 2. Evaluator
|
||||
|
||||
- [x] 2.1 Implement an offline retrieval evaluator script.
|
||||
- [x] 2.2 Support hit-level classification and first expected document rank.
|
||||
- [x] 2.3 Support JSON and Markdown report output.
|
||||
|
||||
## 3. Baseline Report
|
||||
|
||||
- [x] 3.1 Generate the baseline JSON report from the fixed cases and fixtures.
|
||||
- [x] 3.2 Generate the baseline Markdown report from the fixed cases and fixtures.
|
||||
|
||||
## 4. Documentation
|
||||
|
||||
- [x] 4.1 Document the retrieval baseline purpose, file layout, and regeneration command.
|
||||
- [x] 4.2 Link the retrieval baseline from the RAG refactor issue or related MVP documentation.
|
||||
|
||||
## 5. Verification
|
||||
|
||||
- [x] 5.1 Run the evaluator successfully against the fixed baseline cases.
|
||||
- [x] 5.2 Run OpenSpec status/validation for the change and confirm tasks are complete.
|
||||
@@ -0,0 +1,64 @@
|
||||
# rag-retrieval-evaluation Specification
|
||||
|
||||
## Purpose
|
||||
Provide a repeatable offline evaluation baseline for RAG retrieval behavior, so L0, query augmentation, evidence post-processing, and vector store changes can be checked against fixed golden retrieval cases before they affect Agent diagnosis quality.
|
||||
## Requirements
|
||||
### Requirement: Retrieval evaluation SHALL define fixed golden cases
|
||||
The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
|
||||
|
||||
#### Scenario: Golden case includes expected retrieval evidence
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
|
||||
|
||||
#### Scenario: Golden case distinguishes scenario type
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL run offline against fixtures
|
||||
The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
|
||||
|
||||
#### Scenario: Fixture evaluation
|
||||
- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
|
||||
- **THEN** it SHALL evaluate each case against its matching fixture file
|
||||
- **AND** it SHALL not call external services
|
||||
|
||||
#### Scenario: Missing fixture is reported
|
||||
- **WHEN** a golden case has no matching retrieval fixture
|
||||
- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL classify hit quality
|
||||
The evaluator SHALL classify each case into a deterministic hit level.
|
||||
|
||||
#### Scenario: Strong hit classification
|
||||
- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `strong`
|
||||
|
||||
#### Scenario: Medium hit classification
|
||||
- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `medium`
|
||||
|
||||
#### Scenario: Miss classification
|
||||
- **WHEN** retrieved candidates do not include expected documents or expected evidence
|
||||
- **THEN** the evaluator SHALL classify the case as `miss`
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL report ranking signals
|
||||
The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
|
||||
|
||||
#### Scenario: Per-case ranking output
|
||||
- **WHEN** a case is evaluated
|
||||
- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
|
||||
|
||||
#### Scenario: Aggregate metrics output
|
||||
- **WHEN** multiple cases are evaluated
|
||||
- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL preserve baseline reports
|
||||
The system SHALL preserve generated baseline reports in JSON and Markdown formats.
|
||||
|
||||
#### Scenario: Baseline report generation
|
||||
- **WHEN** the baseline evaluator is run for the fixed golden case set
|
||||
- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
|
||||
|
||||
#### Scenario: Baseline regeneration is documented
|
||||
- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
|
||||
- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
|
||||
@@ -0,0 +1,290 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Offline evaluator for RAG retrieval golden cases.
|
||||
|
||||
The evaluator reads fixed golden cases and saved retrieval fixtures. It does not
|
||||
call the running application or any external service.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
DEFAULT_CASES = Path("eval/rag-retrieval/cases/golden-cases.json")
|
||||
DEFAULT_FIXTURES = Path("eval/rag-retrieval/fixtures")
|
||||
DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline.json")
|
||||
DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/baseline.md")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Candidate:
|
||||
rank: int
|
||||
doc_id: str
|
||||
title: str
|
||||
breadcrumb: str
|
||||
content: str
|
||||
score: float | None
|
||||
retrieval_layer: str | None
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, raw: dict[str, Any], fallback_rank: int) -> "Candidate":
|
||||
return cls(
|
||||
rank=int(raw.get("rank") or fallback_rank),
|
||||
doc_id=str(raw.get("docId") or raw.get("id") or ""),
|
||||
title=str(raw.get("title") or ""),
|
||||
breadcrumb=str(raw.get("breadcrumb") or ""),
|
||||
content=str(raw.get("content") or ""),
|
||||
score=_optional_float(raw.get("score")),
|
||||
retrieval_layer=(
|
||||
str(raw.get("retrievalLayer"))
|
||||
if raw.get("retrievalLayer") is not None
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
def searchable_text(self) -> str:
|
||||
return " ".join(
|
||||
[self.doc_id, self.title, self.breadcrumb, self.content]
|
||||
).lower()
|
||||
|
||||
def label(self) -> str:
|
||||
label = self.doc_id or self.title or f"rank-{self.rank}"
|
||||
return f"{self.rank}:{label}"
|
||||
|
||||
|
||||
def _optional_float(value: Any) -> float | None:
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def load_json(path: Path) -> Any:
|
||||
with path.open("r", encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def write_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
json.dump(payload, handle, ensure_ascii=False, indent=2)
|
||||
handle.write("\n")
|
||||
|
||||
|
||||
def write_text(path: Path, content: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
handle.write(content)
|
||||
|
||||
|
||||
def normalize_terms(values: list[Any]) -> list[str]:
|
||||
return [str(value).lower() for value in values if str(value).strip()]
|
||||
|
||||
|
||||
def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[str, Any]:
|
||||
case_id = str(case["caseId"])
|
||||
fixture_path = fixture_dir / f"{case_id}.json"
|
||||
expected_doc_ids = normalize_terms(case.get("expectedDocIds", []))
|
||||
expected_breadcrumbs = normalize_terms(case.get("expectedBreadcrumbs", []))
|
||||
expected_keywords = normalize_terms(case.get("expectedKeywords", []))
|
||||
|
||||
if not fixture_path.exists():
|
||||
return {
|
||||
"caseId": case_id,
|
||||
"scenario": case.get("scenario"),
|
||||
"query": case.get("query"),
|
||||
"hitLevel": "miss",
|
||||
"passed": False,
|
||||
"firstExpectedRank": None,
|
||||
"topCandidates": [],
|
||||
"failedChecks": [f"missing fixture: {fixture_path.as_posix()}"],
|
||||
}
|
||||
|
||||
fixture = load_json(fixture_path)
|
||||
raw_candidates = fixture.get("candidates", [])
|
||||
candidates = [
|
||||
Candidate.from_json(raw, index + 1)
|
||||
for index, raw in enumerate(raw_candidates[:top_k])
|
||||
]
|
||||
|
||||
first_expected = None
|
||||
expected_doc_candidate = None
|
||||
for candidate in candidates:
|
||||
candidate_doc = candidate.doc_id.lower()
|
||||
if any(expected == candidate_doc for expected in expected_doc_ids):
|
||||
first_expected = candidate.rank
|
||||
expected_doc_candidate = candidate
|
||||
break
|
||||
|
||||
breadcrumb_match = False
|
||||
keyword_matches: list[str] = []
|
||||
|
||||
if expected_doc_candidate is not None:
|
||||
breadcrumb_text = expected_doc_candidate.breadcrumb.lower()
|
||||
breadcrumb_match = any(
|
||||
expected in breadcrumb_text or breadcrumb_text in expected
|
||||
for expected in expected_breadcrumbs
|
||||
)
|
||||
|
||||
searchable = expected_doc_candidate.searchable_text()
|
||||
keyword_matches = [
|
||||
keyword for keyword in expected_keywords if keyword in searchable
|
||||
]
|
||||
else:
|
||||
all_text = " ".join(candidate.searchable_text() for candidate in candidates)
|
||||
keyword_matches = [keyword for keyword in expected_keywords if keyword in all_text]
|
||||
|
||||
failed_checks: list[str] = []
|
||||
if expected_doc_candidate is None:
|
||||
failed_checks.append("expected document not found")
|
||||
if expected_doc_candidate is not None and expected_breadcrumbs and not breadcrumb_match:
|
||||
failed_checks.append("expected breadcrumb not found on expected document")
|
||||
if expected_keywords and not keyword_matches:
|
||||
failed_checks.append("expected evidence keywords not found")
|
||||
|
||||
if expected_doc_candidate is not None and (
|
||||
breadcrumb_match or bool(keyword_matches)
|
||||
):
|
||||
hit_level = "strong"
|
||||
elif expected_doc_candidate is not None:
|
||||
hit_level = "medium"
|
||||
elif keyword_matches:
|
||||
hit_level = "weak"
|
||||
else:
|
||||
hit_level = "miss"
|
||||
|
||||
return {
|
||||
"caseId": case_id,
|
||||
"scenario": case.get("scenario"),
|
||||
"query": case.get("query"),
|
||||
"hitLevel": hit_level,
|
||||
"passed": hit_level in {"strong", "medium"},
|
||||
"firstExpectedRank": first_expected,
|
||||
"topCandidates": [candidate.label() for candidate in candidates],
|
||||
"matchedKeywords": keyword_matches,
|
||||
"breadcrumbMatched": breadcrumb_match,
|
||||
"failedChecks": failed_checks,
|
||||
}
|
||||
|
||||
|
||||
def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]:
|
||||
total = len(results)
|
||||
counts = {
|
||||
"strong": sum(1 for item in results if item["hitLevel"] == "strong"),
|
||||
"medium": sum(1 for item in results if item["hitLevel"] == "medium"),
|
||||
"weak": sum(1 for item in results if item["hitLevel"] == "weak"),
|
||||
"miss": sum(1 for item in results if item["hitLevel"] == "miss"),
|
||||
}
|
||||
expected_ranks = [
|
||||
item["firstExpectedRank"]
|
||||
for item in results
|
||||
if item.get("firstExpectedRank") is not None
|
||||
]
|
||||
passed = counts["strong"] + counts["medium"]
|
||||
return {
|
||||
"caseCount": total,
|
||||
"topK": top_k,
|
||||
"strongHitCount": counts["strong"],
|
||||
"mediumHitCount": counts["medium"],
|
||||
"weakHitCount": counts["weak"],
|
||||
"missCount": counts["miss"],
|
||||
"recallAtK": round(passed / total, 4) if total else 0,
|
||||
"strongHitRate": round(counts["strong"] / total, 4) if total else 0,
|
||||
"averageFirstHitRank": (
|
||||
round(sum(expected_ranks) / len(expected_ranks), 4)
|
||||
if expected_ranks
|
||||
else None
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def render_markdown(report: dict[str, Any]) -> str:
|
||||
metrics = report["aggregate"]
|
||||
lines = [
|
||||
"# RAG Retrieval Baseline",
|
||||
"",
|
||||
f"Generated at: `{report['generatedAt']}`",
|
||||
"",
|
||||
"## Aggregate",
|
||||
"",
|
||||
"| Metric | Value |",
|
||||
"|---|---:|",
|
||||
f"| Cases | {metrics['caseCount']} |",
|
||||
f"| Top K | {metrics['topK']} |",
|
||||
f"| Recall@K | {metrics['recallAtK']} |",
|
||||
f"| Strong hit rate | {metrics['strongHitRate']} |",
|
||||
f"| Strong hits | {metrics['strongHitCount']} |",
|
||||
f"| Medium hits | {metrics['mediumHitCount']} |",
|
||||
f"| Weak hits | {metrics['weakHitCount']} |",
|
||||
f"| Misses | {metrics['missCount']} |",
|
||||
f"| Average first hit rank | {metrics['averageFirstHitRank']} |",
|
||||
"",
|
||||
"## Cases",
|
||||
"",
|
||||
"| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |",
|
||||
"|---|---|---|---:|---|---|",
|
||||
]
|
||||
for item in report["results"]:
|
||||
failed = "<br>".join(item["failedChecks"]) if item["failedChecks"] else ""
|
||||
top = "<br>".join(item["topCandidates"])
|
||||
first_rank = item["firstExpectedRank"]
|
||||
lines.append(
|
||||
"| {case} | {scenario} | {hit} | {rank} | {top} | {failed} |".format(
|
||||
case=item["caseId"],
|
||||
scenario=item.get("scenario") or "",
|
||||
hit=item["hitLevel"],
|
||||
rank=first_rank if first_rank is not None else "",
|
||||
top=top,
|
||||
failed=failed,
|
||||
)
|
||||
)
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--cases", type=Path, default=DEFAULT_CASES)
|
||||
parser.add_argument("--fixtures", type=Path, default=DEFAULT_FIXTURES)
|
||||
parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT)
|
||||
parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
case_file = load_json(args.cases)
|
||||
cases = case_file.get("cases", [])
|
||||
top_k = int(case_file.get("topK") or 5)
|
||||
results = [evaluate_case(case, args.fixtures, top_k) for case in cases]
|
||||
report = {
|
||||
"generatedAt": datetime.now(timezone.utc).isoformat(),
|
||||
"caseFile": args.cases.as_posix(),
|
||||
"fixtureDir": args.fixtures.as_posix(),
|
||||
"aggregate": aggregate(results, top_k),
|
||||
"results": results,
|
||||
}
|
||||
write_json(args.json_report, report)
|
||||
write_text(args.markdown_report, render_markdown(report))
|
||||
|
||||
failed = [item for item in results if item["hitLevel"] == "miss"]
|
||||
print(
|
||||
"Evaluated {total} cases: recall@{top_k}={recall}, misses={misses}".format(
|
||||
total=len(results),
|
||||
top_k=top_k,
|
||||
recall=report["aggregate"]["recallAtK"],
|
||||
misses=len(failed),
|
||||
)
|
||||
)
|
||||
return 1 if failed else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user