test: add rag retrieval baseline

This commit is contained in:
aruo
2026-07-05 02:02:27 +08:00
parent 79feed3314
commit 9a2a44d1b5
19 changed files with 946 additions and 0 deletions
+51
View File
@@ -0,0 +1,51 @@
# RAG Retrieval Baseline
This directory contains the offline retrieval baseline for the RAG refactor.
The baseline is intentionally narrower than full diagnosis evaluation. It checks
whether fixed retrieval queries can recover expected documents, breadcrumbs, and
evidence keywords before changing L0 behavior, query augmentation, evidence
post-processing, or Spring AI VectorStore integration.
## Layout
```text
eval/rag-retrieval/
cases/golden-cases.json Fixed retrieval golden cases
fixtures/*.json Saved retrieval candidates for each case
reports/baseline.json Machine-readable baseline report
reports/baseline.md Human-readable baseline report
```
## Run
From the repository root:
```bash
python scripts/eval_rag_retrieval.py
```
Custom paths are also supported:
```bash
python scripts/eval_rag_retrieval.py \
--cases eval/rag-retrieval/cases/golden-cases.json \
--fixtures eval/rag-retrieval/fixtures \
--json-report eval/rag-retrieval/reports/baseline.json \
--markdown-report eval/rag-retrieval/reports/baseline.md
```
## Hit Levels
- `strong`: expected document is found and breadcrumb or evidence keyword coverage is satisfied.
- `medium`: expected document is found, but breadcrumb or keyword coverage is incomplete.
- `weak`: expected evidence keyword is found, but expected document is missing.
- `miss`: expected document and expected evidence are not found.
`Recall@K` counts `strong` and `medium` as retrieved.
## Scope
This baseline runs fully offline and does not call MySQL, Redis, Milvus, an LLM,
or the Spring Boot application. It is a regression harness for retrieval behavior,
not a claim that live production retrieval accuracy is complete.
@@ -0,0 +1,61 @@
{
"version": 1,
"description": "Offline golden retrieval cases for RAG refactor baseline.",
"topK": 5,
"cases": [
{
"caseId": "chat-mysql-connection-pool",
"scenario": "chat",
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
"expectedDocIds": ["mysql-connection-pool"],
"expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
"expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
"notes": "Covers precise database troubleshooting retrieval."
},
{
"caseId": "chat-diagnosis-flow",
"scenario": "chat",
"query": "What is the standard troubleshooting flow for an application incident?",
"expectedDocIds": ["incident-diagnosis-flow"],
"expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
"expectedKeywords": ["collect evidence", "verify", "remediation"],
"notes": "Covers process-style knowledge where breadcrumb matters."
},
{
"caseId": "aiops-payment-latency-alert",
"scenario": "aiops",
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
"expectedDocIds": ["payment-service-latency"],
"expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
"expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
"notes": "Covers alert payload terms that should become retrieval hints."
},
{
"caseId": "aiops-prometheus-alert-scope",
"scenario": "aiops",
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
"expectedDocIds": ["aiops-alert-scope-control"],
"expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
"expectedKeywords": ["payload", "unrelated active alerts", "scope"],
"notes": "Covers scoped alert diagnosis behavior."
},
{
"caseId": "chat-rag-chunk-context",
"scenario": "chat",
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
"expectedDocIds": ["rag-chunk-context-reconstruction"],
"expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
"expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
"notes": "Covers the known RAG refactor issue around context reconstruction."
},
{
"caseId": "chat-l0-domain-hint",
"scenario": "chat",
"query": "Should L0 keyword matching decide the final retrieval result?",
"expectedDocIds": ["rag-l0-domain-entity-hint"],
"expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
"expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
"notes": "Covers the target L0 role after refactor."
}
]
}
@@ -0,0 +1,25 @@
{
"caseId": "aiops-payment-latency-alert",
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
"retrievedAt": "2026-07-05T00:00:00Z",
"candidates": [
{
"rank": 1,
"docId": "payment-service-latency",
"title": "Payment Service Latency Alert Playbook",
"breadcrumb": "AIOps > Service Alerts > Payment Latency",
"content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.",
"score": 0.84,
"retrievalLayer": "L1"
},
{
"rank": 2,
"docId": "mysql-connection-pool",
"title": "MySQL Connection Pool Troubleshooting",
"breadcrumb": "Database > MySQL > Connection Pool",
"content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.",
"score": 0.68,
"retrievalLayer": "L1"
}
]
}
@@ -0,0 +1,16 @@
{
"caseId": "aiops-prometheus-alert-scope",
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
"retrievedAt": "2026-07-05T00:00:00Z",
"candidates": [
{
"rank": 1,
"docId": "aiops-alert-scope-control",
"title": "AIOps Alert Scope Control",
"breadcrumb": "AIOps > Alert Scope Control",
"content": "When payload mode is active, queryPrometheusAlerts can verify the supplied alert, but unrelated active alerts must remain scoped context and should not become full diagnoses.",
"score": 0.9,
"retrievalLayer": "L0+L1"
}
]
}
@@ -0,0 +1,25 @@
{
"caseId": "chat-diagnosis-flow",
"query": "What is the standard troubleshooting flow for an application incident?",
"retrievedAt": "2026-07-05T00:00:00Z",
"candidates": [
{
"rank": 1,
"docId": "incident-diagnosis-flow",
"title": "Incident Diagnosis Flow",
"breadcrumb": "AIOps > Diagnosis Flow",
"content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.",
"score": 0.82,
"retrievalLayer": "L1"
},
{
"rank": 2,
"docId": "rag-chunk-context-reconstruction",
"title": "RAG Chunk Context Reconstruction",
"breadcrumb": "RAG > Chunking > Context Reconstruction",
"content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.",
"score": 0.55,
"retrievalLayer": "L1"
}
]
}
@@ -0,0 +1,25 @@
{
"caseId": "chat-l0-domain-hint",
"query": "Should L0 keyword matching decide the final retrieval result?",
"retrievedAt": "2026-07-05T00:00:00Z",
"candidates": [
{
"rank": 1,
"docId": "rag-l0-domain-entity-hint",
"title": "RAG L0 Domain Entity Hint",
"breadcrumb": "RAG > L0 > Domain Entity Hint",
"content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.",
"score": 0.88,
"retrievalLayer": "L0"
},
{
"rank": 2,
"docId": "rag-l0-l1-fusion-ranking",
"title": "RAG L0 L1 Fusion Ranking",
"breadcrumb": "RAG > Ranking > Fusion",
"content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.",
"score": 0.75,
"retrievalLayer": "L1"
}
]
}
@@ -0,0 +1,25 @@
{
"caseId": "chat-mysql-connection-pool",
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
"retrievedAt": "2026-07-05T00:00:00Z",
"candidates": [
{
"rank": 1,
"docId": "mysql-connection-pool",
"title": "MySQL Connection Pool Troubleshooting",
"breadcrumb": "Database > MySQL > Connection Pool",
"content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.",
"score": 0.86,
"retrievalLayer": "L0+L1"
},
{
"rank": 2,
"docId": "incident-diagnosis-flow",
"title": "Incident Diagnosis Flow",
"breadcrumb": "AIOps > Diagnosis Flow",
"content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.",
"score": 0.61,
"retrievalLayer": "L1"
}
]
}
@@ -0,0 +1,25 @@
{
"caseId": "chat-rag-chunk-context",
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
"retrievedAt": "2026-07-05T00:00:00Z",
"candidates": [
{
"rank": 1,
"docId": "rag-chunk-context-reconstruction",
"title": "RAG Chunk Context Reconstruction",
"breadcrumb": "RAG > Chunking > Context Reconstruction",
"content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.",
"score": 0.79,
"retrievalLayer": "L1"
},
{
"rank": 2,
"docId": "rag-breadcrumb-embedding-gap",
"title": "RAG Breadcrumb Embedding Gap",
"breadcrumb": "RAG > Embedding > Breadcrumb",
"content": "Embedding title and breadcrumb with content helps recover section semantics.",
"score": 0.72,
"retrievalLayer": "L1"
}
]
}
+1
View File
@@ -0,0 +1 @@
+131
View File
@@ -0,0 +1,131 @@
{
"generatedAt": "2026-07-04T17:59:52.172759+00:00",
"caseFile": "eval/rag-retrieval/cases/golden-cases.json",
"fixtureDir": "eval/rag-retrieval/fixtures",
"aggregate": {
"caseCount": 6,
"topK": 5,
"strongHitCount": 6,
"mediumHitCount": 0,
"weakHitCount": 0,
"missCount": 0,
"recallAtK": 1.0,
"strongHitRate": 1.0,
"averageFirstHitRank": 1.0
},
"results": [
{
"caseId": "chat-mysql-connection-pool",
"scenario": "chat",
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:mysql-connection-pool",
"2:incident-diagnosis-flow"
],
"matchedKeywords": [
"connection pool",
"max_connections",
"hikaricp"
],
"breadcrumbMatched": true,
"failedChecks": []
},
{
"caseId": "chat-diagnosis-flow",
"scenario": "chat",
"query": "What is the standard troubleshooting flow for an application incident?",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:incident-diagnosis-flow",
"2:rag-chunk-context-reconstruction"
],
"matchedKeywords": [
"collect evidence",
"verify",
"remediation"
],
"breadcrumbMatched": true,
"failedChecks": []
},
{
"caseId": "aiops-payment-latency-alert",
"scenario": "aiops",
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:payment-service-latency",
"2:mysql-connection-pool"
],
"matchedKeywords": [
"p95 latency",
"payment-service",
"downstream dependency"
],
"breadcrumbMatched": true,
"failedChecks": []
},
{
"caseId": "aiops-prometheus-alert-scope",
"scenario": "aiops",
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:aiops-alert-scope-control"
],
"matchedKeywords": [
"payload",
"unrelated active alerts",
"scope"
],
"breadcrumbMatched": true,
"failedChecks": []
},
{
"caseId": "chat-rag-chunk-context",
"scenario": "chat",
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:rag-chunk-context-reconstruction",
"2:rag-breadcrumb-embedding-gap"
],
"matchedKeywords": [
"neighbor chunk",
"same section",
"breadcrumb"
],
"breadcrumbMatched": true,
"failedChecks": []
},
{
"caseId": "chat-l0-domain-hint",
"scenario": "chat",
"query": "Should L0 keyword matching decide the final retrieval result?",
"hitLevel": "strong",
"passed": true,
"firstExpectedRank": 1,
"topCandidates": [
"1:rag-l0-domain-entity-hint",
"2:rag-l0-l1-fusion-ranking"
],
"matchedKeywords": [
"domain detector",
"entity extractor",
"metadata filter"
],
"breadcrumbMatched": true,
"failedChecks": []
}
]
}
+28
View File
@@ -0,0 +1,28 @@
# RAG Retrieval Baseline
Generated at: `2026-07-04T17:59:52.172759+00:00`
## Aggregate
| Metric | Value |
|---|---:|
| Cases | 6 |
| Top K | 5 |
| Recall@K | 1.0 |
| Strong hit rate | 1.0 |
| Strong hits | 6 |
| Medium hits | 0 |
| Weak hits | 0 |
| Misses | 0 |
| Average first hit rank | 1.0 |
## Cases
| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |
|---|---|---|---:|---|---|
| chat-mysql-connection-pool | chat | strong | 1 | 1:mysql-connection-pool<br>2:incident-diagnosis-flow | |
| chat-diagnosis-flow | chat | strong | 1 | 1:incident-diagnosis-flow<br>2:rag-chunk-context-reconstruction | |
| aiops-payment-latency-alert | aiops | strong | 1 | 1:payment-service-latency<br>2:mysql-connection-pool | |
| aiops-prometheus-alert-scope | aiops | strong | 1 | 1:aiops-alert-scope-control | |
| chat-rag-chunk-context | chat | strong | 1 | 1:rag-chunk-context-reconstruction<br>2:rag-breadcrumb-embedding-gap | |
| chat-l0-domain-hint | chat | strong | 1 | 1:rag-l0-domain-entity-hint<br>2:rag-l0-l1-fusion-ranking | |
+1
View File
@@ -202,6 +202,7 @@ relevanceLevel
- 覆盖 Chat 和 AIOps 场景。
- 每条 query 标注 expected doc、breadcrumb、关键 chunk 或 evidence。
- 用当前链路跑一遍,记录 baseline。
- 初始离线基线落在 `eval/rag-retrieval/`,用于后续 change 对比。
验收:
@@ -0,0 +1,2 @@
schema: spec-driven
created: 2026-07-04
@@ -0,0 +1,62 @@
## Context
The repository already has diagnosis-level evaluation specs and trace observability, but the RAG refactor plan needs a narrower retrieval baseline. The upcoming changes will alter L0 responsibilities, query augmentation, evidence post-processing, and eventually the vector store implementation. Those changes need a fixed set of retrieval cases and deterministic scoring before production retrieval behavior changes.
The first baseline must be offline. It should not require MySQL, Milvus, Redis, LLM calls, or a running Spring Boot application. It can evaluate saved retrieval result fixtures that represent the current behavior and produce reports that future changes can compare against.
## Goals / Non-Goals
**Goals:**
- Add a small fixed golden query set for RAG retrieval.
- Evaluate retrieval fixtures against expected documents, breadcrumbs, evidence keywords, and hit levels.
- Produce JSON and Markdown baseline reports.
- Document how to regenerate the reports.
- Keep the evaluator simple enough to run from the repository with Python.
**Non-Goals:**
- Do not change `lookup_knowledge`, `VectorSearchService`, Milvus schema, L0 matching, or Agent prompts.
- Do not require live services.
- Do not implement Spring AI VectorStore migration in this change.
- Do not implement RRF, BM25, rerank, or evidence packing in this change.
## Decisions
### Decision: Use offline retrieval fixtures first
The evaluator will read saved retrieval fixtures rather than calling the live application.
Rationale: the first change should establish a stable measurement surface before the RAG internals change. Live retrieval depends on embeddings, Milvus state, and service configuration, which makes it a poor first baseline.
Alternative considered: call `SearchController` or `lookup_knowledge` directly. That is useful later, but it would require a running app and seeded knowledge base.
### Decision: Score by hit level, not exact chunk id only
The evaluator will classify each case as:
- `strong`: expected document plus expected breadcrumb or key evidence coverage.
- `medium`: expected document found, but breadcrumb or evidence coverage is incomplete.
- `weak`: related evidence is present but the expected document is missing.
- `miss`: no expected document or expected evidence is found.
Rationale: chunk indexes can change after splitter changes, so exact chunk-only scoring would make later refactors look worse even when evidence quality is preserved.
### Decision: Keep case format explicit and reviewable
Golden cases will be stored as JSON with fields such as `caseId`, `query`, `expectedDocIds`, `expectedBreadcrumbs`, `expectedKeywords`, and optional `notes`.
Rationale: the case file should be easy to inspect in code review and easy to extend during interviews or later refactors.
### Decision: Preserve both machine and human reports
The evaluator will write JSON for automation and Markdown for review.
Rationale: future changes can compare JSON, while the Markdown report is easier to use during design review and interview preparation.
## Risks / Trade-offs
- Offline fixtures can drift from real runtime behavior -> add documentation that this is a baseline harness, not a live retrieval accuracy claim.
- Keyword-based evidence checks are approximate -> use them only as deterministic guardrails, not as a replacement for human review.
- Small golden set may underrepresent production queries -> start with 10-20 cases and expand as new RAG issues are found.
- Fixture schema may not match future retrieval outputs -> normalize fixtures into a simple candidate shape and keep raw fields optional.
@@ -0,0 +1,27 @@
## Why
The RAG refactor needs a repeatable baseline before changing L0, metadata filtering, post-processing, or Spring AI retriever integration. Without fixed retrieval cases and measurable output, later changes can look cleaner architecturally while silently degrading recall or evidence quality.
## What Changes
- Add a retrieval evaluation baseline for RAG queries, separate from full diagnosis evaluation.
- Define golden retrieval cases covering Chat-style knowledge lookup and AIOps-style alert diagnosis retrieval.
- Add a lightweight offline evaluator that compares retrieved candidates against expected documents, breadcrumbs, and evidence keywords.
- Preserve baseline JSON and Markdown reports so future changes can compare retrieval behavior.
- No production retrieval behavior changes in this change.
## Capabilities
### New Capabilities
- `rag-retrieval-evaluation`: Defines fixed retrieval golden cases, deterministic retrieval evaluation, and baseline report preservation.
### Modified Capabilities
- None.
## Impact
- Adds retrieval evaluation fixtures, documentation, and scripts.
- May read existing retrieval/tool trace output or saved fixtures, but does not require live LLM calls.
- Does not change the `lookup_knowledge` runtime behavior, Milvus schema, document upload API, or Agent flow.
@@ -0,0 +1,61 @@
## ADDED Requirements
### Requirement: Retrieval evaluation SHALL define fixed golden cases
The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
#### Scenario: Golden case includes expected retrieval evidence
- **WHEN** a retrieval golden case is defined
- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
#### Scenario: Golden case distinguishes scenario type
- **WHEN** a retrieval golden case is defined
- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
### Requirement: Retrieval evaluation SHALL run offline against fixtures
The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
#### Scenario: Fixture evaluation
- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
- **THEN** it SHALL evaluate each case against its matching fixture file
- **AND** it SHALL not call external services
#### Scenario: Missing fixture is reported
- **WHEN** a golden case has no matching retrieval fixture
- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
### Requirement: Retrieval evaluation SHALL classify hit quality
The evaluator SHALL classify each case into a deterministic hit level.
#### Scenario: Strong hit classification
- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
- **THEN** the evaluator SHALL classify the case as `strong`
#### Scenario: Medium hit classification
- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
- **THEN** the evaluator SHALL classify the case as `medium`
#### Scenario: Miss classification
- **WHEN** retrieved candidates do not include expected documents or expected evidence
- **THEN** the evaluator SHALL classify the case as `miss`
### Requirement: Retrieval evaluation SHALL report ranking signals
The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
#### Scenario: Per-case ranking output
- **WHEN** a case is evaluated
- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
#### Scenario: Aggregate metrics output
- **WHEN** multiple cases are evaluated
- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
### Requirement: Retrieval evaluation SHALL preserve baseline reports
The system SHALL preserve generated baseline reports in JSON and Markdown formats.
#### Scenario: Baseline report generation
- **WHEN** the baseline evaluator is run for the fixed golden case set
- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
#### Scenario: Baseline regeneration is documented
- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
@@ -0,0 +1,26 @@
## 1. Golden Cases
- [x] 1.1 Create retrieval evaluation directory structure.
- [x] 1.2 Add fixed golden retrieval cases covering Chat and AIOps retrieval scenarios.
- [x] 1.3 Add matching offline retrieval fixtures for every golden case.
## 2. Evaluator
- [x] 2.1 Implement an offline retrieval evaluator script.
- [x] 2.2 Support hit-level classification and first expected document rank.
- [x] 2.3 Support JSON and Markdown report output.
## 3. Baseline Report
- [x] 3.1 Generate the baseline JSON report from the fixed cases and fixtures.
- [x] 3.2 Generate the baseline Markdown report from the fixed cases and fixtures.
## 4. Documentation
- [x] 4.1 Document the retrieval baseline purpose, file layout, and regeneration command.
- [x] 4.2 Link the retrieval baseline from the RAG refactor issue or related MVP documentation.
## 5. Verification
- [x] 5.1 Run the evaluator successfully against the fixed baseline cases.
- [x] 5.2 Run OpenSpec status/validation for the change and confirm tasks are complete.
@@ -0,0 +1,64 @@
# rag-retrieval-evaluation Specification
## Purpose
Provide a repeatable offline evaluation baseline for RAG retrieval behavior, so L0, query augmentation, evidence post-processing, and vector store changes can be checked against fixed golden retrieval cases before they affect Agent diagnosis quality.
## Requirements
### Requirement: Retrieval evaluation SHALL define fixed golden cases
The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
#### Scenario: Golden case includes expected retrieval evidence
- **WHEN** a retrieval golden case is defined
- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
#### Scenario: Golden case distinguishes scenario type
- **WHEN** a retrieval golden case is defined
- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
### Requirement: Retrieval evaluation SHALL run offline against fixtures
The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
#### Scenario: Fixture evaluation
- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
- **THEN** it SHALL evaluate each case against its matching fixture file
- **AND** it SHALL not call external services
#### Scenario: Missing fixture is reported
- **WHEN** a golden case has no matching retrieval fixture
- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
### Requirement: Retrieval evaluation SHALL classify hit quality
The evaluator SHALL classify each case into a deterministic hit level.
#### Scenario: Strong hit classification
- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
- **THEN** the evaluator SHALL classify the case as `strong`
#### Scenario: Medium hit classification
- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
- **THEN** the evaluator SHALL classify the case as `medium`
#### Scenario: Miss classification
- **WHEN** retrieved candidates do not include expected documents or expected evidence
- **THEN** the evaluator SHALL classify the case as `miss`
### Requirement: Retrieval evaluation SHALL report ranking signals
The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
#### Scenario: Per-case ranking output
- **WHEN** a case is evaluated
- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
#### Scenario: Aggregate metrics output
- **WHEN** multiple cases are evaluated
- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
### Requirement: Retrieval evaluation SHALL preserve baseline reports
The system SHALL preserve generated baseline reports in JSON and Markdown formats.
#### Scenario: Baseline report generation
- **WHEN** the baseline evaluator is run for the fixed golden case set
- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
#### Scenario: Baseline regeneration is documented
- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
+290
View File
@@ -0,0 +1,290 @@
#!/usr/bin/env python3
"""Offline evaluator for RAG retrieval golden cases.
The evaluator reads fixed golden cases and saved retrieval fixtures. It does not
call the running application or any external service.
"""
from __future__ import annotations
import argparse
import json
from dataclasses import dataclass
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
DEFAULT_CASES = Path("eval/rag-retrieval/cases/golden-cases.json")
DEFAULT_FIXTURES = Path("eval/rag-retrieval/fixtures")
DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline.json")
DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/baseline.md")
@dataclass
class Candidate:
rank: int
doc_id: str
title: str
breadcrumb: str
content: str
score: float | None
retrieval_layer: str | None
@classmethod
def from_json(cls, raw: dict[str, Any], fallback_rank: int) -> "Candidate":
return cls(
rank=int(raw.get("rank") or fallback_rank),
doc_id=str(raw.get("docId") or raw.get("id") or ""),
title=str(raw.get("title") or ""),
breadcrumb=str(raw.get("breadcrumb") or ""),
content=str(raw.get("content") or ""),
score=_optional_float(raw.get("score")),
retrieval_layer=(
str(raw.get("retrievalLayer"))
if raw.get("retrievalLayer") is not None
else None
),
)
def searchable_text(self) -> str:
return " ".join(
[self.doc_id, self.title, self.breadcrumb, self.content]
).lower()
def label(self) -> str:
label = self.doc_id or self.title or f"rank-{self.rank}"
return f"{self.rank}:{label}"
def _optional_float(value: Any) -> float | None:
if value is None:
return None
try:
return float(value)
except (TypeError, ValueError):
return None
def load_json(path: Path) -> Any:
with path.open("r", encoding="utf-8") as handle:
return json.load(handle)
def write_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8", newline="\n") as handle:
json.dump(payload, handle, ensure_ascii=False, indent=2)
handle.write("\n")
def write_text(path: Path, content: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8", newline="\n") as handle:
handle.write(content)
def normalize_terms(values: list[Any]) -> list[str]:
return [str(value).lower() for value in values if str(value).strip()]
def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[str, Any]:
case_id = str(case["caseId"])
fixture_path = fixture_dir / f"{case_id}.json"
expected_doc_ids = normalize_terms(case.get("expectedDocIds", []))
expected_breadcrumbs = normalize_terms(case.get("expectedBreadcrumbs", []))
expected_keywords = normalize_terms(case.get("expectedKeywords", []))
if not fixture_path.exists():
return {
"caseId": case_id,
"scenario": case.get("scenario"),
"query": case.get("query"),
"hitLevel": "miss",
"passed": False,
"firstExpectedRank": None,
"topCandidates": [],
"failedChecks": [f"missing fixture: {fixture_path.as_posix()}"],
}
fixture = load_json(fixture_path)
raw_candidates = fixture.get("candidates", [])
candidates = [
Candidate.from_json(raw, index + 1)
for index, raw in enumerate(raw_candidates[:top_k])
]
first_expected = None
expected_doc_candidate = None
for candidate in candidates:
candidate_doc = candidate.doc_id.lower()
if any(expected == candidate_doc for expected in expected_doc_ids):
first_expected = candidate.rank
expected_doc_candidate = candidate
break
breadcrumb_match = False
keyword_matches: list[str] = []
if expected_doc_candidate is not None:
breadcrumb_text = expected_doc_candidate.breadcrumb.lower()
breadcrumb_match = any(
expected in breadcrumb_text or breadcrumb_text in expected
for expected in expected_breadcrumbs
)
searchable = expected_doc_candidate.searchable_text()
keyword_matches = [
keyword for keyword in expected_keywords if keyword in searchable
]
else:
all_text = " ".join(candidate.searchable_text() for candidate in candidates)
keyword_matches = [keyword for keyword in expected_keywords if keyword in all_text]
failed_checks: list[str] = []
if expected_doc_candidate is None:
failed_checks.append("expected document not found")
if expected_doc_candidate is not None and expected_breadcrumbs and not breadcrumb_match:
failed_checks.append("expected breadcrumb not found on expected document")
if expected_keywords and not keyword_matches:
failed_checks.append("expected evidence keywords not found")
if expected_doc_candidate is not None and (
breadcrumb_match or bool(keyword_matches)
):
hit_level = "strong"
elif expected_doc_candidate is not None:
hit_level = "medium"
elif keyword_matches:
hit_level = "weak"
else:
hit_level = "miss"
return {
"caseId": case_id,
"scenario": case.get("scenario"),
"query": case.get("query"),
"hitLevel": hit_level,
"passed": hit_level in {"strong", "medium"},
"firstExpectedRank": first_expected,
"topCandidates": [candidate.label() for candidate in candidates],
"matchedKeywords": keyword_matches,
"breadcrumbMatched": breadcrumb_match,
"failedChecks": failed_checks,
}
def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]:
total = len(results)
counts = {
"strong": sum(1 for item in results if item["hitLevel"] == "strong"),
"medium": sum(1 for item in results if item["hitLevel"] == "medium"),
"weak": sum(1 for item in results if item["hitLevel"] == "weak"),
"miss": sum(1 for item in results if item["hitLevel"] == "miss"),
}
expected_ranks = [
item["firstExpectedRank"]
for item in results
if item.get("firstExpectedRank") is not None
]
passed = counts["strong"] + counts["medium"]
return {
"caseCount": total,
"topK": top_k,
"strongHitCount": counts["strong"],
"mediumHitCount": counts["medium"],
"weakHitCount": counts["weak"],
"missCount": counts["miss"],
"recallAtK": round(passed / total, 4) if total else 0,
"strongHitRate": round(counts["strong"] / total, 4) if total else 0,
"averageFirstHitRank": (
round(sum(expected_ranks) / len(expected_ranks), 4)
if expected_ranks
else None
),
}
def render_markdown(report: dict[str, Any]) -> str:
metrics = report["aggregate"]
lines = [
"# RAG Retrieval Baseline",
"",
f"Generated at: `{report['generatedAt']}`",
"",
"## Aggregate",
"",
"| Metric | Value |",
"|---|---:|",
f"| Cases | {metrics['caseCount']} |",
f"| Top K | {metrics['topK']} |",
f"| Recall@K | {metrics['recallAtK']} |",
f"| Strong hit rate | {metrics['strongHitRate']} |",
f"| Strong hits | {metrics['strongHitCount']} |",
f"| Medium hits | {metrics['mediumHitCount']} |",
f"| Weak hits | {metrics['weakHitCount']} |",
f"| Misses | {metrics['missCount']} |",
f"| Average first hit rank | {metrics['averageFirstHitRank']} |",
"",
"## Cases",
"",
"| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |",
"|---|---|---|---:|---|---|",
]
for item in report["results"]:
failed = "<br>".join(item["failedChecks"]) if item["failedChecks"] else ""
top = "<br>".join(item["topCandidates"])
first_rank = item["firstExpectedRank"]
lines.append(
"| {case} | {scenario} | {hit} | {rank} | {top} | {failed} |".format(
case=item["caseId"],
scenario=item.get("scenario") or "",
hit=item["hitLevel"],
rank=first_rank if first_rank is not None else "",
top=top,
failed=failed,
)
)
lines.append("")
return "\n".join(lines)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--cases", type=Path, default=DEFAULT_CASES)
parser.add_argument("--fixtures", type=Path, default=DEFAULT_FIXTURES)
parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT)
parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT)
return parser.parse_args()
def main() -> int:
args = parse_args()
case_file = load_json(args.cases)
cases = case_file.get("cases", [])
top_k = int(case_file.get("topK") or 5)
results = [evaluate_case(case, args.fixtures, top_k) for case in cases]
report = {
"generatedAt": datetime.now(timezone.utc).isoformat(),
"caseFile": args.cases.as_posix(),
"fixtureDir": args.fixtures.as_posix(),
"aggregate": aggregate(results, top_k),
"results": results,
}
write_json(args.json_report, report)
write_text(args.markdown_report, render_markdown(report))
failed = [item for item in results if item["hitLevel"] == "miss"]
print(
"Evaluated {total} cases: recall@{top_k}={recall}, misses={misses}".format(
total=len(results),
top_k=top_k,
recall=report["aggregate"]["recallAtK"],
misses=len(failed),
)
)
return 1 if failed else 0
if __name__ == "__main__":
raise SystemExit(main())