diff --git a/eval/rag-retrieval/README.md b/eval/rag-retrieval/README.md new file mode 100644 index 0000000..f85cb02 --- /dev/null +++ b/eval/rag-retrieval/README.md @@ -0,0 +1,51 @@ +# RAG Retrieval Baseline + +This directory contains the offline retrieval baseline for the RAG refactor. + +The baseline is intentionally narrower than full diagnosis evaluation. It checks +whether fixed retrieval queries can recover expected documents, breadcrumbs, and +evidence keywords before changing L0 behavior, query augmentation, evidence +post-processing, or Spring AI VectorStore integration. + +## Layout + +```text +eval/rag-retrieval/ + cases/golden-cases.json Fixed retrieval golden cases + fixtures/*.json Saved retrieval candidates for each case + reports/baseline.json Machine-readable baseline report + reports/baseline.md Human-readable baseline report +``` + +## Run + +From the repository root: + +```bash +python scripts/eval_rag_retrieval.py +``` + +Custom paths are also supported: + +```bash +python scripts/eval_rag_retrieval.py \ + --cases eval/rag-retrieval/cases/golden-cases.json \ + --fixtures eval/rag-retrieval/fixtures \ + --json-report eval/rag-retrieval/reports/baseline.json \ + --markdown-report eval/rag-retrieval/reports/baseline.md +``` + +## Hit Levels + +- `strong`: expected document is found and breadcrumb or evidence keyword coverage is satisfied. +- `medium`: expected document is found, but breadcrumb or keyword coverage is incomplete. +- `weak`: expected evidence keyword is found, but expected document is missing. +- `miss`: expected document and expected evidence are not found. + +`Recall@K` counts `strong` and `medium` as retrieved. + +## Scope + +This baseline runs fully offline and does not call MySQL, Redis, Milvus, an LLM, +or the Spring Boot application. It is a regression harness for retrieval behavior, +not a claim that live production retrieval accuracy is complete. diff --git a/eval/rag-retrieval/cases/golden-cases.json b/eval/rag-retrieval/cases/golden-cases.json new file mode 100644 index 0000000..abbd785 --- /dev/null +++ b/eval/rag-retrieval/cases/golden-cases.json @@ -0,0 +1,61 @@ +{ + "version": 1, + "description": "Offline golden retrieval cases for RAG refactor baseline.", + "topK": 5, + "cases": [ + { + "caseId": "chat-mysql-connection-pool", + "scenario": "chat", + "query": "MySQL connection pool is exhausted. How should I diagnose it?", + "expectedDocIds": ["mysql-connection-pool"], + "expectedBreadcrumbs": ["Database > MySQL > Connection Pool"], + "expectedKeywords": ["connection pool", "max_connections", "HikariCP"], + "notes": "Covers precise database troubleshooting retrieval." + }, + { + "caseId": "chat-diagnosis-flow", + "scenario": "chat", + "query": "What is the standard troubleshooting flow for an application incident?", + "expectedDocIds": ["incident-diagnosis-flow"], + "expectedBreadcrumbs": ["AIOps > Diagnosis Flow"], + "expectedKeywords": ["collect evidence", "verify", "remediation"], + "notes": "Covers process-style knowledge where breadcrumb matters." + }, + { + "caseId": "aiops-payment-latency-alert", + "scenario": "aiops", + "query": "Alert HighLatency on payment-service with p95 latency above threshold", + "expectedDocIds": ["payment-service-latency"], + "expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"], + "expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"], + "notes": "Covers alert payload terms that should become retrieval hints." + }, + { + "caseId": "aiops-prometheus-alert-scope", + "scenario": "aiops", + "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", + "expectedDocIds": ["aiops-alert-scope-control"], + "expectedBreadcrumbs": ["AIOps > Alert Scope Control"], + "expectedKeywords": ["payload", "unrelated active alerts", "scope"], + "notes": "Covers scoped alert diagnosis behavior." + }, + { + "caseId": "chat-rag-chunk-context", + "scenario": "chat", + "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", + "expectedDocIds": ["rag-chunk-context-reconstruction"], + "expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"], + "expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"], + "notes": "Covers the known RAG refactor issue around context reconstruction." + }, + { + "caseId": "chat-l0-domain-hint", + "scenario": "chat", + "query": "Should L0 keyword matching decide the final retrieval result?", + "expectedDocIds": ["rag-l0-domain-entity-hint"], + "expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"], + "expectedKeywords": ["domain detector", "entity extractor", "metadata filter"], + "notes": "Covers the target L0 role after refactor." + } + ] +} diff --git a/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json b/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json new file mode 100644 index 0000000..c35edfa --- /dev/null +++ b/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json @@ -0,0 +1,25 @@ +{ + "caseId": "aiops-payment-latency-alert", + "query": "Alert HighLatency on payment-service with p95 latency above threshold", + "retrievedAt": "2026-07-05T00:00:00Z", + "candidates": [ + { + "rank": 1, + "docId": "payment-service-latency", + "title": "Payment Service Latency Alert Playbook", + "breadcrumb": "AIOps > Service Alerts > Payment Latency", + "content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.", + "score": 0.84, + "retrievalLayer": "L1" + }, + { + "rank": 2, + "docId": "mysql-connection-pool", + "title": "MySQL Connection Pool Troubleshooting", + "breadcrumb": "Database > MySQL > Connection Pool", + "content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.", + "score": 0.68, + "retrievalLayer": "L1" + } + ] +} diff --git a/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json b/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json new file mode 100644 index 0000000..6603515 --- /dev/null +++ b/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json @@ -0,0 +1,16 @@ +{ + "caseId": "aiops-prometheus-alert-scope", + "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", + "retrievedAt": "2026-07-05T00:00:00Z", + "candidates": [ + { + "rank": 1, + "docId": "aiops-alert-scope-control", + "title": "AIOps Alert Scope Control", + "breadcrumb": "AIOps > Alert Scope Control", + "content": "When payload mode is active, queryPrometheusAlerts can verify the supplied alert, but unrelated active alerts must remain scoped context and should not become full diagnoses.", + "score": 0.9, + "retrievalLayer": "L0+L1" + } + ] +} diff --git a/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json b/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json new file mode 100644 index 0000000..7229583 --- /dev/null +++ b/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json @@ -0,0 +1,25 @@ +{ + "caseId": "chat-diagnosis-flow", + "query": "What is the standard troubleshooting flow for an application incident?", + "retrievedAt": "2026-07-05T00:00:00Z", + "candidates": [ + { + "rank": 1, + "docId": "incident-diagnosis-flow", + "title": "Incident Diagnosis Flow", + "breadcrumb": "AIOps > Diagnosis Flow", + "content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.", + "score": 0.82, + "retrievalLayer": "L1" + }, + { + "rank": 2, + "docId": "rag-chunk-context-reconstruction", + "title": "RAG Chunk Context Reconstruction", + "breadcrumb": "RAG > Chunking > Context Reconstruction", + "content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.", + "score": 0.55, + "retrievalLayer": "L1" + } + ] +} diff --git a/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json b/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json new file mode 100644 index 0000000..2e24c15 --- /dev/null +++ b/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json @@ -0,0 +1,25 @@ +{ + "caseId": "chat-l0-domain-hint", + "query": "Should L0 keyword matching decide the final retrieval result?", + "retrievedAt": "2026-07-05T00:00:00Z", + "candidates": [ + { + "rank": 1, + "docId": "rag-l0-domain-entity-hint", + "title": "RAG L0 Domain Entity Hint", + "breadcrumb": "RAG > L0 > Domain Entity Hint", + "content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.", + "score": 0.88, + "retrievalLayer": "L0" + }, + { + "rank": 2, + "docId": "rag-l0-l1-fusion-ranking", + "title": "RAG L0 L1 Fusion Ranking", + "breadcrumb": "RAG > Ranking > Fusion", + "content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.", + "score": 0.75, + "retrievalLayer": "L1" + } + ] +} diff --git a/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json b/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json new file mode 100644 index 0000000..954306b --- /dev/null +++ b/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json @@ -0,0 +1,25 @@ +{ + "caseId": "chat-mysql-connection-pool", + "query": "MySQL connection pool is exhausted. How should I diagnose it?", + "retrievedAt": "2026-07-05T00:00:00Z", + "candidates": [ + { + "rank": 1, + "docId": "mysql-connection-pool", + "title": "MySQL Connection Pool Troubleshooting", + "breadcrumb": "Database > MySQL > Connection Pool", + "content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.", + "score": 0.86, + "retrievalLayer": "L0+L1" + }, + { + "rank": 2, + "docId": "incident-diagnosis-flow", + "title": "Incident Diagnosis Flow", + "breadcrumb": "AIOps > Diagnosis Flow", + "content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.", + "score": 0.61, + "retrievalLayer": "L1" + } + ] +} diff --git a/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json b/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json new file mode 100644 index 0000000..5e9fcc9 --- /dev/null +++ b/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json @@ -0,0 +1,25 @@ +{ + "caseId": "chat-rag-chunk-context", + "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", + "retrievedAt": "2026-07-05T00:00:00Z", + "candidates": [ + { + "rank": 1, + "docId": "rag-chunk-context-reconstruction", + "title": "RAG Chunk Context Reconstruction", + "breadcrumb": "RAG > Chunking > Context Reconstruction", + "content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.", + "score": 0.79, + "retrievalLayer": "L1" + }, + { + "rank": 2, + "docId": "rag-breadcrumb-embedding-gap", + "title": "RAG Breadcrumb Embedding Gap", + "breadcrumb": "RAG > Embedding > Breadcrumb", + "content": "Embedding title and breadcrumb with content helps recover section semantics.", + "score": 0.72, + "retrievalLayer": "L1" + } + ] +} diff --git a/eval/rag-retrieval/reports/.gitkeep b/eval/rag-retrieval/reports/.gitkeep new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/eval/rag-retrieval/reports/.gitkeep @@ -0,0 +1 @@ + diff --git a/eval/rag-retrieval/reports/baseline.json b/eval/rag-retrieval/reports/baseline.json new file mode 100644 index 0000000..a4eaa8e --- /dev/null +++ b/eval/rag-retrieval/reports/baseline.json @@ -0,0 +1,131 @@ +{ + "generatedAt": "2026-07-04T17:59:52.172759+00:00", + "caseFile": "eval/rag-retrieval/cases/golden-cases.json", + "fixtureDir": "eval/rag-retrieval/fixtures", + "aggregate": { + "caseCount": 6, + "topK": 5, + "strongHitCount": 6, + "mediumHitCount": 0, + "weakHitCount": 0, + "missCount": 0, + "recallAtK": 1.0, + "strongHitRate": 1.0, + "averageFirstHitRank": 1.0 + }, + "results": [ + { + "caseId": "chat-mysql-connection-pool", + "scenario": "chat", + "query": "MySQL connection pool is exhausted. How should I diagnose it?", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:mysql-connection-pool", + "2:incident-diagnosis-flow" + ], + "matchedKeywords": [ + "connection pool", + "max_connections", + "hikaricp" + ], + "breadcrumbMatched": true, + "failedChecks": [] + }, + { + "caseId": "chat-diagnosis-flow", + "scenario": "chat", + "query": "What is the standard troubleshooting flow for an application incident?", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:incident-diagnosis-flow", + "2:rag-chunk-context-reconstruction" + ], + "matchedKeywords": [ + "collect evidence", + "verify", + "remediation" + ], + "breadcrumbMatched": true, + "failedChecks": [] + }, + { + "caseId": "aiops-payment-latency-alert", + "scenario": "aiops", + "query": "Alert HighLatency on payment-service with p95 latency above threshold", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:payment-service-latency", + "2:mysql-connection-pool" + ], + "matchedKeywords": [ + "p95 latency", + "payment-service", + "downstream dependency" + ], + "breadcrumbMatched": true, + "failedChecks": [] + }, + { + "caseId": "aiops-prometheus-alert-scope", + "scenario": "aiops", + "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:aiops-alert-scope-control" + ], + "matchedKeywords": [ + "payload", + "unrelated active alerts", + "scope" + ], + "breadcrumbMatched": true, + "failedChecks": [] + }, + { + "caseId": "chat-rag-chunk-context", + "scenario": "chat", + "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:rag-chunk-context-reconstruction", + "2:rag-breadcrumb-embedding-gap" + ], + "matchedKeywords": [ + "neighbor chunk", + "same section", + "breadcrumb" + ], + "breadcrumbMatched": true, + "failedChecks": [] + }, + { + "caseId": "chat-l0-domain-hint", + "scenario": "chat", + "query": "Should L0 keyword matching decide the final retrieval result?", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:rag-l0-domain-entity-hint", + "2:rag-l0-l1-fusion-ranking" + ], + "matchedKeywords": [ + "domain detector", + "entity extractor", + "metadata filter" + ], + "breadcrumbMatched": true, + "failedChecks": [] + } + ] +} diff --git a/eval/rag-retrieval/reports/baseline.md b/eval/rag-retrieval/reports/baseline.md new file mode 100644 index 0000000..7b9a403 --- /dev/null +++ b/eval/rag-retrieval/reports/baseline.md @@ -0,0 +1,28 @@ +# RAG Retrieval Baseline + +Generated at: `2026-07-04T17:59:52.172759+00:00` + +## Aggregate + +| Metric | Value | +|---|---:| +| Cases | 6 | +| Top K | 5 | +| Recall@K | 1.0 | +| Strong hit rate | 1.0 | +| Strong hits | 6 | +| Medium hits | 0 | +| Weak hits | 0 | +| Misses | 0 | +| Average first hit rank | 1.0 | + +## Cases + +| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks | +|---|---|---|---:|---|---| +| chat-mysql-connection-pool | chat | strong | 1 | 1:mysql-connection-pool
2:incident-diagnosis-flow | | +| chat-diagnosis-flow | chat | strong | 1 | 1:incident-diagnosis-flow
2:rag-chunk-context-reconstruction | | +| aiops-payment-latency-alert | aiops | strong | 1 | 1:payment-service-latency
2:mysql-connection-pool | | +| aiops-prometheus-alert-scope | aiops | strong | 1 | 1:aiops-alert-scope-control | | +| chat-rag-chunk-context | chat | strong | 1 | 1:rag-chunk-context-reconstruction
2:rag-breadcrumb-embedding-gap | | +| chat-l0-domain-hint | chat | strong | 1 | 1:rag-l0-domain-entity-hint
2:rag-l0-l1-fusion-ranking | | diff --git a/mvp/issues/rag-refactor-plan.md b/mvp/issues/rag-refactor-plan.md index edde42a..24eef56 100644 --- a/mvp/issues/rag-refactor-plan.md +++ b/mvp/issues/rag-refactor-plan.md @@ -202,6 +202,7 @@ relevanceLevel - 覆盖 Chat 和 AIOps 场景。 - 每条 query 标注 expected doc、breadcrumb、关键 chunk 或 evidence。 - 用当前链路跑一遍,记录 baseline。 +- 初始离线基线落在 `eval/rag-retrieval/`,用于后续 change 对比。 验收: diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/.openspec.yaml b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/.openspec.yaml new file mode 100644 index 0000000..d86f152 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-04 diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/design.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/design.md new file mode 100644 index 0000000..c63669a --- /dev/null +++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/design.md @@ -0,0 +1,62 @@ +## Context + +The repository already has diagnosis-level evaluation specs and trace observability, but the RAG refactor plan needs a narrower retrieval baseline. The upcoming changes will alter L0 responsibilities, query augmentation, evidence post-processing, and eventually the vector store implementation. Those changes need a fixed set of retrieval cases and deterministic scoring before production retrieval behavior changes. + +The first baseline must be offline. It should not require MySQL, Milvus, Redis, LLM calls, or a running Spring Boot application. It can evaluate saved retrieval result fixtures that represent the current behavior and produce reports that future changes can compare against. + +## Goals / Non-Goals + +**Goals:** + +- Add a small fixed golden query set for RAG retrieval. +- Evaluate retrieval fixtures against expected documents, breadcrumbs, evidence keywords, and hit levels. +- Produce JSON and Markdown baseline reports. +- Document how to regenerate the reports. +- Keep the evaluator simple enough to run from the repository with Python. + +**Non-Goals:** + +- Do not change `lookup_knowledge`, `VectorSearchService`, Milvus schema, L0 matching, or Agent prompts. +- Do not require live services. +- Do not implement Spring AI VectorStore migration in this change. +- Do not implement RRF, BM25, rerank, or evidence packing in this change. + +## Decisions + +### Decision: Use offline retrieval fixtures first + +The evaluator will read saved retrieval fixtures rather than calling the live application. + +Rationale: the first change should establish a stable measurement surface before the RAG internals change. Live retrieval depends on embeddings, Milvus state, and service configuration, which makes it a poor first baseline. + +Alternative considered: call `SearchController` or `lookup_knowledge` directly. That is useful later, but it would require a running app and seeded knowledge base. + +### Decision: Score by hit level, not exact chunk id only + +The evaluator will classify each case as: + +- `strong`: expected document plus expected breadcrumb or key evidence coverage. +- `medium`: expected document found, but breadcrumb or evidence coverage is incomplete. +- `weak`: related evidence is present but the expected document is missing. +- `miss`: no expected document or expected evidence is found. + +Rationale: chunk indexes can change after splitter changes, so exact chunk-only scoring would make later refactors look worse even when evidence quality is preserved. + +### Decision: Keep case format explicit and reviewable + +Golden cases will be stored as JSON with fields such as `caseId`, `query`, `expectedDocIds`, `expectedBreadcrumbs`, `expectedKeywords`, and optional `notes`. + +Rationale: the case file should be easy to inspect in code review and easy to extend during interviews or later refactors. + +### Decision: Preserve both machine and human reports + +The evaluator will write JSON for automation and Markdown for review. + +Rationale: future changes can compare JSON, while the Markdown report is easier to use during design review and interview preparation. + +## Risks / Trade-offs + +- Offline fixtures can drift from real runtime behavior -> add documentation that this is a baseline harness, not a live retrieval accuracy claim. +- Keyword-based evidence checks are approximate -> use them only as deterministic guardrails, not as a replacement for human review. +- Small golden set may underrepresent production queries -> start with 10-20 cases and expand as new RAG issues are found. +- Fixture schema may not match future retrieval outputs -> normalize fixtures into a simple candidate shape and keep raw fields optional. diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/proposal.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/proposal.md new file mode 100644 index 0000000..572fe71 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/proposal.md @@ -0,0 +1,27 @@ +## Why + +The RAG refactor needs a repeatable baseline before changing L0, metadata filtering, post-processing, or Spring AI retriever integration. Without fixed retrieval cases and measurable output, later changes can look cleaner architecturally while silently degrading recall or evidence quality. + +## What Changes + +- Add a retrieval evaluation baseline for RAG queries, separate from full diagnosis evaluation. +- Define golden retrieval cases covering Chat-style knowledge lookup and AIOps-style alert diagnosis retrieval. +- Add a lightweight offline evaluator that compares retrieved candidates against expected documents, breadcrumbs, and evidence keywords. +- Preserve baseline JSON and Markdown reports so future changes can compare retrieval behavior. +- No production retrieval behavior changes in this change. + +## Capabilities + +### New Capabilities + +- `rag-retrieval-evaluation`: Defines fixed retrieval golden cases, deterministic retrieval evaluation, and baseline report preservation. + +### Modified Capabilities + +- None. + +## Impact + +- Adds retrieval evaluation fixtures, documentation, and scripts. +- May read existing retrieval/tool trace output or saved fixtures, but does not require live LLM calls. +- Does not change the `lookup_knowledge` runtime behavior, Milvus schema, document upload API, or Agent flow. diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/specs/rag-retrieval-evaluation/spec.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/specs/rag-retrieval-evaluation/spec.md new file mode 100644 index 0000000..3d053f2 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/specs/rag-retrieval-evaluation/spec.md @@ -0,0 +1,61 @@ +## ADDED Requirements + +### Requirement: Retrieval evaluation SHALL define fixed golden cases +The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically. + +#### Scenario: Golden case includes expected retrieval evidence +- **WHEN** a retrieval golden case is defined +- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs + +#### Scenario: Golden case distinguishes scenario type +- **WHEN** a retrieval golden case is defined +- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type + +### Requirement: Retrieval evaluation SHALL run offline against fixtures +The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services. + +#### Scenario: Fixture evaluation +- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory +- **THEN** it SHALL evaluate each case against its matching fixture file +- **AND** it SHALL not call external services + +#### Scenario: Missing fixture is reported +- **WHEN** a golden case has no matching retrieval fixture +- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason + +### Requirement: Retrieval evaluation SHALL classify hit quality +The evaluator SHALL classify each case into a deterministic hit level. + +#### Scenario: Strong hit classification +- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage +- **THEN** the evaluator SHALL classify the case as `strong` + +#### Scenario: Medium hit classification +- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage +- **THEN** the evaluator SHALL classify the case as `medium` + +#### Scenario: Miss classification +- **WHEN** retrieved candidates do not include expected documents or expected evidence +- **THEN** the evaluator SHALL classify the case as `miss` + +### Requirement: Retrieval evaluation SHALL report ranking signals +The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks. + +#### Scenario: Per-case ranking output +- **WHEN** a case is evaluated +- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks + +#### Scenario: Aggregate metrics output +- **WHEN** multiple cases are evaluated +- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available + +### Requirement: Retrieval evaluation SHALL preserve baseline reports +The system SHALL preserve generated baseline reports in JSON and Markdown formats. + +#### Scenario: Baseline report generation +- **WHEN** the baseline evaluator is run for the fixed golden case set +- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area + +#### Scenario: Baseline regeneration is documented +- **WHEN** a developer changes golden cases, fixtures, or evaluator logic +- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/tasks.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/tasks.md new file mode 100644 index 0000000..1118dd7 --- /dev/null +++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/tasks.md @@ -0,0 +1,26 @@ +## 1. Golden Cases + +- [x] 1.1 Create retrieval evaluation directory structure. +- [x] 1.2 Add fixed golden retrieval cases covering Chat and AIOps retrieval scenarios. +- [x] 1.3 Add matching offline retrieval fixtures for every golden case. + +## 2. Evaluator + +- [x] 2.1 Implement an offline retrieval evaluator script. +- [x] 2.2 Support hit-level classification and first expected document rank. +- [x] 2.3 Support JSON and Markdown report output. + +## 3. Baseline Report + +- [x] 3.1 Generate the baseline JSON report from the fixed cases and fixtures. +- [x] 3.2 Generate the baseline Markdown report from the fixed cases and fixtures. + +## 4. Documentation + +- [x] 4.1 Document the retrieval baseline purpose, file layout, and regeneration command. +- [x] 4.2 Link the retrieval baseline from the RAG refactor issue or related MVP documentation. + +## 5. Verification + +- [x] 5.1 Run the evaluator successfully against the fixed baseline cases. +- [x] 5.2 Run OpenSpec status/validation for the change and confirm tasks are complete. diff --git a/openspec/specs/rag-retrieval-evaluation/spec.md b/openspec/specs/rag-retrieval-evaluation/spec.md new file mode 100644 index 0000000..6a4861c --- /dev/null +++ b/openspec/specs/rag-retrieval-evaluation/spec.md @@ -0,0 +1,64 @@ +# rag-retrieval-evaluation Specification + +## Purpose +Provide a repeatable offline evaluation baseline for RAG retrieval behavior, so L0, query augmentation, evidence post-processing, and vector store changes can be checked against fixed golden retrieval cases before they affect Agent diagnosis quality. +## Requirements +### Requirement: Retrieval evaluation SHALL define fixed golden cases +The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically. + +#### Scenario: Golden case includes expected retrieval evidence +- **WHEN** a retrieval golden case is defined +- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs + +#### Scenario: Golden case distinguishes scenario type +- **WHEN** a retrieval golden case is defined +- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type + +### Requirement: Retrieval evaluation SHALL run offline against fixtures +The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services. + +#### Scenario: Fixture evaluation +- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory +- **THEN** it SHALL evaluate each case against its matching fixture file +- **AND** it SHALL not call external services + +#### Scenario: Missing fixture is reported +- **WHEN** a golden case has no matching retrieval fixture +- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason + +### Requirement: Retrieval evaluation SHALL classify hit quality +The evaluator SHALL classify each case into a deterministic hit level. + +#### Scenario: Strong hit classification +- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage +- **THEN** the evaluator SHALL classify the case as `strong` + +#### Scenario: Medium hit classification +- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage +- **THEN** the evaluator SHALL classify the case as `medium` + +#### Scenario: Miss classification +- **WHEN** retrieved candidates do not include expected documents or expected evidence +- **THEN** the evaluator SHALL classify the case as `miss` + +### Requirement: Retrieval evaluation SHALL report ranking signals +The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks. + +#### Scenario: Per-case ranking output +- **WHEN** a case is evaluated +- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks + +#### Scenario: Aggregate metrics output +- **WHEN** multiple cases are evaluated +- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available + +### Requirement: Retrieval evaluation SHALL preserve baseline reports +The system SHALL preserve generated baseline reports in JSON and Markdown formats. + +#### Scenario: Baseline report generation +- **WHEN** the baseline evaluator is run for the fixed golden case set +- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area + +#### Scenario: Baseline regeneration is documented +- **WHEN** a developer changes golden cases, fixtures, or evaluator logic +- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports diff --git a/scripts/eval_rag_retrieval.py b/scripts/eval_rag_retrieval.py new file mode 100644 index 0000000..c6c1e95 --- /dev/null +++ b/scripts/eval_rag_retrieval.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +"""Offline evaluator for RAG retrieval golden cases. + +The evaluator reads fixed golden cases and saved retrieval fixtures. It does not +call the running application or any external service. +""" + +from __future__ import annotations + +import argparse +import json +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + + +DEFAULT_CASES = Path("eval/rag-retrieval/cases/golden-cases.json") +DEFAULT_FIXTURES = Path("eval/rag-retrieval/fixtures") +DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline.json") +DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/baseline.md") + + +@dataclass +class Candidate: + rank: int + doc_id: str + title: str + breadcrumb: str + content: str + score: float | None + retrieval_layer: str | None + + @classmethod + def from_json(cls, raw: dict[str, Any], fallback_rank: int) -> "Candidate": + return cls( + rank=int(raw.get("rank") or fallback_rank), + doc_id=str(raw.get("docId") or raw.get("id") or ""), + title=str(raw.get("title") or ""), + breadcrumb=str(raw.get("breadcrumb") or ""), + content=str(raw.get("content") or ""), + score=_optional_float(raw.get("score")), + retrieval_layer=( + str(raw.get("retrievalLayer")) + if raw.get("retrievalLayer") is not None + else None + ), + ) + + def searchable_text(self) -> str: + return " ".join( + [self.doc_id, self.title, self.breadcrumb, self.content] + ).lower() + + def label(self) -> str: + label = self.doc_id or self.title or f"rank-{self.rank}" + return f"{self.rank}:{label}" + + +def _optional_float(value: Any) -> float | None: + if value is None: + return None + try: + return float(value) + except (TypeError, ValueError): + return None + + +def load_json(path: Path) -> Any: + with path.open("r", encoding="utf-8") as handle: + return json.load(handle) + + +def write_json(path: Path, payload: Any) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="\n") as handle: + json.dump(payload, handle, ensure_ascii=False, indent=2) + handle.write("\n") + + +def write_text(path: Path, content: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="\n") as handle: + handle.write(content) + + +def normalize_terms(values: list[Any]) -> list[str]: + return [str(value).lower() for value in values if str(value).strip()] + + +def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[str, Any]: + case_id = str(case["caseId"]) + fixture_path = fixture_dir / f"{case_id}.json" + expected_doc_ids = normalize_terms(case.get("expectedDocIds", [])) + expected_breadcrumbs = normalize_terms(case.get("expectedBreadcrumbs", [])) + expected_keywords = normalize_terms(case.get("expectedKeywords", [])) + + if not fixture_path.exists(): + return { + "caseId": case_id, + "scenario": case.get("scenario"), + "query": case.get("query"), + "hitLevel": "miss", + "passed": False, + "firstExpectedRank": None, + "topCandidates": [], + "failedChecks": [f"missing fixture: {fixture_path.as_posix()}"], + } + + fixture = load_json(fixture_path) + raw_candidates = fixture.get("candidates", []) + candidates = [ + Candidate.from_json(raw, index + 1) + for index, raw in enumerate(raw_candidates[:top_k]) + ] + + first_expected = None + expected_doc_candidate = None + for candidate in candidates: + candidate_doc = candidate.doc_id.lower() + if any(expected == candidate_doc for expected in expected_doc_ids): + first_expected = candidate.rank + expected_doc_candidate = candidate + break + + breadcrumb_match = False + keyword_matches: list[str] = [] + + if expected_doc_candidate is not None: + breadcrumb_text = expected_doc_candidate.breadcrumb.lower() + breadcrumb_match = any( + expected in breadcrumb_text or breadcrumb_text in expected + for expected in expected_breadcrumbs + ) + + searchable = expected_doc_candidate.searchable_text() + keyword_matches = [ + keyword for keyword in expected_keywords if keyword in searchable + ] + else: + all_text = " ".join(candidate.searchable_text() for candidate in candidates) + keyword_matches = [keyword for keyword in expected_keywords if keyword in all_text] + + failed_checks: list[str] = [] + if expected_doc_candidate is None: + failed_checks.append("expected document not found") + if expected_doc_candidate is not None and expected_breadcrumbs and not breadcrumb_match: + failed_checks.append("expected breadcrumb not found on expected document") + if expected_keywords and not keyword_matches: + failed_checks.append("expected evidence keywords not found") + + if expected_doc_candidate is not None and ( + breadcrumb_match or bool(keyword_matches) + ): + hit_level = "strong" + elif expected_doc_candidate is not None: + hit_level = "medium" + elif keyword_matches: + hit_level = "weak" + else: + hit_level = "miss" + + return { + "caseId": case_id, + "scenario": case.get("scenario"), + "query": case.get("query"), + "hitLevel": hit_level, + "passed": hit_level in {"strong", "medium"}, + "firstExpectedRank": first_expected, + "topCandidates": [candidate.label() for candidate in candidates], + "matchedKeywords": keyword_matches, + "breadcrumbMatched": breadcrumb_match, + "failedChecks": failed_checks, + } + + +def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]: + total = len(results) + counts = { + "strong": sum(1 for item in results if item["hitLevel"] == "strong"), + "medium": sum(1 for item in results if item["hitLevel"] == "medium"), + "weak": sum(1 for item in results if item["hitLevel"] == "weak"), + "miss": sum(1 for item in results if item["hitLevel"] == "miss"), + } + expected_ranks = [ + item["firstExpectedRank"] + for item in results + if item.get("firstExpectedRank") is not None + ] + passed = counts["strong"] + counts["medium"] + return { + "caseCount": total, + "topK": top_k, + "strongHitCount": counts["strong"], + "mediumHitCount": counts["medium"], + "weakHitCount": counts["weak"], + "missCount": counts["miss"], + "recallAtK": round(passed / total, 4) if total else 0, + "strongHitRate": round(counts["strong"] / total, 4) if total else 0, + "averageFirstHitRank": ( + round(sum(expected_ranks) / len(expected_ranks), 4) + if expected_ranks + else None + ), + } + + +def render_markdown(report: dict[str, Any]) -> str: + metrics = report["aggregate"] + lines = [ + "# RAG Retrieval Baseline", + "", + f"Generated at: `{report['generatedAt']}`", + "", + "## Aggregate", + "", + "| Metric | Value |", + "|---|---:|", + f"| Cases | {metrics['caseCount']} |", + f"| Top K | {metrics['topK']} |", + f"| Recall@K | {metrics['recallAtK']} |", + f"| Strong hit rate | {metrics['strongHitRate']} |", + f"| Strong hits | {metrics['strongHitCount']} |", + f"| Medium hits | {metrics['mediumHitCount']} |", + f"| Weak hits | {metrics['weakHitCount']} |", + f"| Misses | {metrics['missCount']} |", + f"| Average first hit rank | {metrics['averageFirstHitRank']} |", + "", + "## Cases", + "", + "| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |", + "|---|---|---|---:|---|---|", + ] + for item in report["results"]: + failed = "
".join(item["failedChecks"]) if item["failedChecks"] else "" + top = "
".join(item["topCandidates"]) + first_rank = item["firstExpectedRank"] + lines.append( + "| {case} | {scenario} | {hit} | {rank} | {top} | {failed} |".format( + case=item["caseId"], + scenario=item.get("scenario") or "", + hit=item["hitLevel"], + rank=first_rank if first_rank is not None else "", + top=top, + failed=failed, + ) + ) + lines.append("") + return "\n".join(lines) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--cases", type=Path, default=DEFAULT_CASES) + parser.add_argument("--fixtures", type=Path, default=DEFAULT_FIXTURES) + parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT) + parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + case_file = load_json(args.cases) + cases = case_file.get("cases", []) + top_k = int(case_file.get("topK") or 5) + results = [evaluate_case(case, args.fixtures, top_k) for case in cases] + report = { + "generatedAt": datetime.now(timezone.utc).isoformat(), + "caseFile": args.cases.as_posix(), + "fixtureDir": args.fixtures.as_posix(), + "aggregate": aggregate(results, top_k), + "results": results, + } + write_json(args.json_report, report) + write_text(args.markdown_report, render_markdown(report)) + + failed = [item for item in results if item["hitLevel"] == "miss"] + print( + "Evaluated {total} cases: recall@{top_k}={recall}, misses={misses}".format( + total=len(results), + top_k=top_k, + recall=report["aggregate"]["recallAtK"], + misses=len(failed), + ) + ) + return 1 if failed else 0 + + +if __name__ == "__main__": + raise SystemExit(main())