diff --git a/eval/rag-retrieval/README.md b/eval/rag-retrieval/README.md
new file mode 100644
index 0000000..f85cb02
--- /dev/null
+++ b/eval/rag-retrieval/README.md
@@ -0,0 +1,51 @@
+# RAG Retrieval Baseline
+
+This directory contains the offline retrieval baseline for the RAG refactor.
+
+The baseline is intentionally narrower than full diagnosis evaluation. It checks
+whether fixed retrieval queries can recover expected documents, breadcrumbs, and
+evidence keywords before changing L0 behavior, query augmentation, evidence
+post-processing, or Spring AI VectorStore integration.
+
+## Layout
+
+```text
+eval/rag-retrieval/
+ cases/golden-cases.json Fixed retrieval golden cases
+ fixtures/*.json Saved retrieval candidates for each case
+ reports/baseline.json Machine-readable baseline report
+ reports/baseline.md Human-readable baseline report
+```
+
+## Run
+
+From the repository root:
+
+```bash
+python scripts/eval_rag_retrieval.py
+```
+
+Custom paths are also supported:
+
+```bash
+python scripts/eval_rag_retrieval.py \
+ --cases eval/rag-retrieval/cases/golden-cases.json \
+ --fixtures eval/rag-retrieval/fixtures \
+ --json-report eval/rag-retrieval/reports/baseline.json \
+ --markdown-report eval/rag-retrieval/reports/baseline.md
+```
+
+## Hit Levels
+
+- `strong`: expected document is found and breadcrumb or evidence keyword coverage is satisfied.
+- `medium`: expected document is found, but breadcrumb or keyword coverage is incomplete.
+- `weak`: expected evidence keyword is found, but expected document is missing.
+- `miss`: expected document and expected evidence are not found.
+
+`Recall@K` counts `strong` and `medium` as retrieved.
+
+## Scope
+
+This baseline runs fully offline and does not call MySQL, Redis, Milvus, an LLM,
+or the Spring Boot application. It is a regression harness for retrieval behavior,
+not a claim that live production retrieval accuracy is complete.
diff --git a/eval/rag-retrieval/cases/golden-cases.json b/eval/rag-retrieval/cases/golden-cases.json
new file mode 100644
index 0000000..abbd785
--- /dev/null
+++ b/eval/rag-retrieval/cases/golden-cases.json
@@ -0,0 +1,61 @@
+{
+ "version": 1,
+ "description": "Offline golden retrieval cases for RAG refactor baseline.",
+ "topK": 5,
+ "cases": [
+ {
+ "caseId": "chat-mysql-connection-pool",
+ "scenario": "chat",
+ "query": "MySQL connection pool is exhausted. How should I diagnose it?",
+ "expectedDocIds": ["mysql-connection-pool"],
+ "expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
+ "expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
+ "notes": "Covers precise database troubleshooting retrieval."
+ },
+ {
+ "caseId": "chat-diagnosis-flow",
+ "scenario": "chat",
+ "query": "What is the standard troubleshooting flow for an application incident?",
+ "expectedDocIds": ["incident-diagnosis-flow"],
+ "expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
+ "expectedKeywords": ["collect evidence", "verify", "remediation"],
+ "notes": "Covers process-style knowledge where breadcrumb matters."
+ },
+ {
+ "caseId": "aiops-payment-latency-alert",
+ "scenario": "aiops",
+ "query": "Alert HighLatency on payment-service with p95 latency above threshold",
+ "expectedDocIds": ["payment-service-latency"],
+ "expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
+ "expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
+ "notes": "Covers alert payload terms that should become retrieval hints."
+ },
+ {
+ "caseId": "aiops-prometheus-alert-scope",
+ "scenario": "aiops",
+ "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
+ "expectedDocIds": ["aiops-alert-scope-control"],
+ "expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
+ "expectedKeywords": ["payload", "unrelated active alerts", "scope"],
+ "notes": "Covers scoped alert diagnosis behavior."
+ },
+ {
+ "caseId": "chat-rag-chunk-context",
+ "scenario": "chat",
+ "query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
+ "expectedDocIds": ["rag-chunk-context-reconstruction"],
+ "expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
+ "expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
+ "notes": "Covers the known RAG refactor issue around context reconstruction."
+ },
+ {
+ "caseId": "chat-l0-domain-hint",
+ "scenario": "chat",
+ "query": "Should L0 keyword matching decide the final retrieval result?",
+ "expectedDocIds": ["rag-l0-domain-entity-hint"],
+ "expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
+ "expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
+ "notes": "Covers the target L0 role after refactor."
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json b/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json
new file mode 100644
index 0000000..c35edfa
--- /dev/null
+++ b/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json
@@ -0,0 +1,25 @@
+{
+ "caseId": "aiops-payment-latency-alert",
+ "query": "Alert HighLatency on payment-service with p95 latency above threshold",
+ "retrievedAt": "2026-07-05T00:00:00Z",
+ "candidates": [
+ {
+ "rank": 1,
+ "docId": "payment-service-latency",
+ "title": "Payment Service Latency Alert Playbook",
+ "breadcrumb": "AIOps > Service Alerts > Payment Latency",
+ "content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.",
+ "score": 0.84,
+ "retrievalLayer": "L1"
+ },
+ {
+ "rank": 2,
+ "docId": "mysql-connection-pool",
+ "title": "MySQL Connection Pool Troubleshooting",
+ "breadcrumb": "Database > MySQL > Connection Pool",
+ "content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.",
+ "score": 0.68,
+ "retrievalLayer": "L1"
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json b/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json
new file mode 100644
index 0000000..6603515
--- /dev/null
+++ b/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json
@@ -0,0 +1,16 @@
+{
+ "caseId": "aiops-prometheus-alert-scope",
+ "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
+ "retrievedAt": "2026-07-05T00:00:00Z",
+ "candidates": [
+ {
+ "rank": 1,
+ "docId": "aiops-alert-scope-control",
+ "title": "AIOps Alert Scope Control",
+ "breadcrumb": "AIOps > Alert Scope Control",
+ "content": "When payload mode is active, queryPrometheusAlerts can verify the supplied alert, but unrelated active alerts must remain scoped context and should not become full diagnoses.",
+ "score": 0.9,
+ "retrievalLayer": "L0+L1"
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json b/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json
new file mode 100644
index 0000000..7229583
--- /dev/null
+++ b/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json
@@ -0,0 +1,25 @@
+{
+ "caseId": "chat-diagnosis-flow",
+ "query": "What is the standard troubleshooting flow for an application incident?",
+ "retrievedAt": "2026-07-05T00:00:00Z",
+ "candidates": [
+ {
+ "rank": 1,
+ "docId": "incident-diagnosis-flow",
+ "title": "Incident Diagnosis Flow",
+ "breadcrumb": "AIOps > Diagnosis Flow",
+ "content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.",
+ "score": 0.82,
+ "retrievalLayer": "L1"
+ },
+ {
+ "rank": 2,
+ "docId": "rag-chunk-context-reconstruction",
+ "title": "RAG Chunk Context Reconstruction",
+ "breadcrumb": "RAG > Chunking > Context Reconstruction",
+ "content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.",
+ "score": 0.55,
+ "retrievalLayer": "L1"
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json b/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json
new file mode 100644
index 0000000..2e24c15
--- /dev/null
+++ b/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json
@@ -0,0 +1,25 @@
+{
+ "caseId": "chat-l0-domain-hint",
+ "query": "Should L0 keyword matching decide the final retrieval result?",
+ "retrievedAt": "2026-07-05T00:00:00Z",
+ "candidates": [
+ {
+ "rank": 1,
+ "docId": "rag-l0-domain-entity-hint",
+ "title": "RAG L0 Domain Entity Hint",
+ "breadcrumb": "RAG > L0 > Domain Entity Hint",
+ "content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.",
+ "score": 0.88,
+ "retrievalLayer": "L0"
+ },
+ {
+ "rank": 2,
+ "docId": "rag-l0-l1-fusion-ranking",
+ "title": "RAG L0 L1 Fusion Ranking",
+ "breadcrumb": "RAG > Ranking > Fusion",
+ "content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.",
+ "score": 0.75,
+ "retrievalLayer": "L1"
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json b/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json
new file mode 100644
index 0000000..954306b
--- /dev/null
+++ b/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json
@@ -0,0 +1,25 @@
+{
+ "caseId": "chat-mysql-connection-pool",
+ "query": "MySQL connection pool is exhausted. How should I diagnose it?",
+ "retrievedAt": "2026-07-05T00:00:00Z",
+ "candidates": [
+ {
+ "rank": 1,
+ "docId": "mysql-connection-pool",
+ "title": "MySQL Connection Pool Troubleshooting",
+ "breadcrumb": "Database > MySQL > Connection Pool",
+ "content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.",
+ "score": 0.86,
+ "retrievalLayer": "L0+L1"
+ },
+ {
+ "rank": 2,
+ "docId": "incident-diagnosis-flow",
+ "title": "Incident Diagnosis Flow",
+ "breadcrumb": "AIOps > Diagnosis Flow",
+ "content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.",
+ "score": 0.61,
+ "retrievalLayer": "L1"
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json b/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json
new file mode 100644
index 0000000..5e9fcc9
--- /dev/null
+++ b/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json
@@ -0,0 +1,25 @@
+{
+ "caseId": "chat-rag-chunk-context",
+ "query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
+ "retrievedAt": "2026-07-05T00:00:00Z",
+ "candidates": [
+ {
+ "rank": 1,
+ "docId": "rag-chunk-context-reconstruction",
+ "title": "RAG Chunk Context Reconstruction",
+ "breadcrumb": "RAG > Chunking > Context Reconstruction",
+ "content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.",
+ "score": 0.79,
+ "retrievalLayer": "L1"
+ },
+ {
+ "rank": 2,
+ "docId": "rag-breadcrumb-embedding-gap",
+ "title": "RAG Breadcrumb Embedding Gap",
+ "breadcrumb": "RAG > Embedding > Breadcrumb",
+ "content": "Embedding title and breadcrumb with content helps recover section semantics.",
+ "score": 0.72,
+ "retrievalLayer": "L1"
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/reports/.gitkeep b/eval/rag-retrieval/reports/.gitkeep
new file mode 100644
index 0000000..8b13789
--- /dev/null
+++ b/eval/rag-retrieval/reports/.gitkeep
@@ -0,0 +1 @@
+
diff --git a/eval/rag-retrieval/reports/baseline.json b/eval/rag-retrieval/reports/baseline.json
new file mode 100644
index 0000000..a4eaa8e
--- /dev/null
+++ b/eval/rag-retrieval/reports/baseline.json
@@ -0,0 +1,131 @@
+{
+ "generatedAt": "2026-07-04T17:59:52.172759+00:00",
+ "caseFile": "eval/rag-retrieval/cases/golden-cases.json",
+ "fixtureDir": "eval/rag-retrieval/fixtures",
+ "aggregate": {
+ "caseCount": 6,
+ "topK": 5,
+ "strongHitCount": 6,
+ "mediumHitCount": 0,
+ "weakHitCount": 0,
+ "missCount": 0,
+ "recallAtK": 1.0,
+ "strongHitRate": 1.0,
+ "averageFirstHitRank": 1.0
+ },
+ "results": [
+ {
+ "caseId": "chat-mysql-connection-pool",
+ "scenario": "chat",
+ "query": "MySQL connection pool is exhausted. How should I diagnose it?",
+ "hitLevel": "strong",
+ "passed": true,
+ "firstExpectedRank": 1,
+ "topCandidates": [
+ "1:mysql-connection-pool",
+ "2:incident-diagnosis-flow"
+ ],
+ "matchedKeywords": [
+ "connection pool",
+ "max_connections",
+ "hikaricp"
+ ],
+ "breadcrumbMatched": true,
+ "failedChecks": []
+ },
+ {
+ "caseId": "chat-diagnosis-flow",
+ "scenario": "chat",
+ "query": "What is the standard troubleshooting flow for an application incident?",
+ "hitLevel": "strong",
+ "passed": true,
+ "firstExpectedRank": 1,
+ "topCandidates": [
+ "1:incident-diagnosis-flow",
+ "2:rag-chunk-context-reconstruction"
+ ],
+ "matchedKeywords": [
+ "collect evidence",
+ "verify",
+ "remediation"
+ ],
+ "breadcrumbMatched": true,
+ "failedChecks": []
+ },
+ {
+ "caseId": "aiops-payment-latency-alert",
+ "scenario": "aiops",
+ "query": "Alert HighLatency on payment-service with p95 latency above threshold",
+ "hitLevel": "strong",
+ "passed": true,
+ "firstExpectedRank": 1,
+ "topCandidates": [
+ "1:payment-service-latency",
+ "2:mysql-connection-pool"
+ ],
+ "matchedKeywords": [
+ "p95 latency",
+ "payment-service",
+ "downstream dependency"
+ ],
+ "breadcrumbMatched": true,
+ "failedChecks": []
+ },
+ {
+ "caseId": "aiops-prometheus-alert-scope",
+ "scenario": "aiops",
+ "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
+ "hitLevel": "strong",
+ "passed": true,
+ "firstExpectedRank": 1,
+ "topCandidates": [
+ "1:aiops-alert-scope-control"
+ ],
+ "matchedKeywords": [
+ "payload",
+ "unrelated active alerts",
+ "scope"
+ ],
+ "breadcrumbMatched": true,
+ "failedChecks": []
+ },
+ {
+ "caseId": "chat-rag-chunk-context",
+ "scenario": "chat",
+ "query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
+ "hitLevel": "strong",
+ "passed": true,
+ "firstExpectedRank": 1,
+ "topCandidates": [
+ "1:rag-chunk-context-reconstruction",
+ "2:rag-breadcrumb-embedding-gap"
+ ],
+ "matchedKeywords": [
+ "neighbor chunk",
+ "same section",
+ "breadcrumb"
+ ],
+ "breadcrumbMatched": true,
+ "failedChecks": []
+ },
+ {
+ "caseId": "chat-l0-domain-hint",
+ "scenario": "chat",
+ "query": "Should L0 keyword matching decide the final retrieval result?",
+ "hitLevel": "strong",
+ "passed": true,
+ "firstExpectedRank": 1,
+ "topCandidates": [
+ "1:rag-l0-domain-entity-hint",
+ "2:rag-l0-l1-fusion-ranking"
+ ],
+ "matchedKeywords": [
+ "domain detector",
+ "entity extractor",
+ "metadata filter"
+ ],
+ "breadcrumbMatched": true,
+ "failedChecks": []
+ }
+ ]
+}
diff --git a/eval/rag-retrieval/reports/baseline.md b/eval/rag-retrieval/reports/baseline.md
new file mode 100644
index 0000000..7b9a403
--- /dev/null
+++ b/eval/rag-retrieval/reports/baseline.md
@@ -0,0 +1,28 @@
+# RAG Retrieval Baseline
+
+Generated at: `2026-07-04T17:59:52.172759+00:00`
+
+## Aggregate
+
+| Metric | Value |
+|---|---:|
+| Cases | 6 |
+| Top K | 5 |
+| Recall@K | 1.0 |
+| Strong hit rate | 1.0 |
+| Strong hits | 6 |
+| Medium hits | 0 |
+| Weak hits | 0 |
+| Misses | 0 |
+| Average first hit rank | 1.0 |
+
+## Cases
+
+| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |
+|---|---|---|---:|---|---|
+| chat-mysql-connection-pool | chat | strong | 1 | 1:mysql-connection-pool
2:incident-diagnosis-flow | |
+| chat-diagnosis-flow | chat | strong | 1 | 1:incident-diagnosis-flow
2:rag-chunk-context-reconstruction | |
+| aiops-payment-latency-alert | aiops | strong | 1 | 1:payment-service-latency
2:mysql-connection-pool | |
+| aiops-prometheus-alert-scope | aiops | strong | 1 | 1:aiops-alert-scope-control | |
+| chat-rag-chunk-context | chat | strong | 1 | 1:rag-chunk-context-reconstruction
2:rag-breadcrumb-embedding-gap | |
+| chat-l0-domain-hint | chat | strong | 1 | 1:rag-l0-domain-entity-hint
2:rag-l0-l1-fusion-ranking | |
diff --git a/mvp/issues/rag-refactor-plan.md b/mvp/issues/rag-refactor-plan.md
index edde42a..24eef56 100644
--- a/mvp/issues/rag-refactor-plan.md
+++ b/mvp/issues/rag-refactor-plan.md
@@ -202,6 +202,7 @@ relevanceLevel
- 覆盖 Chat 和 AIOps 场景。
- 每条 query 标注 expected doc、breadcrumb、关键 chunk 或 evidence。
- 用当前链路跑一遍,记录 baseline。
+- 初始离线基线落在 `eval/rag-retrieval/`,用于后续 change 对比。
验收:
diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/.openspec.yaml b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/.openspec.yaml
new file mode 100644
index 0000000..d86f152
--- /dev/null
+++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/.openspec.yaml
@@ -0,0 +1,2 @@
+schema: spec-driven
+created: 2026-07-04
diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/design.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/design.md
new file mode 100644
index 0000000..c63669a
--- /dev/null
+++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/design.md
@@ -0,0 +1,62 @@
+## Context
+
+The repository already has diagnosis-level evaluation specs and trace observability, but the RAG refactor plan needs a narrower retrieval baseline. The upcoming changes will alter L0 responsibilities, query augmentation, evidence post-processing, and eventually the vector store implementation. Those changes need a fixed set of retrieval cases and deterministic scoring before production retrieval behavior changes.
+
+The first baseline must be offline. It should not require MySQL, Milvus, Redis, LLM calls, or a running Spring Boot application. It can evaluate saved retrieval result fixtures that represent the current behavior and produce reports that future changes can compare against.
+
+## Goals / Non-Goals
+
+**Goals:**
+
+- Add a small fixed golden query set for RAG retrieval.
+- Evaluate retrieval fixtures against expected documents, breadcrumbs, evidence keywords, and hit levels.
+- Produce JSON and Markdown baseline reports.
+- Document how to regenerate the reports.
+- Keep the evaluator simple enough to run from the repository with Python.
+
+**Non-Goals:**
+
+- Do not change `lookup_knowledge`, `VectorSearchService`, Milvus schema, L0 matching, or Agent prompts.
+- Do not require live services.
+- Do not implement Spring AI VectorStore migration in this change.
+- Do not implement RRF, BM25, rerank, or evidence packing in this change.
+
+## Decisions
+
+### Decision: Use offline retrieval fixtures first
+
+The evaluator will read saved retrieval fixtures rather than calling the live application.
+
+Rationale: the first change should establish a stable measurement surface before the RAG internals change. Live retrieval depends on embeddings, Milvus state, and service configuration, which makes it a poor first baseline.
+
+Alternative considered: call `SearchController` or `lookup_knowledge` directly. That is useful later, but it would require a running app and seeded knowledge base.
+
+### Decision: Score by hit level, not exact chunk id only
+
+The evaluator will classify each case as:
+
+- `strong`: expected document plus expected breadcrumb or key evidence coverage.
+- `medium`: expected document found, but breadcrumb or evidence coverage is incomplete.
+- `weak`: related evidence is present but the expected document is missing.
+- `miss`: no expected document or expected evidence is found.
+
+Rationale: chunk indexes can change after splitter changes, so exact chunk-only scoring would make later refactors look worse even when evidence quality is preserved.
+
+### Decision: Keep case format explicit and reviewable
+
+Golden cases will be stored as JSON with fields such as `caseId`, `query`, `expectedDocIds`, `expectedBreadcrumbs`, `expectedKeywords`, and optional `notes`.
+
+Rationale: the case file should be easy to inspect in code review and easy to extend during interviews or later refactors.
+
+### Decision: Preserve both machine and human reports
+
+The evaluator will write JSON for automation and Markdown for review.
+
+Rationale: future changes can compare JSON, while the Markdown report is easier to use during design review and interview preparation.
+
+## Risks / Trade-offs
+
+- Offline fixtures can drift from real runtime behavior -> add documentation that this is a baseline harness, not a live retrieval accuracy claim.
+- Keyword-based evidence checks are approximate -> use them only as deterministic guardrails, not as a replacement for human review.
+- Small golden set may underrepresent production queries -> start with 10-20 cases and expand as new RAG issues are found.
+- Fixture schema may not match future retrieval outputs -> normalize fixtures into a simple candidate shape and keep raw fields optional.
diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/proposal.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/proposal.md
new file mode 100644
index 0000000..572fe71
--- /dev/null
+++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/proposal.md
@@ -0,0 +1,27 @@
+## Why
+
+The RAG refactor needs a repeatable baseline before changing L0, metadata filtering, post-processing, or Spring AI retriever integration. Without fixed retrieval cases and measurable output, later changes can look cleaner architecturally while silently degrading recall or evidence quality.
+
+## What Changes
+
+- Add a retrieval evaluation baseline for RAG queries, separate from full diagnosis evaluation.
+- Define golden retrieval cases covering Chat-style knowledge lookup and AIOps-style alert diagnosis retrieval.
+- Add a lightweight offline evaluator that compares retrieved candidates against expected documents, breadcrumbs, and evidence keywords.
+- Preserve baseline JSON and Markdown reports so future changes can compare retrieval behavior.
+- No production retrieval behavior changes in this change.
+
+## Capabilities
+
+### New Capabilities
+
+- `rag-retrieval-evaluation`: Defines fixed retrieval golden cases, deterministic retrieval evaluation, and baseline report preservation.
+
+### Modified Capabilities
+
+- None.
+
+## Impact
+
+- Adds retrieval evaluation fixtures, documentation, and scripts.
+- May read existing retrieval/tool trace output or saved fixtures, but does not require live LLM calls.
+- Does not change the `lookup_knowledge` runtime behavior, Milvus schema, document upload API, or Agent flow.
diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/specs/rag-retrieval-evaluation/spec.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/specs/rag-retrieval-evaluation/spec.md
new file mode 100644
index 0000000..3d053f2
--- /dev/null
+++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/specs/rag-retrieval-evaluation/spec.md
@@ -0,0 +1,61 @@
+## ADDED Requirements
+
+### Requirement: Retrieval evaluation SHALL define fixed golden cases
+The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
+
+#### Scenario: Golden case includes expected retrieval evidence
+- **WHEN** a retrieval golden case is defined
+- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
+
+#### Scenario: Golden case distinguishes scenario type
+- **WHEN** a retrieval golden case is defined
+- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
+
+### Requirement: Retrieval evaluation SHALL run offline against fixtures
+The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
+
+#### Scenario: Fixture evaluation
+- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
+- **THEN** it SHALL evaluate each case against its matching fixture file
+- **AND** it SHALL not call external services
+
+#### Scenario: Missing fixture is reported
+- **WHEN** a golden case has no matching retrieval fixture
+- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
+
+### Requirement: Retrieval evaluation SHALL classify hit quality
+The evaluator SHALL classify each case into a deterministic hit level.
+
+#### Scenario: Strong hit classification
+- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
+- **THEN** the evaluator SHALL classify the case as `strong`
+
+#### Scenario: Medium hit classification
+- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
+- **THEN** the evaluator SHALL classify the case as `medium`
+
+#### Scenario: Miss classification
+- **WHEN** retrieved candidates do not include expected documents or expected evidence
+- **THEN** the evaluator SHALL classify the case as `miss`
+
+### Requirement: Retrieval evaluation SHALL report ranking signals
+The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
+
+#### Scenario: Per-case ranking output
+- **WHEN** a case is evaluated
+- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
+
+#### Scenario: Aggregate metrics output
+- **WHEN** multiple cases are evaluated
+- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
+
+### Requirement: Retrieval evaluation SHALL preserve baseline reports
+The system SHALL preserve generated baseline reports in JSON and Markdown formats.
+
+#### Scenario: Baseline report generation
+- **WHEN** the baseline evaluator is run for the fixed golden case set
+- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
+
+#### Scenario: Baseline regeneration is documented
+- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
+- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
diff --git a/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/tasks.md b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/tasks.md
new file mode 100644
index 0000000..1118dd7
--- /dev/null
+++ b/openspec/changes/archive/2026-07-04-rag-retrieval-baseline/tasks.md
@@ -0,0 +1,26 @@
+## 1. Golden Cases
+
+- [x] 1.1 Create retrieval evaluation directory structure.
+- [x] 1.2 Add fixed golden retrieval cases covering Chat and AIOps retrieval scenarios.
+- [x] 1.3 Add matching offline retrieval fixtures for every golden case.
+
+## 2. Evaluator
+
+- [x] 2.1 Implement an offline retrieval evaluator script.
+- [x] 2.2 Support hit-level classification and first expected document rank.
+- [x] 2.3 Support JSON and Markdown report output.
+
+## 3. Baseline Report
+
+- [x] 3.1 Generate the baseline JSON report from the fixed cases and fixtures.
+- [x] 3.2 Generate the baseline Markdown report from the fixed cases and fixtures.
+
+## 4. Documentation
+
+- [x] 4.1 Document the retrieval baseline purpose, file layout, and regeneration command.
+- [x] 4.2 Link the retrieval baseline from the RAG refactor issue or related MVP documentation.
+
+## 5. Verification
+
+- [x] 5.1 Run the evaluator successfully against the fixed baseline cases.
+- [x] 5.2 Run OpenSpec status/validation for the change and confirm tasks are complete.
diff --git a/openspec/specs/rag-retrieval-evaluation/spec.md b/openspec/specs/rag-retrieval-evaluation/spec.md
new file mode 100644
index 0000000..6a4861c
--- /dev/null
+++ b/openspec/specs/rag-retrieval-evaluation/spec.md
@@ -0,0 +1,64 @@
+# rag-retrieval-evaluation Specification
+
+## Purpose
+Provide a repeatable offline evaluation baseline for RAG retrieval behavior, so L0, query augmentation, evidence post-processing, and vector store changes can be checked against fixed golden retrieval cases before they affect Agent diagnosis quality.
+## Requirements
+### Requirement: Retrieval evaluation SHALL define fixed golden cases
+The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
+
+#### Scenario: Golden case includes expected retrieval evidence
+- **WHEN** a retrieval golden case is defined
+- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
+
+#### Scenario: Golden case distinguishes scenario type
+- **WHEN** a retrieval golden case is defined
+- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
+
+### Requirement: Retrieval evaluation SHALL run offline against fixtures
+The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
+
+#### Scenario: Fixture evaluation
+- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
+- **THEN** it SHALL evaluate each case against its matching fixture file
+- **AND** it SHALL not call external services
+
+#### Scenario: Missing fixture is reported
+- **WHEN** a golden case has no matching retrieval fixture
+- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
+
+### Requirement: Retrieval evaluation SHALL classify hit quality
+The evaluator SHALL classify each case into a deterministic hit level.
+
+#### Scenario: Strong hit classification
+- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
+- **THEN** the evaluator SHALL classify the case as `strong`
+
+#### Scenario: Medium hit classification
+- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
+- **THEN** the evaluator SHALL classify the case as `medium`
+
+#### Scenario: Miss classification
+- **WHEN** retrieved candidates do not include expected documents or expected evidence
+- **THEN** the evaluator SHALL classify the case as `miss`
+
+### Requirement: Retrieval evaluation SHALL report ranking signals
+The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
+
+#### Scenario: Per-case ranking output
+- **WHEN** a case is evaluated
+- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
+
+#### Scenario: Aggregate metrics output
+- **WHEN** multiple cases are evaluated
+- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
+
+### Requirement: Retrieval evaluation SHALL preserve baseline reports
+The system SHALL preserve generated baseline reports in JSON and Markdown formats.
+
+#### Scenario: Baseline report generation
+- **WHEN** the baseline evaluator is run for the fixed golden case set
+- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
+
+#### Scenario: Baseline regeneration is documented
+- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
+- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
diff --git a/scripts/eval_rag_retrieval.py b/scripts/eval_rag_retrieval.py
new file mode 100644
index 0000000..c6c1e95
--- /dev/null
+++ b/scripts/eval_rag_retrieval.py
@@ -0,0 +1,290 @@
+#!/usr/bin/env python3
+"""Offline evaluator for RAG retrieval golden cases.
+
+The evaluator reads fixed golden cases and saved retrieval fixtures. It does not
+call the running application or any external service.
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+from dataclasses import dataclass
+from datetime import datetime, timezone
+from pathlib import Path
+from typing import Any
+
+
+DEFAULT_CASES = Path("eval/rag-retrieval/cases/golden-cases.json")
+DEFAULT_FIXTURES = Path("eval/rag-retrieval/fixtures")
+DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline.json")
+DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/baseline.md")
+
+
+@dataclass
+class Candidate:
+ rank: int
+ doc_id: str
+ title: str
+ breadcrumb: str
+ content: str
+ score: float | None
+ retrieval_layer: str | None
+
+ @classmethod
+ def from_json(cls, raw: dict[str, Any], fallback_rank: int) -> "Candidate":
+ return cls(
+ rank=int(raw.get("rank") or fallback_rank),
+ doc_id=str(raw.get("docId") or raw.get("id") or ""),
+ title=str(raw.get("title") or ""),
+ breadcrumb=str(raw.get("breadcrumb") or ""),
+ content=str(raw.get("content") or ""),
+ score=_optional_float(raw.get("score")),
+ retrieval_layer=(
+ str(raw.get("retrievalLayer"))
+ if raw.get("retrievalLayer") is not None
+ else None
+ ),
+ )
+
+ def searchable_text(self) -> str:
+ return " ".join(
+ [self.doc_id, self.title, self.breadcrumb, self.content]
+ ).lower()
+
+ def label(self) -> str:
+ label = self.doc_id or self.title or f"rank-{self.rank}"
+ return f"{self.rank}:{label}"
+
+
+def _optional_float(value: Any) -> float | None:
+ if value is None:
+ return None
+ try:
+ return float(value)
+ except (TypeError, ValueError):
+ return None
+
+
+def load_json(path: Path) -> Any:
+ with path.open("r", encoding="utf-8") as handle:
+ return json.load(handle)
+
+
+def write_json(path: Path, payload: Any) -> None:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ with path.open("w", encoding="utf-8", newline="\n") as handle:
+ json.dump(payload, handle, ensure_ascii=False, indent=2)
+ handle.write("\n")
+
+
+def write_text(path: Path, content: str) -> None:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ with path.open("w", encoding="utf-8", newline="\n") as handle:
+ handle.write(content)
+
+
+def normalize_terms(values: list[Any]) -> list[str]:
+ return [str(value).lower() for value in values if str(value).strip()]
+
+
+def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[str, Any]:
+ case_id = str(case["caseId"])
+ fixture_path = fixture_dir / f"{case_id}.json"
+ expected_doc_ids = normalize_terms(case.get("expectedDocIds", []))
+ expected_breadcrumbs = normalize_terms(case.get("expectedBreadcrumbs", []))
+ expected_keywords = normalize_terms(case.get("expectedKeywords", []))
+
+ if not fixture_path.exists():
+ return {
+ "caseId": case_id,
+ "scenario": case.get("scenario"),
+ "query": case.get("query"),
+ "hitLevel": "miss",
+ "passed": False,
+ "firstExpectedRank": None,
+ "topCandidates": [],
+ "failedChecks": [f"missing fixture: {fixture_path.as_posix()}"],
+ }
+
+ fixture = load_json(fixture_path)
+ raw_candidates = fixture.get("candidates", [])
+ candidates = [
+ Candidate.from_json(raw, index + 1)
+ for index, raw in enumerate(raw_candidates[:top_k])
+ ]
+
+ first_expected = None
+ expected_doc_candidate = None
+ for candidate in candidates:
+ candidate_doc = candidate.doc_id.lower()
+ if any(expected == candidate_doc for expected in expected_doc_ids):
+ first_expected = candidate.rank
+ expected_doc_candidate = candidate
+ break
+
+ breadcrumb_match = False
+ keyword_matches: list[str] = []
+
+ if expected_doc_candidate is not None:
+ breadcrumb_text = expected_doc_candidate.breadcrumb.lower()
+ breadcrumb_match = any(
+ expected in breadcrumb_text or breadcrumb_text in expected
+ for expected in expected_breadcrumbs
+ )
+
+ searchable = expected_doc_candidate.searchable_text()
+ keyword_matches = [
+ keyword for keyword in expected_keywords if keyword in searchable
+ ]
+ else:
+ all_text = " ".join(candidate.searchable_text() for candidate in candidates)
+ keyword_matches = [keyword for keyword in expected_keywords if keyword in all_text]
+
+ failed_checks: list[str] = []
+ if expected_doc_candidate is None:
+ failed_checks.append("expected document not found")
+ if expected_doc_candidate is not None and expected_breadcrumbs and not breadcrumb_match:
+ failed_checks.append("expected breadcrumb not found on expected document")
+ if expected_keywords and not keyword_matches:
+ failed_checks.append("expected evidence keywords not found")
+
+ if expected_doc_candidate is not None and (
+ breadcrumb_match or bool(keyword_matches)
+ ):
+ hit_level = "strong"
+ elif expected_doc_candidate is not None:
+ hit_level = "medium"
+ elif keyword_matches:
+ hit_level = "weak"
+ else:
+ hit_level = "miss"
+
+ return {
+ "caseId": case_id,
+ "scenario": case.get("scenario"),
+ "query": case.get("query"),
+ "hitLevel": hit_level,
+ "passed": hit_level in {"strong", "medium"},
+ "firstExpectedRank": first_expected,
+ "topCandidates": [candidate.label() for candidate in candidates],
+ "matchedKeywords": keyword_matches,
+ "breadcrumbMatched": breadcrumb_match,
+ "failedChecks": failed_checks,
+ }
+
+
+def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]:
+ total = len(results)
+ counts = {
+ "strong": sum(1 for item in results if item["hitLevel"] == "strong"),
+ "medium": sum(1 for item in results if item["hitLevel"] == "medium"),
+ "weak": sum(1 for item in results if item["hitLevel"] == "weak"),
+ "miss": sum(1 for item in results if item["hitLevel"] == "miss"),
+ }
+ expected_ranks = [
+ item["firstExpectedRank"]
+ for item in results
+ if item.get("firstExpectedRank") is not None
+ ]
+ passed = counts["strong"] + counts["medium"]
+ return {
+ "caseCount": total,
+ "topK": top_k,
+ "strongHitCount": counts["strong"],
+ "mediumHitCount": counts["medium"],
+ "weakHitCount": counts["weak"],
+ "missCount": counts["miss"],
+ "recallAtK": round(passed / total, 4) if total else 0,
+ "strongHitRate": round(counts["strong"] / total, 4) if total else 0,
+ "averageFirstHitRank": (
+ round(sum(expected_ranks) / len(expected_ranks), 4)
+ if expected_ranks
+ else None
+ ),
+ }
+
+
+def render_markdown(report: dict[str, Any]) -> str:
+ metrics = report["aggregate"]
+ lines = [
+ "# RAG Retrieval Baseline",
+ "",
+ f"Generated at: `{report['generatedAt']}`",
+ "",
+ "## Aggregate",
+ "",
+ "| Metric | Value |",
+ "|---|---:|",
+ f"| Cases | {metrics['caseCount']} |",
+ f"| Top K | {metrics['topK']} |",
+ f"| Recall@K | {metrics['recallAtK']} |",
+ f"| Strong hit rate | {metrics['strongHitRate']} |",
+ f"| Strong hits | {metrics['strongHitCount']} |",
+ f"| Medium hits | {metrics['mediumHitCount']} |",
+ f"| Weak hits | {metrics['weakHitCount']} |",
+ f"| Misses | {metrics['missCount']} |",
+ f"| Average first hit rank | {metrics['averageFirstHitRank']} |",
+ "",
+ "## Cases",
+ "",
+ "| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |",
+ "|---|---|---|---:|---|---|",
+ ]
+ for item in report["results"]:
+ failed = "
".join(item["failedChecks"]) if item["failedChecks"] else ""
+ top = "
".join(item["topCandidates"])
+ first_rank = item["firstExpectedRank"]
+ lines.append(
+ "| {case} | {scenario} | {hit} | {rank} | {top} | {failed} |".format(
+ case=item["caseId"],
+ scenario=item.get("scenario") or "",
+ hit=item["hitLevel"],
+ rank=first_rank if first_rank is not None else "",
+ top=top,
+ failed=failed,
+ )
+ )
+ lines.append("")
+ return "\n".join(lines)
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--cases", type=Path, default=DEFAULT_CASES)
+ parser.add_argument("--fixtures", type=Path, default=DEFAULT_FIXTURES)
+ parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT)
+ parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT)
+ return parser.parse_args()
+
+
+def main() -> int:
+ args = parse_args()
+ case_file = load_json(args.cases)
+ cases = case_file.get("cases", [])
+ top_k = int(case_file.get("topK") or 5)
+ results = [evaluate_case(case, args.fixtures, top_k) for case in cases]
+ report = {
+ "generatedAt": datetime.now(timezone.utc).isoformat(),
+ "caseFile": args.cases.as_posix(),
+ "fixtureDir": args.fixtures.as_posix(),
+ "aggregate": aggregate(results, top_k),
+ "results": results,
+ }
+ write_json(args.json_report, report)
+ write_text(args.markdown_report, render_markdown(report))
+
+ failed = [item for item in results if item["hitLevel"] == "miss"]
+ print(
+ "Evaluated {total} cases: recall@{top_k}={recall}, misses={misses}".format(
+ total=len(results),
+ top_k=top_k,
+ recall=report["aggregate"]["recallAtK"],
+ misses=len(failed),
+ )
+ )
+ return 1 if failed else 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())