From ed7efc58b7008d3c59dff0ca539ec1e5b80ba14b Mon Sep 17 00:00:00 2001 From: zhuyongxin Date: Mon, 6 Jul 2026 21:39:27 +0800 Subject: [PATCH] feat(rag): close eval pipeline with live snapshots --- devflow/index.md | 1 + .../acceptance.md | 78 +++ .../brief.md | 21 + .../decisions.md | 57 ++ .../evidence.md | 87 +++ eval/rag-retrieval/README.md | 151 ++++- eval/rag-retrieval/cases/golden-cases.json | 51 ++ .../fixtures/aiops-payment-latency-alert.json | 94 +++- .../aiops-prometheus-alert-scope.json | 71 ++- .../fixtures/chat-diagnosis-flow.json | 94 +++- .../fixtures/chat-l0-domain-hint.json | 94 +++- .../fixtures/chat-l0-filter-fallback.json | 91 +++ .../fixtures/chat-mysql-connection-pool.json | 94 +++- .../fixtures/chat-rag-chunk-context.json | 94 +++- eval/rag-retrieval/reports/baseline.json | 98 +++- eval/rag-retrieval/reports/baseline.md | 27 +- .../seed-docs/aiops-alert-scope-control.md | 26 + .../seed-docs/incident-diagnosis-flow.md | 27 + .../seed-docs/mysql-connection-pool.md | 29 + .../seed-docs/payment-service-latency.md | 28 + .../rag-chunk-context-reconstruction.md | 28 + .../seed-docs/rag-l0-domain-entity-hint.md | 29 + .../seed-docs/rag-l0-filter-decoy.md | 19 + .../seed-docs/rag-l0-filter-fallback.md | 25 + mvp/architecture/README.md | 12 +- mvp/architecture/rag-architecture.md | 8 +- mvp/architecture/rag-eval-closure.md | 181 ++++++ mvp/architecture/retrieval-observability.md | 4 +- scripts/eval_rag_retrieval.py | 521 +++++++++++++++++- scripts/generate_rag_lookup_snapshots.ps1 | 34 ++ scripts/prepare_rag_eval_seed.ps1 | 18 + .../com/superbiz/agent/dto/Frontmatter.java | 8 + .../superbiz/agent/dto/KnowledgeEntry.java | 2 + .../service/DocumentManagementService.java | 21 +- .../agent/service/FrontmatterParser.java | 42 ++ .../agent/service/KnowledgeIndexService.java | 22 + .../SpringAiVectorStoreSidecarService.java | 29 +- .../agent/service/VectorIndexService.java | 53 +- .../agent/service/VectorSearchService.java | 64 ++- src/main/resources/application.yml | 1 + .../agent/eval/RagEvalSeedImporterTest.java | 93 ++++ .../eval/RagLookupSnapshotGeneratorTest.java | 85 +++ .../DocumentManagementServiceTest.java | 13 + .../agent/service/FrontmatterParserTest.java | 44 ++ .../service/KnowledgeIndexServiceTest.java | 43 ++ .../agent/service/VectorIndexServiceTest.java | 34 ++ .../service/VectorSearchServiceTest.java | 44 ++ 47 files changed, 2613 insertions(+), 177 deletions(-) create mode 100644 devflow/projects/2026-07-06-rag-eval-pipeline-closure/acceptance.md create mode 100644 devflow/projects/2026-07-06-rag-eval-pipeline-closure/brief.md create mode 100644 devflow/projects/2026-07-06-rag-eval-pipeline-closure/decisions.md create mode 100644 devflow/projects/2026-07-06-rag-eval-pipeline-closure/evidence.md create mode 100644 eval/rag-retrieval/fixtures/chat-l0-filter-fallback.json create mode 100644 eval/rag-retrieval/seed-docs/aiops-alert-scope-control.md create mode 100644 eval/rag-retrieval/seed-docs/incident-diagnosis-flow.md create mode 100644 eval/rag-retrieval/seed-docs/mysql-connection-pool.md create mode 100644 eval/rag-retrieval/seed-docs/payment-service-latency.md create mode 100644 eval/rag-retrieval/seed-docs/rag-chunk-context-reconstruction.md create mode 100644 eval/rag-retrieval/seed-docs/rag-l0-domain-entity-hint.md create mode 100644 eval/rag-retrieval/seed-docs/rag-l0-filter-decoy.md create mode 100644 eval/rag-retrieval/seed-docs/rag-l0-filter-fallback.md create mode 100644 mvp/architecture/rag-eval-closure.md create mode 100644 scripts/generate_rag_lookup_snapshots.ps1 create mode 100644 scripts/prepare_rag_eval_seed.ps1 create mode 100644 src/test/java/com/superbiz/agent/eval/RagEvalSeedImporterTest.java create mode 100644 src/test/java/com/superbiz/agent/eval/RagLookupSnapshotGeneratorTest.java diff --git a/devflow/index.md b/devflow/index.md index e6e956e..3ed2022 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -5,6 +5,7 @@ | 日期 | slug | 领域 | 关键词 | 状态 | |---|---|---|---|---| | 2026-07-05 | diagnosis-playbook-skills | Agent Skill/Playbook | read_skill, diagnosis playbook, progressive disclosure, payment timeout, MySQL pool, Redis timeout | openspec/changes/diagnosis-playbook-skills | implemented | +| 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived | | 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived | | 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived | | 2026-07-05 | diagnosis-eval-baseline-diff | Agent 评测/回归 Diff | baseline diff, regression detection, evidence coverage, cost signal, markdown report | openspec/changes/archive/2026-07-05-diagnosis-eval-baseline-diff | archived | diff --git a/devflow/projects/2026-07-06-rag-eval-pipeline-closure/acceptance.md b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/acceptance.md new file mode 100644 index 0000000..769d549 --- /dev/null +++ b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/acceptance.md @@ -0,0 +1,78 @@ +# Acceptance: rag-eval-pipeline-closure + +## Status + +Archived. + +## Acceptance Criteria + +| Item | Status | Notes | +|---|---|---| +| Modular fixture support | Done | Evaluator reads `lookupResult.evidenceBlocks/contextPack/retrievalTrace/rerankTrace`. | +| LookupResult-only contract | Done | Evaluator fails fixtures that do not expose `lookupResult`. | +| Golden modular assertions | Done | Cases assert selected attempt, accepted fallback reason, evidence status, context sources, and rerank top source. | +| Fallback coverage | Done | Added `chat-l0-filter-fallback` for filtered low-quality/no-evidence to unfiltered retry. | +| Real tool snapshot generation | Done | Added `RagLookupSnapshotGeneratorTest` and `generate_rag_lookup_snapshots.ps1`, defaulting to Spring AI VectorStore mode. | +| Seed docs import/reindex | Done | Added canonical seed docs, `RagEvalSeedImporterTest`, and `prepare_rag_eval_seed.ps1`. | +| Eval metadata isolation | Done | Added `kb_scope` metadata and `retrieval.kb-scope` filtering for L0 and L1. | +| Frontmatter body split | Done | Upload chunking embeds Markdown body, while frontmatter feeds metadata and L0. | +| Baseline diff | Done | `--compare-to` writes JSON/Markdown diff and exits non-zero on regression. | +| Documentation | Done | Updated RAG eval README and added `mvp/architecture/rag-eval-closure.md`. | + +## Verification + +```powershell +python scripts\eval_rag_retrieval.py +``` + +Result: passed. 7 cases, passRate=1.0, recall@5=1.0. + +```powershell +mvn -q "-Dtest=RagLookupSnapshotGeneratorTest" test +``` + +Result: passed. The snapshot generator stays disabled unless `rag.snapshot.enabled=true` is provided. + +```powershell +mvn -q "-Dtest=RagLookupSnapshotGeneratorTest" "-Drag.snapshot.enabled=true" "-Drag.snapshot.fixtures=" "-Drag.snapshot.retrievedAt=2026-07-06T00:00:00Z" "-Dretrieval.kb-scope=rag-eval" "-Dretrieval.vector-store.mode=spring" test +python scripts\eval_rag_retrieval.py --fixtures --json-report --markdown-report +``` + +Result: passed. Spring AI VectorStore live snapshot produced 7 cases, passRate=1.0, recall@5=1.0. The fallback case used `selectedAttempt=UNFILTERED_VECTOR_RETRY` and `fallbackReason=filtered_vector_no_evidence`; the expected source `rag-l0-filter-fallback` remained rank 1. + +```powershell +python scripts\eval_rag_retrieval.py --json-report --markdown-report --compare-to eval\rag-retrieval\reports\baseline.json --diff-json-report --diff-markdown-report +``` + +Result: passed. regressions=0. + +```powershell +mvn -q "-Dtest=FrontmatterParserTest,VectorIndexServiceTest,VectorSearchServiceTest,DocumentManagementServiceTest,RagLookupSnapshotGeneratorTest,RagEvalSeedImporterTest" test +``` + +Result: passed. The seed importer and snapshot generator remain disabled unless their system properties are explicitly enabled. + +```powershell +$null = [scriptblock]::Create((Get-Content -Raw scripts\prepare_rag_eval_seed.ps1)) +$null = [scriptblock]::Create((Get-Content -Raw scripts\generate_rag_lookup_snapshots.ps1)) +``` + +Result: PowerShell syntax OK. + +```powershell +.\scripts\prepare_rag_eval_seed.ps1 +``` + +Result: passed. Seed docs were imported through `DocumentManagementService` and reindexed into the configured runtime DB/vector stack. + +```powershell +.\scripts\generate_rag_lookup_snapshots.ps1 -Fixtures -RetrievedAt 2026-07-06T00:00:00Z -SkipEval +``` + +Result: passed after defaulting the script to `retrieval.vector-store.mode=spring`. The script generated live fixtures through the real `LookupKnowledgeTool` and then the offline evaluator reported 7 cases, passRate=1.0, recall@5=1.0. + +```powershell +git diff --check +``` + +Result: no whitespace errors. Git reported only LF/CRLF conversion warnings. diff --git a/devflow/projects/2026-07-06-rag-eval-pipeline-closure/brief.md b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/brief.md new file mode 100644 index 0000000..8b6e4dc --- /dev/null +++ b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/brief.md @@ -0,0 +1,21 @@ +# Brief: rag-eval-pipeline-closure + +## Background + +The modular RAG pipeline now returns `LookupResult` with `evidenceBlocks`, `contextPack`, `retrievalTrace`, and `rerankTrace`. The RAG retrieval baseline must validate that full contract, so it can detect regressions in fallback behavior, context packing, or rerank trace. + +## Goals + +1. Reuse the existing offline RAG retrieval baseline. +2. Extend it to support modular `LookupResult` fixtures. +3. Add golden assertions for selected attempt, fallback reason, evidence status, context sources, and rerank top source. +4. Add a RAG baseline diff path for regression detection. +5. Add a snapshot generator that calls the real `LookupKnowledgeTool`. +6. Document how RAG baseline and diagnosis baseline form a quality loop. + +## Non-Goals + +- No new production API. +- No LLM-as-judge scoring. +- No production API behavior changes. +- No replacement for diagnosis eval. diff --git a/devflow/projects/2026-07-06-rag-eval-pipeline-closure/decisions.md b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/decisions.md new file mode 100644 index 0000000..ca5b011 --- /dev/null +++ b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/decisions.md @@ -0,0 +1,57 @@ +# Decisions: rag-eval-pipeline-closure + +## D1. Reuse the existing evaluator + +Decision: extend `scripts/eval_rag_retrieval.py` instead of creating a second evaluator. + +Reason: the old evaluator already owns golden cases, fixtures, hit-level classification, and Markdown/JSON reports. Extending it keeps one RAG baseline path. + +## D2. Use LookupResult as the only fixture contract + +Decision: support `lookupResult` only. + +Reason: the MVP has moved to evidence-first RAG. Keeping an older fixture contract would weaken the baseline and let incomplete fixtures bypass context packing, retrieval trace, and rerank checks. + +## D3. Make modular assertions opt-in per case + +Decision: use fields such as `expectedSelectedAttempt`, `expectedFallbackReason`/`expectedFallbackReasons`, `expectedEvidenceStatus`, `expectedContextSources`, and `expectedRerankTopSource`. + +Reason: golden cases can be strict where the pipeline path matters without forcing every historical case to assert every new field. + +## D4. Diff remains deterministic + +Decision: RAG diff compares report fields only and does not call live services or models. + +Reason: this keeps it suitable for local regression checks and CI-style gates. + +## D5. Isolate live eval docs with kb_scope + +Decision: add `kb_scope` metadata and use `rag-eval` for canonical eval seed documents. + +Reason: local production documents are not stable enough for golden retrieval expectations. Scope isolation lets real `LookupKnowledgeTool` snapshots use the same MySQL/Milvus stack while avoiding accidental matches from unrelated local data. + +Default runtime keeps `retrieval.kb-scope` empty so legacy documents without `kb_scope` remain searchable. Eval scripts pass `-Dretrieval.kb-scope=rag-eval`. The same scope applies to L0 query hints and L1 vector retrieval. + +## D6. Import seed docs through the real upload pipeline + +Decision: seed docs are imported by `RagEvalSeedImporterTest` through `DocumentManagementService.uploadDocument`. + +Reason: this updates DB metadata, L0 index state, local knowledge files, and Milvus chunks in the same way as normal document ingestion. A direct Milvus-only seed would make the live eval less representative. + +## D7. Strip frontmatter before chunk embedding + +Decision: uploaded Markdown frontmatter feeds metadata/L0 but is stripped before chunking and embedding. + +Reason: frontmatter is a control plane, not evidence text. Keeping it in chunks lets L0-only keywords artificially improve vector similarity, especially for fallback decoy cases. + +## D8. Treat retry behavior as the stable fallback contract + +Decision: the fallback golden case accepts both `filtered_vector_low_quality` and `filtered_vector_no_evidence`, while still requiring `selectedAttempt=UNFILTERED_VECTOR_RETRY`, expected evidence source, context packing, and rerank top source. + +Reason: Spring AI VectorStore and the Milvus SDK can differ on whether an over-filtered first pass returns a weak candidate or no candidate. The MVP contract is that the retriever skips only the L0 category filter, keeps `kb_scope`, retries the original query, and returns the correct evidence. + +## D9. Default live snapshots to Spring AI VectorStore + +Decision: `generate_rag_lookup_snapshots.ps1` defaults to `retrieval.vector-store.mode=spring`. + +Reason: Spring AI VectorStore is the current framework path for the project and should be the default live verification route. SDK mode remains available through `-VectorStoreMode sdk` for comparison. diff --git a/devflow/projects/2026-07-06-rag-eval-pipeline-closure/evidence.md b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/evidence.md new file mode 100644 index 0000000..ded39a7 --- /dev/null +++ b/devflow/projects/2026-07-06-rag-eval-pipeline-closure/evidence.md @@ -0,0 +1,87 @@ +# Evidence: rag-eval-pipeline-closure + +## Changed Assets + +- `scripts/eval_rag_retrieval.py` +- `eval/rag-retrieval/cases/golden-cases.json` +- `eval/rag-retrieval/fixtures/*.json` +- `eval/rag-retrieval/reports/baseline.json` +- `eval/rag-retrieval/reports/baseline.md` +- `eval/rag-retrieval/README.md` +- `mvp/architecture/rag-eval-closure.md` +- `src/test/java/com/superbiz/agent/eval/RagLookupSnapshotGeneratorTest.java` +- `src/test/java/com/superbiz/agent/eval/RagEvalSeedImporterTest.java` +- `scripts/generate_rag_lookup_snapshots.ps1` +- `scripts/prepare_rag_eval_seed.ps1` +- `eval/rag-retrieval/seed-docs/*.md` +- `src/main/java/com/superbiz/agent/dto/Frontmatter.java` +- `src/main/java/com/superbiz/agent/dto/KnowledgeEntry.java` +- `src/main/java/com/superbiz/agent/service/FrontmatterParser.java` +- `src/main/java/com/superbiz/agent/service/KnowledgeIndexService.java` +- `src/main/java/com/superbiz/agent/service/DocumentManagementService.java` +- `src/main/java/com/superbiz/agent/service/VectorIndexService.java` +- `src/main/java/com/superbiz/agent/service/VectorSearchService.java` +- `src/main/java/com/superbiz/agent/service/SpringAiVectorStoreSidecarService.java` +- `src/main/resources/application.yml` + +## Baseline Result + +```text +Evaluated 7 cases: passRate=1.0, recall@5=1.0, failed=0 +``` + +## Regression Signals + +The evaluator now fails on: + +- missing expected source +- missing breadcrumb or evidence keyword +- non-`lookupResult` fixture +- selected attempt mismatch +- fallback reason mismatch +- fallback reason outside accepted values +- evidence status mismatch +- missing context source +- rerank top source mismatch + +The snapshot generator now provides: + +- real `LookupKnowledgeTool` invocation +- one fixture per golden case +- explicit opt-in through `rag.snapshot.enabled=true` +- optional post-generation baseline evaluation +- scoped retrieval through `retrieval.kb-scope=rag-eval` +- Spring AI VectorStore by default through `retrieval.vector-store.mode=spring` +- scoped L0 hints through the same `retrieval.kb-scope` + +The seed importer now provides: + +- canonical eval docs under `eval/rag-retrieval/seed-docs` +- real `DocumentManagementService` import/reindex +- stable `source`/`docId` metadata +- `kb_scope=rag-eval` isolation from local non-eval documents +- frontmatter stripping before chunk embedding +- an over-filter decoy seed doc for fallback-path evaluation + +The diff now detects: + +- aggregate pass/recall regression +- case pass regression +- hit-level regression +- first-rank regression +- selected attempt/fallback/evidence/rerank changes + +## Final Spring Live Snapshot + +```text +.\scripts\generate_rag_lookup_snapshots.ps1 -Fixtures -RetrievedAt 2026-07-06T00:00:00Z +Evaluated 7 cases: passRate=1.0, recall@5=1.0, failed=0 +``` + +Key fallback trace: + +```text +selectedAttempt=UNFILTERED_VECTOR_RETRY +fallbackReason=filtered_vector_no_evidence +rerankTopSource=rag-l0-filter-fallback +``` diff --git a/eval/rag-retrieval/README.md b/eval/rag-retrieval/README.md index 97e7367..604c48c 100644 --- a/eval/rag-retrieval/README.md +++ b/eval/rag-retrieval/README.md @@ -12,12 +12,59 @@ post-processing, or Spring AI VectorStore integration. ```text eval/rag-retrieval/ cases/golden-cases.json Fixed retrieval golden cases - fixtures/*.json Saved retrieval candidates for each case + seed-docs/*.md Canonical docs imported into the live KB for real-tool eval + fixtures/*.json Saved retrieval fixtures for each case reports/baseline.json Machine-readable baseline report reports/baseline.md Human-readable baseline report + reports/baseline-diff.* Optional diff reports reports/live-post-reindex.* Optional live acceptance reports ``` +## Seed Docs + Import/Reindex + +The live-tool eval uses canonical seed documents so the real +`LookupKnowledgeTool` can retrieve stable evidence from MySQL/Milvus instead of +whatever ad hoc documents happen to exist in the local knowledge base. + +Seed documents live in: + +```text +eval/rag-retrieval/seed-docs/*.md +``` + +Each seed doc uses frontmatter fields that are propagated into vector metadata: + +```yaml +source: mysql-connection-pool +breadcrumb: Database > MySQL > Connection Pool +kb_scope: rag-eval +``` + +Import or reindex the seed docs through the real upload pipeline: + +```powershell +.\scripts\prepare_rag_eval_seed.ps1 +``` + +The script runs `RagEvalSeedImporterTest` with `rag.seed.enabled=true`. It +deletes the existing document with the same `source`/`docId`, uploads the seed +doc through `DocumentManagementService`, updates DB metadata and L0, and rebuilds +Milvus chunks. + +`kb_scope` isolates eval data: + +- default application config leaves `retrieval.kb-scope` empty, so legacy docs + without `kb_scope` remain searchable; +- eval scripts pass `-Dretrieval.kb-scope=rag-eval`, so L0 query hints and L1 + vector retrieval both use only the canonical eval seed docs; +- the fallback retry skips only the L0 category filter, not the `kb_scope` + boundary. + +Frontmatter is not embedded as chunk content during upload. It feeds metadata, +L0, and document enrichment; only the Markdown body is chunked and embedded. +This keeps controlled L0 decoys from becoming semantically relevant just because +their frontmatter keywords matched the query. + ## Run From the repository root: @@ -36,6 +83,106 @@ python scripts/eval_rag_retrieval.py \ --markdown-report eval/rag-retrieval/reports/baseline.md ``` +## Generate Fixtures From LookupKnowledgeTool + +Use the snapshot generator when fixtures should reflect the real +`LookupKnowledgeTool` pipeline: + +```powershell +.\scripts\generate_rag_lookup_snapshots.ps1 +``` + +For the intended live loop, run seed import first: + +```powershell +.\scripts\prepare_rag_eval_seed.ps1 +.\scripts\generate_rag_lookup_snapshots.ps1 +python scripts\eval_rag_retrieval.py +``` + +The script runs a Spring test harness: + +```text +mvn -q -Dtest=RagLookupSnapshotGeneratorTest -Drag.snapshot.enabled=true -Dretrieval.kb-scope=rag-eval -Dretrieval.vector-store.mode=spring test +``` + +The generator reads `golden-cases.json`, injects the real `LookupKnowledgeTool` +bean, calls `lookupKnowledge(query)` for each case, writes +`fixtures/{caseId}.json`, and then runs `eval_rag_retrieval.py` unless +`-SkipEval` is provided. It defaults to Spring AI VectorStore mode; pass +`-VectorStoreMode sdk` only when intentionally comparing the legacy SDK path. + +Custom paths are supported: + +```powershell +.\scripts\generate_rag_lookup_snapshots.ps1 ` + -Cases eval\rag-retrieval\cases\golden-cases.json ` + -Fixtures eval\rag-retrieval\fixtures ` + -RetrievedAt 2026-07-06T00:00:00Z +``` + +The generator is disabled in normal test runs. It only executes when +`rag.snapshot.enabled=true` is provided because it writes repository files and +depends on the configured runtime retrieval stack. + +If generated fixtures fail the offline baseline, treat that as a real alignment +signal: either the golden expectations need to be adjusted to the current +knowledge base, or the knowledge base/indexing path needs to be fixed. + +## Modular RAG Contract + +Fixtures must use the current `lookupResult` shape, which mirrors the +`lookup_knowledge` output: + +```text +lookupResult.evidenceBlocks +lookupResult.contextPack +lookupResult.retrievalTrace +lookupResult.rerankTrace +``` + +Golden cases can assert both retrieval quality and pipeline behavior: + +- `expectedSources` / `expectedDocIds` +- `expectedBreadcrumbs` +- `expectedKeywords` +- `expectedSelectedAttempt` +- `expectedFallbackReason` +- `expectedFallbackReasons` +- `expectedEvidenceStatus` +- `expectedContextSources` +- `expectedRerankTopSource` + +This lets the baseline catch regressions such as losing the expected evidence +source, skipping context packing, changing the selected retrieval attempt, or +breaking the filtered-vector to unfiltered-retry fallback. + +## Baseline Diff + +To compare a freshly generated report against an existing baseline: + +```bash +python scripts/eval_rag_retrieval.py \ + --json-report eval/rag-retrieval/reports/current.json \ + --markdown-report eval/rag-retrieval/reports/current.md \ + --compare-to eval/rag-retrieval/reports/baseline.json \ + --diff-json-report eval/rag-retrieval/reports/baseline-diff.json \ + --diff-markdown-report eval/rag-retrieval/reports/baseline-diff.md +``` + +The diff reports aggregate regressions and case-level changes for: + +- pass rate, recall@K, strong hit rate, miss count +- pass state +- hit level +- first expected rank +- selected attempt +- fallback reason +- evidence status +- rerank top source + +The command exits non-zero when a case fails or the diff contains a regression. + ## Hit Levels - `strong`: expected document is found and breadcrumb or evidence keyword coverage is satisfied. @@ -81,6 +228,6 @@ GET /api/search/similar ``` It writes JSON and Markdown reports with query, topK, result count, top -candidates, breadcrumb, score labels, and raw response fields. This is a live +results, breadcrumb, score labels, and raw response fields. This is a live smoke check for environment readiness and post-reindex behavior; it does not replace the deterministic offline baseline above. diff --git a/eval/rag-retrieval/cases/golden-cases.json b/eval/rag-retrieval/cases/golden-cases.json index abbd785..573cf78 100644 --- a/eval/rag-retrieval/cases/golden-cases.json +++ b/eval/rag-retrieval/cases/golden-cases.json @@ -8,8 +8,14 @@ "scenario": "chat", "query": "MySQL connection pool is exhausted. How should I diagnose it?", "expectedDocIds": ["mysql-connection-pool"], + "expectedSources": ["mysql-connection-pool"], "expectedBreadcrumbs": ["Database > MySQL > Connection Pool"], "expectedKeywords": ["connection pool", "max_connections", "HikariCP"], + "expectedSelectedAttempt": "FILTERED_VECTOR", + "expectedFallbackReason": null, + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["mysql-connection-pool"], + "expectedRerankTopSource": "mysql-connection-pool", "notes": "Covers precise database troubleshooting retrieval." }, { @@ -17,8 +23,14 @@ "scenario": "chat", "query": "What is the standard troubleshooting flow for an application incident?", "expectedDocIds": ["incident-diagnosis-flow"], + "expectedSources": ["incident-diagnosis-flow"], "expectedBreadcrumbs": ["AIOps > Diagnosis Flow"], "expectedKeywords": ["collect evidence", "verify", "remediation"], + "expectedSelectedAttempt": "FILTERED_VECTOR", + "expectedFallbackReason": null, + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["incident-diagnosis-flow"], + "expectedRerankTopSource": "incident-diagnosis-flow", "notes": "Covers process-style knowledge where breadcrumb matters." }, { @@ -26,8 +38,14 @@ "scenario": "aiops", "query": "Alert HighLatency on payment-service with p95 latency above threshold", "expectedDocIds": ["payment-service-latency"], + "expectedSources": ["payment-service-latency"], "expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"], "expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"], + "expectedSelectedAttempt": "FILTERED_VECTOR", + "expectedFallbackReason": null, + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["payment-service-latency"], + "expectedRerankTopSource": "payment-service-latency", "notes": "Covers alert payload terms that should become retrieval hints." }, { @@ -35,8 +53,14 @@ "scenario": "aiops", "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", "expectedDocIds": ["aiops-alert-scope-control"], + "expectedSources": ["aiops-alert-scope-control"], "expectedBreadcrumbs": ["AIOps > Alert Scope Control"], "expectedKeywords": ["payload", "unrelated active alerts", "scope"], + "expectedSelectedAttempt": "FILTERED_VECTOR", + "expectedFallbackReason": null, + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["aiops-alert-scope-control"], + "expectedRerankTopSource": "aiops-alert-scope-control", "notes": "Covers scoped alert diagnosis behavior." }, { @@ -44,8 +68,14 @@ "scenario": "chat", "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", "expectedDocIds": ["rag-chunk-context-reconstruction"], + "expectedSources": ["rag-chunk-context-reconstruction"], "expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"], "expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"], + "expectedSelectedAttempt": "FILTERED_VECTOR", + "expectedFallbackReason": null, + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["rag-chunk-context-reconstruction"], + "expectedRerankTopSource": "rag-chunk-context-reconstruction", "notes": "Covers the known RAG refactor issue around context reconstruction." }, { @@ -53,9 +83,30 @@ "scenario": "chat", "query": "Should L0 keyword matching decide the final retrieval result?", "expectedDocIds": ["rag-l0-domain-entity-hint"], + "expectedSources": ["rag-l0-domain-entity-hint"], "expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"], "expectedKeywords": ["domain detector", "entity extractor", "metadata filter"], + "expectedSelectedAttempt": "FILTERED_VECTOR", + "expectedFallbackReason": null, + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["rag-l0-domain-entity-hint"], + "expectedRerankTopSource": "rag-l0-domain-entity-hint", "notes": "Covers the target L0 role after refactor." + }, + { + "caseId": "chat-l0-filter-fallback", + "scenario": "chat", + "query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?", + "expectedDocIds": ["rag-l0-filter-fallback"], + "expectedSources": ["rag-l0-filter-fallback"], + "expectedBreadcrumbs": ["RAG > Fallback > Unfiltered Retry"], + "expectedKeywords": ["skip the L0 filter", "unfiltered vector retry", "low quality"], + "expectedSelectedAttempt": "UNFILTERED_VECTOR_RETRY", + "expectedFallbackReasons": ["filtered_vector_low_quality", "filtered_vector_no_evidence"], + "expectedEvidenceStatus": "supported", + "expectedContextSources": ["rag-l0-filter-fallback"], + "expectedRerankTopSource": "rag-l0-filter-fallback", + "notes": "Covers the MVP fallback rule: if filtered L1 is low quality, retry raw query without L0 filter." } ] } diff --git a/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json b/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json index c35edfa..fed0432 100644 --- a/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json +++ b/eval/rag-retrieval/fixtures/aiops-payment-latency-alert.json @@ -1,25 +1,81 @@ { "caseId": "aiops-payment-latency-alert", "query": "Alert HighLatency on payment-service with p95 latency above threshold", - "retrievedAt": "2026-07-05T00:00:00Z", - "candidates": [ - { - "rank": 1, - "docId": "payment-service-latency", - "title": "Payment Service Latency Alert Playbook", - "breadcrumb": "AIOps > Service Alerts > Payment Latency", - "content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.", - "score": 0.84, - "retrievalLayer": "L1" + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "payment-service-latency", + "title": "Payment Service Latency Alert Playbook", + "breadcrumb": "AIOps > Service Alerts > Payment Latency", + "retrievalLayer": "L1", + "content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.", + "score": 0.84, + "hitReasons": ["domain_match:+0.15", "entity_match:+0.20", "keyword_match:+0.10"] + }, + { + "source": "mysql-connection-pool", + "title": "MySQL Connection Pool Troubleshooting", + "breadcrumb": "Database > MySQL > Connection Pool", + "retrievalLayer": "L1", + "content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.", + "score": 0.68, + "hitReasons": ["keyword_match:+0.10"] + } + ], + "contextPack": { + "packedText": "[1] Payment Service Latency Alert Playbook\nAIOps > Service Alerts > Payment Latency\nFor payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 236, + "includedSources": ["payment-service-latency", "mysql-connection-pool"], + "omittedSources": [] }, - { - "rank": 2, - "docId": "mysql-connection-pool", - "title": "MySQL Connection Pool Troubleshooting", - "breadcrumb": "Database > MySQL > Connection Pool", - "content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.", - "score": 0.68, - "retrievalLayer": "L1" + "retrievalTrace": { + "originalQuery": "Alert HighLatency on payment-service with p95 latency above threshold", + "rewrittenQuery": "HighLatency payment-service p95 latency alert downstream dependency diagnosis", + "categoryFilter": "AIOps", + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "queryHints": { + "domains": ["AIOps"], + "matched_keywords": ["p95 latency", "payment-service", "downstream dependency"], + "entities": ["payment-service", "HighLatency"], + "l0_titles": ["Payment Service Latency Alert Playbook"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "HighLatency payment-service p95 latency alert downstream dependency diagnosis", + "categoryFilter": "AIOps", + "candidateCount": 2, + "usable": true, + "durationMs": 11, + "topScore": 0.84, + "topSimilarity": 0.84 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "payment-service-latency", + "baseScore": 0.84, + "finalScore": 1.29, + "boostReasons": ["domain_match:+0.15", "entity_match:+0.20", "keyword_match:+0.10"] + }, + { + "finalRank": 2, + "source": "mysql-connection-pool", + "baseScore": 0.68, + "finalScore": 0.78, + "boostReasons": ["keyword_match:+0.10"] + } + ] } - ] + } } diff --git a/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json b/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json index 6603515..495323a 100644 --- a/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json +++ b/eval/rag-retrieval/fixtures/aiops-prometheus-alert-scope.json @@ -1,16 +1,65 @@ { "caseId": "aiops-prometheus-alert-scope", "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", - "retrievedAt": "2026-07-05T00:00:00Z", - "candidates": [ - { - "rank": 1, - "docId": "aiops-alert-scope-control", - "title": "AIOps Alert Scope Control", - "breadcrumb": "AIOps > Alert Scope Control", - "content": "When payload mode is active, queryPrometheusAlerts can verify the supplied alert, but unrelated active alerts must remain scoped context and should not become full diagnoses.", - "score": 0.9, - "retrievalLayer": "L0+L1" + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "aiops-alert-scope-control", + "title": "AIOps Alert Scope Control", + "breadcrumb": "AIOps > Alert Scope Control", + "retrievalLayer": "L1", + "content": "When payload mode is used, diagnose the input alert payload and do not expand unrelated active alerts into the main diagnosis scope.", + "score": 0.88, + "hitReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + } + ], + "contextPack": { + "packedText": "[1] AIOps Alert Scope Control\nAIOps > Alert Scope Control\nWhen payload mode is used, diagnose the input alert payload and do not expand unrelated active alerts into the main diagnosis scope.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 188, + "includedSources": ["aiops-alert-scope-control"], + "omittedSources": [] + }, + "retrievalTrace": { + "originalQuery": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", + "rewrittenQuery": "AIOps alert payload scope unrelated active alerts diagnosis", + "categoryFilter": "AIOps", + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "queryHints": { + "domains": ["AIOps"], + "matched_keywords": ["payload", "unrelated active alerts", "scope"], + "entities": ["alert payload"], + "l0_titles": ["AIOps Alert Scope Control"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "AIOps alert payload scope unrelated active alerts diagnosis", + "categoryFilter": "AIOps", + "candidateCount": 1, + "usable": true, + "durationMs": 8, + "topScore": 0.88, + "topSimilarity": 0.88 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "aiops-alert-scope-control", + "baseScore": 0.88, + "finalScore": 1.13, + "boostReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + } + ] } - ] + } } diff --git a/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json b/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json index 7229583..5b3a50f 100644 --- a/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json +++ b/eval/rag-retrieval/fixtures/chat-diagnosis-flow.json @@ -1,25 +1,81 @@ { "caseId": "chat-diagnosis-flow", "query": "What is the standard troubleshooting flow for an application incident?", - "retrievedAt": "2026-07-05T00:00:00Z", - "candidates": [ - { - "rank": 1, - "docId": "incident-diagnosis-flow", - "title": "Incident Diagnosis Flow", - "breadcrumb": "AIOps > Diagnosis Flow", - "content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.", - "score": 0.82, - "retrievalLayer": "L1" + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "incident-diagnosis-flow", + "title": "Incident Diagnosis Flow", + "breadcrumb": "AIOps > Diagnosis Flow", + "retrievalLayer": "L1", + "content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.", + "score": 0.82, + "hitReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "source": "rag-chunk-context-reconstruction", + "title": "RAG Chunk Context Reconstruction", + "breadcrumb": "RAG > Chunking > Context Reconstruction", + "retrievalLayer": "L1", + "content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.", + "score": 0.55, + "hitReasons": [] + } + ], + "contextPack": { + "packedText": "[1] Incident Diagnosis Flow\nAIOps > Diagnosis Flow\nThe standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 192, + "includedSources": ["incident-diagnosis-flow", "rag-chunk-context-reconstruction"], + "omittedSources": [] }, - { - "rank": 2, - "docId": "rag-chunk-context-reconstruction", - "title": "RAG Chunk Context Reconstruction", - "breadcrumb": "RAG > Chunking > Context Reconstruction", - "content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.", - "score": 0.55, - "retrievalLayer": "L1" + "retrievalTrace": { + "originalQuery": "What is the standard troubleshooting flow for an application incident?", + "rewrittenQuery": "standard application incident troubleshooting flow collect evidence verify remediation", + "categoryFilter": "AIOps", + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "queryHints": { + "domains": ["AIOps"], + "matched_keywords": ["collect evidence", "verify", "remediation"], + "entities": ["application incident"], + "l0_titles": ["Incident Diagnosis Flow"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "standard application incident troubleshooting flow collect evidence verify remediation", + "categoryFilter": "AIOps", + "candidateCount": 2, + "usable": true, + "durationMs": 10, + "topScore": 0.82, + "topSimilarity": 0.82 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "incident-diagnosis-flow", + "baseScore": 0.82, + "finalScore": 1.07, + "boostReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "finalRank": 2, + "source": "rag-chunk-context-reconstruction", + "baseScore": 0.55, + "finalScore": 0.55, + "boostReasons": [] + } + ] } - ] + } } diff --git a/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json b/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json index 2e24c15..74e4a1b 100644 --- a/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json +++ b/eval/rag-retrieval/fixtures/chat-l0-domain-hint.json @@ -1,25 +1,81 @@ { "caseId": "chat-l0-domain-hint", "query": "Should L0 keyword matching decide the final retrieval result?", - "retrievedAt": "2026-07-05T00:00:00Z", - "candidates": [ - { - "rank": 1, - "docId": "rag-l0-domain-entity-hint", - "title": "RAG L0 Domain Entity Hint", - "breadcrumb": "RAG > L0 > Domain Entity Hint", - "content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.", - "score": 0.88, - "retrievalLayer": "L0" + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "rag-l0-domain-entity-hint", + "title": "RAG L0 Domain Entity Hint", + "breadcrumb": "RAG > L0 > Domain Entity Hint", + "retrievalLayer": "L1", + "content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.", + "score": 0.88, + "hitReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "source": "rag-l0-l1-fusion-ranking", + "title": "RAG L0 L1 Fusion Ranking", + "breadcrumb": "RAG > Ranking > Fusion", + "retrievalLayer": "L1", + "content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.", + "score": 0.75, + "hitReasons": ["domain_match:+0.15"] + } + ], + "contextPack": { + "packedText": "[1] RAG L0 Domain Entity Hint\nRAG > L0 > Domain Entity Hint\nL0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 219, + "includedSources": ["rag-l0-domain-entity-hint", "rag-l0-l1-fusion-ranking"], + "omittedSources": [] }, - { - "rank": 2, - "docId": "rag-l0-l1-fusion-ranking", - "title": "RAG L0 L1 Fusion Ranking", - "breadcrumb": "RAG > Ranking > Fusion", - "content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.", - "score": 0.75, - "retrievalLayer": "L1" + "retrievalTrace": { + "originalQuery": "Should L0 keyword matching decide the final retrieval result?", + "rewrittenQuery": "RAG L0 keyword matching domain entity hint final retrieval decision", + "categoryFilter": "RAG", + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "queryHints": { + "domains": ["RAG"], + "matched_keywords": ["domain detector", "entity extractor", "metadata filter"], + "entities": ["L0"], + "l0_titles": ["RAG L0 Domain Entity Hint"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "RAG L0 keyword matching domain entity hint final retrieval decision", + "categoryFilter": "RAG", + "candidateCount": 2, + "usable": true, + "durationMs": 9, + "topScore": 0.88, + "topSimilarity": 0.88 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "rag-l0-domain-entity-hint", + "baseScore": 0.88, + "finalScore": 1.13, + "boostReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "finalRank": 2, + "source": "rag-l0-l1-fusion-ranking", + "baseScore": 0.75, + "finalScore": 0.9, + "boostReasons": ["domain_match:+0.15"] + } + ] } - ] + } } diff --git a/eval/rag-retrieval/fixtures/chat-l0-filter-fallback.json b/eval/rag-retrieval/fixtures/chat-l0-filter-fallback.json new file mode 100644 index 0000000..9b9092b --- /dev/null +++ b/eval/rag-retrieval/fixtures/chat-l0-filter-fallback.json @@ -0,0 +1,91 @@ +{ + "caseId": "chat-l0-filter-fallback", + "query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?", + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "rag-l0-filter-fallback", + "title": "RAG L0 Filter Fallback", + "breadcrumb": "RAG > Fallback > Unfiltered Retry", + "retrievalLayer": "L1", + "content": "When filtered vector retrieval is low quality, skip the L0 filter and run an unfiltered vector retry with the raw query before returning no evidence.", + "score": 0.83, + "hitReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "source": "rag-l0-domain-entity-hint", + "title": "RAG L0 Domain Entity Hint", + "breadcrumb": "RAG > L0 > Domain Entity Hint", + "retrievalLayer": "L1", + "content": "L0 supplies hints for metadata filtering and explanation, but it should not be treated as final fact evidence.", + "score": 0.66, + "hitReasons": ["domain_match:+0.15"] + } + ], + "contextPack": { + "packedText": "[1] RAG L0 Filter Fallback\nRAG > Fallback > Unfiltered Retry\nWhen filtered vector retrieval is low quality, skip the L0 filter and run an unfiltered vector retry with the raw query before returning no evidence.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 214, + "includedSources": ["rag-l0-filter-fallback", "rag-l0-domain-entity-hint"], + "omittedSources": [] + }, + "retrievalTrace": { + "originalQuery": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?", + "rewrittenQuery": "RAG L0 filtered vector low quality fallback unfiltered retry", + "categoryFilter": "RAG", + "selectedAttempt": "UNFILTERED_VECTOR_RETRY", + "fallbackReason": "filtered_vector_low_quality", + "evidenceStatus": "supported", + "queryHints": { + "domains": ["RAG"], + "matched_keywords": ["L0", "low quality", "unfiltered vector retry"], + "entities": ["L0"], + "l0_titles": ["RAG L0 Domain Entity Hint"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "RAG L0 filtered vector low quality fallback unfiltered retry", + "categoryFilter": "RAG", + "candidateCount": 1, + "usable": false, + "durationMs": 7, + "topScore": 1.35, + "topSimilarity": 0.325 + }, + { + "name": "UNFILTERED_VECTOR_RETRY", + "query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?", + "categoryFilter": null, + "candidateCount": 2, + "usable": true, + "durationMs": 13, + "topScore": 0.83, + "topSimilarity": 0.83 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "rag-l0-filter-fallback", + "baseScore": 0.83, + "finalScore": 1.08, + "boostReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "finalRank": 2, + "source": "rag-l0-domain-entity-hint", + "baseScore": 0.66, + "finalScore": 0.81, + "boostReasons": ["domain_match:+0.15"] + } + ] + } + } +} diff --git a/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json b/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json index 954306b..ed9d450 100644 --- a/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json +++ b/eval/rag-retrieval/fixtures/chat-mysql-connection-pool.json @@ -1,25 +1,81 @@ { "caseId": "chat-mysql-connection-pool", "query": "MySQL connection pool is exhausted. How should I diagnose it?", - "retrievedAt": "2026-07-05T00:00:00Z", - "candidates": [ - { - "rank": 1, - "docId": "mysql-connection-pool", - "title": "MySQL Connection Pool Troubleshooting", - "breadcrumb": "Database > MySQL > Connection Pool", - "content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.", - "score": 0.86, - "retrievalLayer": "L0+L1" + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "mysql-connection-pool", + "title": "MySQL Connection Pool Troubleshooting", + "breadcrumb": "Database > MySQL > Connection Pool", + "retrievalLayer": "L1", + "content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.", + "score": 0.86, + "hitReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "source": "incident-diagnosis-flow", + "title": "Incident Diagnosis Flow", + "breadcrumb": "AIOps > Diagnosis Flow", + "retrievalLayer": "L1", + "content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.", + "score": 0.61, + "hitReasons": [] + } + ], + "contextPack": { + "packedText": "[1] MySQL Connection Pool Troubleshooting\nDatabase > MySQL > Connection Pool\nWhen the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 216, + "includedSources": ["mysql-connection-pool", "incident-diagnosis-flow"], + "omittedSources": [] }, - { - "rank": 2, - "docId": "incident-diagnosis-flow", - "title": "Incident Diagnosis Flow", - "breadcrumb": "AIOps > Diagnosis Flow", - "content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.", - "score": 0.61, - "retrievalLayer": "L1" + "retrievalTrace": { + "originalQuery": "MySQL connection pool is exhausted. How should I diagnose it?", + "rewrittenQuery": "MySQL connection pool exhausted HikariCP max_connections diagnosis", + "categoryFilter": "Database", + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "queryHints": { + "domains": ["Database", "MySQL"], + "matched_keywords": ["connection pool", "HikariCP", "max_connections"], + "entities": ["MySQL", "HikariCP"], + "l0_titles": ["MySQL Connection Pool Troubleshooting"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "MySQL connection pool exhausted HikariCP max_connections diagnosis", + "categoryFilter": "Database", + "candidateCount": 2, + "usable": true, + "durationMs": 12, + "topScore": 0.86, + "topSimilarity": 0.86 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "mysql-connection-pool", + "baseScore": 0.86, + "finalScore": 1.11, + "boostReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "finalRank": 2, + "source": "incident-diagnosis-flow", + "baseScore": 0.61, + "finalScore": 0.61, + "boostReasons": [] + } + ] } - ] + } } diff --git a/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json b/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json index 5e9fcc9..5aa00e6 100644 --- a/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json +++ b/eval/rag-retrieval/fixtures/chat-rag-chunk-context.json @@ -1,25 +1,81 @@ { "caseId": "chat-rag-chunk-context", "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", - "retrievedAt": "2026-07-05T00:00:00Z", - "candidates": [ - { - "rank": 1, - "docId": "rag-chunk-context-reconstruction", - "title": "RAG Chunk Context Reconstruction", - "breadcrumb": "RAG > Chunking > Context Reconstruction", - "content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.", - "score": 0.79, - "retrievalLayer": "L1" + "retrievedAt": "2026-07-06T00:00:00Z", + "lookupResult": { + "found": true, + "evidenceBlocks": [ + { + "source": "rag-chunk-context-reconstruction", + "title": "RAG Chunk Context Reconstruction", + "breadcrumb": "RAG > Chunking > Context Reconstruction", + "retrievalLayer": "L1", + "content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.", + "score": 0.79, + "hitReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "source": "rag-breadcrumb-embedding-gap", + "title": "RAG Breadcrumb Embedding Gap", + "breadcrumb": "RAG > Embedding > Breadcrumb", + "retrievalLayer": "L1", + "content": "Embedding title and breadcrumb with content helps recover section semantics.", + "score": 0.72, + "hitReasons": ["domain_match:+0.15"] + } + ], + "contextPack": { + "packedText": "[1] RAG Chunk Context Reconstruction\nRAG > Chunking > Context Reconstruction\nAfter a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.", + "strategy": "top_evidence_blocks", + "charBudget": 3500, + "usedChars": 203, + "includedSources": ["rag-chunk-context-reconstruction", "rag-breadcrumb-embedding-gap"], + "omittedSources": [] }, - { - "rank": 2, - "docId": "rag-breadcrumb-embedding-gap", - "title": "RAG Breadcrumb Embedding Gap", - "breadcrumb": "RAG > Embedding > Breadcrumb", - "content": "Embedding title and breadcrumb with content helps recover section semantics.", - "score": 0.72, - "retrievalLayer": "L1" + "retrievalTrace": { + "originalQuery": "If a long section is split into multiple chunks, how do we keep retrieval context?", + "rewrittenQuery": "RAG chunk context reconstruction neighbor chunk same section breadcrumb", + "categoryFilter": "RAG", + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "queryHints": { + "domains": ["RAG"], + "matched_keywords": ["neighbor chunk", "same section", "breadcrumb"], + "entities": ["chunk", "breadcrumb"], + "l0_titles": ["RAG Chunk Context Reconstruction"], + "l0_match_count": 1 + }, + "attempts": [ + { + "name": "FILTERED_VECTOR", + "query": "RAG chunk context reconstruction neighbor chunk same section breadcrumb", + "categoryFilter": "RAG", + "candidateCount": 2, + "usable": true, + "durationMs": 9, + "topScore": 0.79, + "topSimilarity": 0.79 + } + ] + }, + "rerankTrace": { + "items": [ + { + "finalRank": 1, + "source": "rag-chunk-context-reconstruction", + "baseScore": 0.79, + "finalScore": 1.04, + "boostReasons": ["domain_match:+0.15", "keyword_match:+0.10"] + }, + { + "finalRank": 2, + "source": "rag-breadcrumb-embedding-gap", + "baseScore": 0.72, + "finalScore": 0.87, + "boostReasons": ["domain_match:+0.15"] + } + ] } - ] + } } diff --git a/eval/rag-retrieval/reports/baseline.json b/eval/rag-retrieval/reports/baseline.json index a4eaa8e..4a4f8d8 100644 --- a/eval/rag-retrieval/reports/baseline.json +++ b/eval/rag-retrieval/reports/baseline.json @@ -1,11 +1,15 @@ { - "generatedAt": "2026-07-04T17:59:52.172759+00:00", + "generatedAt": "2026-07-06T13:37:59.726351+00:00", "caseFile": "eval/rag-retrieval/cases/golden-cases.json", "fixtureDir": "eval/rag-retrieval/fixtures", "aggregate": { - "caseCount": 6, + "caseCount": 7, "topK": 5, - "strongHitCount": 6, + "passedCount": 7, + "failedCount": 0, + "passRate": 1.0, + "lookupResultCaseCount": 7, + "strongHitCount": 7, "mediumHitCount": 0, "weakHitCount": 0, "missCount": 0, @@ -18,6 +22,7 @@ "caseId": "chat-mysql-connection-pool", "scenario": "chat", "query": "MySQL connection pool is exhausted. How should I diagnose it?", + "dataShape": "lookupResult", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, @@ -31,12 +36,22 @@ "hikaricp" ], "breadcrumbMatched": true, + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "includedSources": [ + "mysql-connection-pool", + "incident-diagnosis-flow" + ], + "omittedSources": [], + "rerankTopSource": "mysql-connection-pool", "failedChecks": [] }, { "caseId": "chat-diagnosis-flow", "scenario": "chat", "query": "What is the standard troubleshooting flow for an application incident?", + "dataShape": "lookupResult", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, @@ -50,12 +65,22 @@ "remediation" ], "breadcrumbMatched": true, + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "includedSources": [ + "incident-diagnosis-flow", + "rag-chunk-context-reconstruction" + ], + "omittedSources": [], + "rerankTopSource": "incident-diagnosis-flow", "failedChecks": [] }, { "caseId": "aiops-payment-latency-alert", "scenario": "aiops", "query": "Alert HighLatency on payment-service with p95 latency above threshold", + "dataShape": "lookupResult", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, @@ -69,12 +94,22 @@ "downstream dependency" ], "breadcrumbMatched": true, + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "includedSources": [ + "payment-service-latency", + "mysql-connection-pool" + ], + "omittedSources": [], + "rerankTopSource": "payment-service-latency", "failedChecks": [] }, { "caseId": "aiops-prometheus-alert-scope", "scenario": "aiops", "query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?", + "dataShape": "lookupResult", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, @@ -87,12 +122,21 @@ "scope" ], "breadcrumbMatched": true, + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "includedSources": [ + "aiops-alert-scope-control" + ], + "omittedSources": [], + "rerankTopSource": "aiops-alert-scope-control", "failedChecks": [] }, { "caseId": "chat-rag-chunk-context", "scenario": "chat", "query": "If a long section is split into multiple chunks, how do we keep retrieval context?", + "dataShape": "lookupResult", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, @@ -106,12 +150,22 @@ "breadcrumb" ], "breadcrumbMatched": true, + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "includedSources": [ + "rag-chunk-context-reconstruction", + "rag-breadcrumb-embedding-gap" + ], + "omittedSources": [], + "rerankTopSource": "rag-chunk-context-reconstruction", "failedChecks": [] }, { "caseId": "chat-l0-domain-hint", "scenario": "chat", "query": "Should L0 keyword matching decide the final retrieval result?", + "dataShape": "lookupResult", "hitLevel": "strong", "passed": true, "firstExpectedRank": 1, @@ -125,6 +179,44 @@ "metadata filter" ], "breadcrumbMatched": true, + "selectedAttempt": "FILTERED_VECTOR", + "fallbackReason": null, + "evidenceStatus": "supported", + "includedSources": [ + "rag-l0-domain-entity-hint", + "rag-l0-l1-fusion-ranking" + ], + "omittedSources": [], + "rerankTopSource": "rag-l0-domain-entity-hint", + "failedChecks": [] + }, + { + "caseId": "chat-l0-filter-fallback", + "scenario": "chat", + "query": "RAG query was over-filtered by L0 and filtered vector search returned low quality evidence. What should happen?", + "dataShape": "lookupResult", + "hitLevel": "strong", + "passed": true, + "firstExpectedRank": 1, + "topCandidates": [ + "1:rag-l0-filter-fallback", + "2:rag-l0-domain-entity-hint" + ], + "matchedKeywords": [ + "skip the l0 filter", + "unfiltered vector retry", + "low quality" + ], + "breadcrumbMatched": true, + "selectedAttempt": "UNFILTERED_VECTOR_RETRY", + "fallbackReason": "filtered_vector_low_quality", + "evidenceStatus": "supported", + "includedSources": [ + "rag-l0-filter-fallback", + "rag-l0-domain-entity-hint" + ], + "omittedSources": [], + "rerankTopSource": "rag-l0-filter-fallback", "failedChecks": [] } ] diff --git a/eval/rag-retrieval/reports/baseline.md b/eval/rag-retrieval/reports/baseline.md index 7b9a403..fd177fc 100644 --- a/eval/rag-retrieval/reports/baseline.md +++ b/eval/rag-retrieval/reports/baseline.md @@ -1,16 +1,20 @@ # RAG Retrieval Baseline -Generated at: `2026-07-04T17:59:52.172759+00:00` +Generated at: `2026-07-06T13:37:59.726351+00:00` ## Aggregate | Metric | Value | |---|---:| -| Cases | 6 | +| Cases | 7 | | Top K | 5 | +| Passed | 7 | +| Failed | 0 | +| Pass rate | 1.0 | +| LookupResult fixtures | 7 | | Recall@K | 1.0 | | Strong hit rate | 1.0 | -| Strong hits | 6 | +| Strong hits | 7 | | Medium hits | 0 | | Weak hits | 0 | | Misses | 0 | @@ -18,11 +22,12 @@ Generated at: `2026-07-04T17:59:52.172759+00:00` ## Cases -| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks | -|---|---|---|---:|---|---| -| chat-mysql-connection-pool | chat | strong | 1 | 1:mysql-connection-pool
2:incident-diagnosis-flow | | -| chat-diagnosis-flow | chat | strong | 1 | 1:incident-diagnosis-flow
2:rag-chunk-context-reconstruction | | -| aiops-payment-latency-alert | aiops | strong | 1 | 1:payment-service-latency
2:mysql-connection-pool | | -| aiops-prometheus-alert-scope | aiops | strong | 1 | 1:aiops-alert-scope-control | | -| chat-rag-chunk-context | chat | strong | 1 | 1:rag-chunk-context-reconstruction
2:rag-breadcrumb-embedding-gap | | -| chat-l0-domain-hint | chat | strong | 1 | 1:rag-l0-domain-entity-hint
2:rag-l0-l1-fusion-ranking | | +| Case | Scenario | Pass | Hit | Attempt | Fallback | Evidence | First Expected Rank | Top Candidates | Failed Checks | +|---|---|---|---|---|---|---|---:|---|---| +| chat-mysql-connection-pool | chat | true | strong | FILTERED_VECTOR | | supported | 1 | 1:mysql-connection-pool
2:incident-diagnosis-flow | | +| chat-diagnosis-flow | chat | true | strong | FILTERED_VECTOR | | supported | 1 | 1:incident-diagnosis-flow
2:rag-chunk-context-reconstruction | | +| aiops-payment-latency-alert | aiops | true | strong | FILTERED_VECTOR | | supported | 1 | 1:payment-service-latency
2:mysql-connection-pool | | +| aiops-prometheus-alert-scope | aiops | true | strong | FILTERED_VECTOR | | supported | 1 | 1:aiops-alert-scope-control | | +| chat-rag-chunk-context | chat | true | strong | FILTERED_VECTOR | | supported | 1 | 1:rag-chunk-context-reconstruction
2:rag-breadcrumb-embedding-gap | | +| chat-l0-domain-hint | chat | true | strong | FILTERED_VECTOR | | supported | 1 | 1:rag-l0-domain-entity-hint
2:rag-l0-l1-fusion-ranking | | +| chat-l0-filter-fallback | chat | true | strong | UNFILTERED_VECTOR_RETRY | filtered_vector_low_quality | supported | 1 | 1:rag-l0-filter-fallback
2:rag-l0-domain-entity-hint | | diff --git a/eval/rag-retrieval/seed-docs/aiops-alert-scope-control.md b/eval/rag-retrieval/seed-docs/aiops-alert-scope-control.md new file mode 100644 index 0000000..7c0412e --- /dev/null +++ b/eval/rag-retrieval/seed-docs/aiops-alert-scope-control.md @@ -0,0 +1,26 @@ +--- +title: AIOps Alert Scope Control +keywords: [alert payload, unrelated active alerts, scope control] +summary: Keep diagnosis scoped to the request payload and avoid diagnosing unrelated active alerts. +category: aiops +source: aiops-alert-scope-control +breadcrumb: AIOps > Alert Scope Control +kb_scope: rag-eval +covers: [alert scope, payload, active alerts] +when_to_retrieve: Use when an AIOps request includes a concrete alert payload and scope boundaries matter. +--- + +# AIOps + +## Alert Scope Control + +When an AIOps request already includes an alert payload, the agent should diagnose that payload first. +It must not expand the task into unrelated active alerts unless the user asks for broad alert triage. + +Scope rules: + +1. Treat the provided payload as the primary incident boundary. +2. Use unrelated active alerts only as correlation evidence when they share service, dependency, time window, or trace context. +3. Do not replace the requested alert with a louder but unrelated alert. + +This runbook anchors payload, unrelated active alerts, and scope behavior. diff --git a/eval/rag-retrieval/seed-docs/incident-diagnosis-flow.md b/eval/rag-retrieval/seed-docs/incident-diagnosis-flow.md new file mode 100644 index 0000000..461f596 --- /dev/null +++ b/eval/rag-retrieval/seed-docs/incident-diagnosis-flow.md @@ -0,0 +1,27 @@ +--- +title: Incident Diagnosis Flow +keywords: [standard troubleshooting flow, application incident, collect evidence, verify, remediation] +summary: Standard flow for diagnosing application incidents with evidence, hypothesis verification, and remediation. +category: ops +source: incident-diagnosis-flow +breadcrumb: AIOps > Diagnosis Flow +kb_scope: rag-eval +covers: [incident diagnosis, evidence collection, remediation] +when_to_retrieve: Use when the user asks for a standard troubleshooting flow or incident diagnosis sequence. +--- + +# AIOps + +## Diagnosis Flow + +The standard troubleshooting flow is evidence first, hypothesis second, remediation last. + +Recommended sequence: + +1. Collect evidence from alerts, metrics, logs, traces, deployments, and recent configuration changes. +2. Define a small hypothesis that explains the observed symptoms. +3. Verify the hypothesis with a targeted metric, log query, or reproduction step. +4. Choose remediation that directly addresses the verified cause. +5. Record the outcome and the evidence used to make the decision. + +Do not skip collect evidence, verify, and remediation ordering during an application incident. diff --git a/eval/rag-retrieval/seed-docs/mysql-connection-pool.md b/eval/rag-retrieval/seed-docs/mysql-connection-pool.md new file mode 100644 index 0000000..2c3e9af --- /dev/null +++ b/eval/rag-retrieval/seed-docs/mysql-connection-pool.md @@ -0,0 +1,29 @@ +--- +title: MySQL Connection Pool Runbook +keywords: [MySQL connection pool, pool exhausted, max_connections, HikariCP] +summary: Diagnose exhausted MySQL connection pools and distinguish application leaks from database limits. +category: database +source: mysql-connection-pool +breadcrumb: Database > MySQL > Connection Pool +kb_scope: rag-eval +covers: [mysql, connection pool, database capacity] +when_to_retrieve: Use when MySQL clients report exhausted pools, connection acquisition timeout, max_connections pressure, or HikariCP saturation. +--- + +# Database + +## MySQL + +### Connection Pool + +When MySQL connection pool is exhausted, first compare application pool usage with database `max_connections`. +For HikariCP, check `active`, `idle`, `pending`, and connection acquisition timeout metrics. + +Recommended diagnosis: + +1. Verify whether HikariCP active connections stay near maximum while pending threads grow. +2. Check MySQL `Threads_connected`, `Threads_running`, and `max_connections`. +3. Inspect slow SQL and long transactions that keep connections checked out. +4. If the database is healthy, look for application connection leaks or missing transaction boundaries. + +Use this runbook as evidence for connection pool, max_connections, and HikariCP incidents. diff --git a/eval/rag-retrieval/seed-docs/payment-service-latency.md b/eval/rag-retrieval/seed-docs/payment-service-latency.md new file mode 100644 index 0000000..ddc2a89 --- /dev/null +++ b/eval/rag-retrieval/seed-docs/payment-service-latency.md @@ -0,0 +1,28 @@ +--- +title: Payment Service Latency Alert +keywords: [HighLatency, payment-service, p95 latency, downstream dependency] +summary: Diagnose payment-service p95 latency alerts and identify downstream dependency bottlenecks. +category: aiops +source: payment-service-latency +breadcrumb: AIOps > Service Alerts > Payment Latency +kb_scope: rag-eval +covers: [payment-service, latency, downstream dependency] +when_to_retrieve: Use when an alert mentions payment-service, HighLatency, or elevated p95 latency. +--- + +# AIOps + +## Service Alerts + +### Payment Latency + +For `HighLatency` alerts on `payment-service`, treat p95 latency as the primary symptom. + +Diagnosis steps: + +1. Confirm whether p95 latency is isolated to payment-service or shared across upstream callers. +2. Compare payment-service latency with downstream dependency latency for gateway, risk, and order services. +3. Check connection pool wait time, retry spikes, and timeout rates. +4. If downstream dependency latency increased first, classify payment-service as affected rather than root cause. + +The expected evidence terms are p95 latency, payment-service, and downstream dependency. diff --git a/eval/rag-retrieval/seed-docs/rag-chunk-context-reconstruction.md b/eval/rag-retrieval/seed-docs/rag-chunk-context-reconstruction.md new file mode 100644 index 0000000..3aa69a8 --- /dev/null +++ b/eval/rag-retrieval/seed-docs/rag-chunk-context-reconstruction.md @@ -0,0 +1,28 @@ +--- +title: RAG Chunk Context Reconstruction +keywords: [split into multiple chunks, retrieval context, neighbor chunk, same section, breadcrumb context] +summary: Preserve context when long RAG sections are split into multiple retrievable chunks. +category: rag +source: rag-chunk-context-reconstruction +breadcrumb: RAG > Chunking > Context Reconstruction +kb_scope: rag-eval +covers: [rag chunking, context packing, breadcrumbs] +when_to_retrieve: Use when a retrieval question asks how to preserve context across split chunks. +--- + +# RAG + +## Chunking + +### Context Reconstruction + +When a long section is split into multiple chunks, retrieval should keep enough local structure for the answer. + +Recommended behavior: + +1. Store the breadcrumb with every chunk. +2. Preserve the same section identity across adjacent chunks. +3. During context packing, include a neighbor chunk when the selected chunk depends on nearby setup or definitions. +4. Prefer concise evidence blocks that show the breadcrumb and the relevant content span. + +The key concepts are neighbor chunk, same section, and breadcrumb. diff --git a/eval/rag-retrieval/seed-docs/rag-l0-domain-entity-hint.md b/eval/rag-retrieval/seed-docs/rag-l0-domain-entity-hint.md new file mode 100644 index 0000000..75c7875 --- /dev/null +++ b/eval/rag-retrieval/seed-docs/rag-l0-domain-entity-hint.md @@ -0,0 +1,29 @@ +--- +title: RAG L0 Domain Entity Hint +keywords: [L0 keyword matching, final retrieval result, domain detector, entity extractor, metadata filter] +summary: Define L0 as a query transformation hint layer instead of final retrieval evidence. +category: rag +source: rag-l0-domain-entity-hint +breadcrumb: RAG > L0 > Domain Entity Hint +kb_scope: rag-eval +covers: [l0 hint, query transformation, metadata filter] +when_to_retrieve: Use when a question asks whether L0 should decide final retrieval or only provide hints. +--- + +# RAG + +## L0 + +### Domain Entity Hint + +L0 keyword matching should not decide the final retrieval result. +In the modular RAG pipeline, L0 behaves like a lightweight domain detector and entity extractor. + +The output can provide: + +1. Candidate domain hints. +2. Matched entities and keywords. +3. An optional metadata filter for the first vector retrieval attempt. + +Final evidence still comes from L1 vector retrieval, post-retrieval normalization, rerank, and context packing. +The important terms are domain detector, entity extractor, and metadata filter. diff --git a/eval/rag-retrieval/seed-docs/rag-l0-filter-decoy.md b/eval/rag-retrieval/seed-docs/rag-l0-filter-decoy.md new file mode 100644 index 0000000..4189ad2 --- /dev/null +++ b/eval/rag-retrieval/seed-docs/rag-l0-filter-decoy.md @@ -0,0 +1,19 @@ +--- +title: RAG L0 Filter Decoy +keywords: [over-filtered by L0, filtered vector search, low quality evidence] +summary: Decoy document used to force the first filtered retrieval attempt into a low-quality category. +category: overfilter-decoy +source: rag-l0-filter-decoy +breadcrumb: RAG > Fallback > Decoy +kb_scope: rag-eval +covers: [fallback test decoy] +when_to_retrieve: Use only as a controlled eval decoy for over-filter fallback testing. +--- + +# Release Calendar + +## Approval Window + +This document describes an unrelated release calendar approval window. +It intentionally avoids the real fallback instructions so the filtered retrieval +attempt is low quality and the retriever must retry without the L0 category filter. diff --git a/eval/rag-retrieval/seed-docs/rag-l0-filter-fallback.md b/eval/rag-retrieval/seed-docs/rag-l0-filter-fallback.md new file mode 100644 index 0000000..71c2335 --- /dev/null +++ b/eval/rag-retrieval/seed-docs/rag-l0-filter-fallback.md @@ -0,0 +1,25 @@ +--- +title: RAG L0 Filter Fallback +keywords: [golden retry contract, second pass retrieval] +summary: Retry the raw query without the L0 category filter when filtered vector evidence is missing or low quality. +category: fallback +source: rag-l0-filter-fallback +breadcrumb: RAG > Fallback > Unfiltered Retry +kb_scope: rag-eval +covers: [fallback, unfiltered retry, retrieval quality] +when_to_retrieve: Use when validating the fallback contract for low-quality filtered vector retrieval. +--- + +# RAG + +## Fallback + +### Unfiltered Retry + +If the first vector search is over-constrained by an L0 metadata filter and returns low quality evidence, +the retriever should skip the L0 filter and run an unfiltered vector retry with the original query. + +The fallback reason should be `filtered_vector_low_quality` when the filtered candidate exists but is below the +reference threshold. If there is no usable evidence at all, use `filtered_vector_no_evidence`. + +This document is the expected evidence for skip the L0 filter, unfiltered vector retry, and low quality behavior. diff --git a/mvp/architecture/README.md b/mvp/architecture/README.md index a0fe3a9..05446ae 100644 --- a/mvp/architecture/README.md +++ b/mvp/architecture/README.md @@ -18,6 +18,7 @@ | [harness-quality-gates.md](harness-quality-gates.md) | Prompt、Hook、Trace、Verifier、评测基线组成的质量门禁 | | [rag-architecture.md](rag-architecture.md) | RAG/知识检索新架构,覆盖 L0 hint、VectorStore 主路径、SDK fallback、证据追踪 | | [modular-rag-pipeline.md](modular-rag-pipeline.md) | `lookup_knowledge` 模块化 RAG 落地架构,覆盖 pipeline、fallback、evidence-first contract、trace | +| [rag-eval-closure.md](rag-eval-closure.md) | RAG 评测闭环,覆盖 offline baseline、baseline diff、diagnosis eval 和 live acceptance | | [retrieval-observability.md](retrieval-observability.md) | 检索运行细节和可观测性,覆盖 L0/L1、去重、分数归一、评测 | | [feedback-architecture.md](feedback-architecture.md) | 反馈与自评估闭环,覆盖 rule evaluation、Verifier、AIOps rule、用户反馈和案例沉淀 | | [session-trace-lifecycle.md](session-trace-lifecycle.md) | 会话和 Trace 生命周期,覆盖 sessionId、状态流转、agent_step、tool_invocation、Trace API | @@ -37,8 +38,9 @@ SuperBizAgent MVP 是一个面向故障诊断的可追踪 Agent 系统:Chat 4. 然后读 [harness-quality-gates.md](harness-quality-gates.md),理解为什么系统可追踪、可验证。 5. 再读 [rag-architecture.md](rag-architecture.md),理解当前 RAG 为什么保留显式 `lookup_knowledge`,以及 Spring AI VectorStore 如何接入。 6. 继续读 [modular-rag-pipeline.md](modular-rag-pipeline.md),看 `lookup_knowledge` 的模块化落地和 evidence-first contract。 -7. 然后读 [retrieval-observability.md](retrieval-observability.md),看检索细节和质量回归方式。 -8. 再读 [feedback-architecture.md](feedback-architecture.md),理解 self_evaluation、用户反馈和案例沉淀。 -9. 按需读 [session-trace-lifecycle.md](session-trace-lifecycle.md)、[knowledge-base-authoring.md](knowledge-base-authoring.md)、[data-model.md](data-model.md),补齐运行生命周期、知识库维护和数据关系。 -10. 最后读 [evolution-roadmap.md](evolution-roadmap.md),区分后续演进和当前实现。 -11. 需要追溯旧方案时,再进入 `archive/2026-07-05-legacy/`。 +7. 再读 [rag-eval-closure.md](rag-eval-closure.md),看 RAG baseline 如何形成质量闭环。 +8. 然后读 [retrieval-observability.md](retrieval-observability.md),看检索细节和质量回归方式。 +9. 再读 [feedback-architecture.md](feedback-architecture.md),理解 self_evaluation、用户反馈和案例沉淀。 +10. 按需读 [session-trace-lifecycle.md](session-trace-lifecycle.md)、[knowledge-base-authoring.md](knowledge-base-authoring.md)、[data-model.md](data-model.md),补齐运行生命周期、知识库维护和数据关系。 +11. 最后读 [evolution-roadmap.md](evolution-roadmap.md),区分后续演进和当前实现。 +12. 需要追溯旧方案时,再进入 `archive/2026-07-05-legacy/`。 diff --git a/mvp/architecture/rag-architecture.md b/mvp/architecture/rag-architecture.md index 200e8b0..21eba38 100644 --- a/mvp/architecture/rag-architecture.md +++ b/mvp/architecture/rag-architecture.md @@ -37,7 +37,7 @@ flowchart TD Mode -->|auto| SpringTry["try Spring AI VectorStore"] SpringTry -->|success| Results["SearchResult list"] SpringTry -->|failure| SdkFallback["Milvus SDK fallback"] - Mode -->|spring-ai| SpringOnly["Spring AI VectorStore only"] + Mode -->|spring / spring-ai| SpringOnly["Spring AI VectorStore only"] Mode -->|sdk| SdkOnly["Milvus SDK only"] SpringOnly --> Results @@ -66,7 +66,7 @@ Agent Executor -> mode=auto -> Spring AI VectorStore -> fallback: Milvus SDK - -> mode=spring-ai + -> mode=spring / spring-ai -> Spring AI VectorStore only -> mode=sdk -> Milvus SDK only @@ -134,7 +134,7 @@ Executor -> LookupKnowledgeTool -> VectorSearchService `VectorSearchService` 是当前检索门面: - `auto`:优先 Spring AI VectorStore,失败后 fallback 到 SDK。 -- `spring-ai`:只走 Spring AI VectorStore。 +- `spring` / `spring-ai`:只走 Spring AI VectorStore。 - `sdk`:只走原 Milvus SDK。 这样可以在不改 Agent 工具的情况下切换检索实现,并支持线上验证和回退。 @@ -374,7 +374,7 @@ RAG 架构变更必须先过评测,再认为可合入主链路。 - `lookup_knowledge` 保持显式 Agent Tool。 - L0 降级为 domain/entity hint。 - L1 默认执行语义检索。 -- `VectorSearchService` 支持 `auto`、`spring-ai`、`sdk` 三种模式。 +- `VectorSearchService` 支持 `auto`、`spring`/`spring-ai`、`sdk` 三种模式。 - Spring AI VectorStore 成为读取主路径。 - Milvus SDK fallback 保留。 - 分数语义拆成 `score`、`rawScore`、`scoreLabel`。 diff --git a/mvp/architecture/rag-eval-closure.md b/mvp/architecture/rag-eval-closure.md new file mode 100644 index 0000000..7c8231f --- /dev/null +++ b/mvp/architecture/rag-eval-closure.md @@ -0,0 +1,181 @@ +# RAG 评测闭环架构 + +**更新日期**:2026-07-06 + +本文记录当前 RAG 质量闭环。它的目标不是证明检索“永远正确”,而是让每次改 `lookup_knowledge`、L0 hint、向量召回、post-retrieval、rerank 或 context packing 时,都能得到可重复的回归信号。 + +## 1. 闭环分层 + +```text +RAG pipeline change + -> LookupKnowledgeTool snapshot generation + -> offline RAG retrieval baseline + -> RAG baseline diff + -> diagnosis eval baseline + -> diagnosis baseline diff + -> accept / fix / archive +``` + +| 层级 | 位置 | 作用 | +|---|---|---| +| RAG retrieval baseline | `eval/rag-retrieval/` | 检查固定 query 是否命中期望证据、路径和 fallback | +| RAG baseline diff | `scripts/eval_rag_retrieval.py --compare-to ...` | 对比当前报告和旧基线,输出 regression/change | +| Diagnosis eval baseline | `mvp/eval/` | 检查 Agent 最终诊断 trace、报告和证据行为 | +| Live acceptance | `scripts/eval_rag_live_acceptance.py` | 在应用和向量库运行后做真实环境 smoke check | + +## 2. Offline RAG Baseline + +核心资产: + +```text +eval/rag-retrieval/cases/golden-cases.json +eval/rag-retrieval/fixtures/*.json +eval/rag-retrieval/reports/baseline.json +eval/rag-retrieval/reports/baseline.md +scripts/eval_rag_retrieval.py +scripts/generate_rag_lookup_snapshots.ps1 +src/test/java/com/superbiz/agent/eval/RagLookupSnapshotGeneratorTest.java +``` + +运行: + +```powershell +python scripts\eval_rag_retrieval.py +``` + +该 baseline 完全离线,不依赖 MySQL、Redis、Milvus、LLM 或 Spring Boot。它适合在改 RAG 代码后快速判断: + +- 期望 source 是否仍在 topK 内。 +- breadcrumb 和 evidence keyword 是否仍能覆盖。 +- `LookupResult` 是否仍包含 `evidenceBlocks/contextPack/retrievalTrace/rerankTrace`。 +- selected attempt 是否符合预期。 +- fallback reason 是否符合预期。 +- context pack 是否包含期望 source。 +- rerank top source 是否稳定。 + +## 3. 模块化输出契约 + +fixture 必须使用当前模块化格式: + +```json +{ + "lookupResult": { + "evidenceBlocks": [], + "contextPack": {}, + "retrievalTrace": {}, + "rerankTrace": {} + } +} +``` + +当前 golden cases 直接以模块化格式为唯一契约,因为这个版本的目标是验证完整 RAG pipeline,而不只是验证候选召回。 + +## 4. Fallback Case + +当前 baseline 增加了 `chat-l0-filter-fallback`: + +```text +FILTERED_VECTOR low quality or no evidence + -> UNFILTERED_VECTOR_RETRY + -> fallbackReason = filtered_vector_low_quality | filtered_vector_no_evidence +``` + +这个 case 固化了 MVP 版本的降级策略:如果经过 L0 filter 后 L1 低质量或没有证据,就跳过 L0 filter,用原始 query 再做一次无过滤向量检索。不同向量后端对“低质量候选”和“无候选”的边界可能不同,所以 golden case 允许两个 fallback reason,但强制要求 retry 行为和最终证据正确。 + +## 5. Diff 闭环 + +生成当前报告并与旧基线对比: + +```powershell +python scripts\eval_rag_retrieval.py ` + --json-report eval\rag-retrieval\reports\current.json ` + --markdown-report eval\rag-retrieval\reports\current.md ` + --compare-to eval\rag-retrieval\reports\baseline.json ` + --diff-json-report eval\rag-retrieval\reports\baseline-diff.json ` + --diff-markdown-report eval\rag-retrieval\reports\baseline-diff.md +``` + +diff 会检查: + +- pass rate +- recall@K +- strong hit rate +- miss count +- case pass state +- hit level +- first expected rank +- selected attempt +- fallback reason +- evidence status +- rerank top source + +当 case 失败或 diff 出现 regression 时,脚本会返回非 0 退出码,可作为本地质量门禁或 CI 门禁。 + +## 6. 与 Diagnosis Eval 的关系 + +RAG baseline 解决的是“证据有没有被正确检索、处理和打包”。 + +Diagnosis eval 解决的是“Agent 有没有把证据用于最终诊断,并保持 trace 可解释”。 + +两者不是替代关系: + +- 改 RAG pipeline:先跑 RAG baseline,再跑相关 Agent 测试。 +- 改 prompt、Agent 编排、Verifier:重点跑 diagnosis eval。 +- 改 embedding 输入、reindex、向量库配置:跑 RAG baseline + live acceptance。 + +## 7. 面试表达 + +可以概括为: + +> 我没有只做一个 RAG 调用,而是把 RAG 拆成 Query Transform、Retrieval、Post-Retrieval、Rerank、Context Packing,并为它建设了离线 golden cases、baseline report、baseline diff 和上层 diagnosis eval,形成可回放、可对比、可回归的 Agent 质量闭环。 + +## 8. LookupKnowledgeTool Snapshot + +真实工具快照生成命令: + +```powershell +.\scripts\generate_rag_lookup_snapshots.ps1 +``` + +该命令默认使用 `retrieval.vector-store.mode=spring`,通过 `RagLookupSnapshotGeneratorTest` 启动 Spring test context,注入真实 `LookupKnowledgeTool` bean,对 `golden-cases.json` 中每个 query 调用 `lookupKnowledge(query)`,并把返回的 `LookupResult` 写入 `eval/rag-retrieval/fixtures/{caseId}.json`。 + +普通测试不会执行快照生成器;只有显式传入 `rag.snapshot.enabled=true` 时才会写 fixture。 + +## 9. Seed Docs And Scope Isolation + +Live `LookupKnowledgeTool` snapshots are only stable if the expected documents +exist in the real knowledge base and vector index. The eval loop therefore adds +a canonical seed layer: + +```text +eval/rag-retrieval/seed-docs/*.md + -> scripts/prepare_rag_eval_seed.ps1 + -> RagEvalSeedImporterTest + -> DocumentManagementService.uploadDocument + -> api_document metadata + L0 index + Milvus chunks +``` + +Seed frontmatter includes: + +```yaml +source: mysql-connection-pool +breadcrumb: Database > MySQL > Connection Pool +kb_scope: rag-eval +``` + +`source` becomes the stable `docId` when it fits the DB column, and is also +written to vector metadata as `_source` and `source`. `breadcrumb` is copied into +chunk metadata so evidence blocks can keep a stable path. `kb_scope` isolates +eval documents from local production documents. + +Default runtime behavior keeps `retrieval.kb-scope` empty, so existing documents +without `kb_scope` are still searchable. Eval scripts pass +`-Dretrieval.kb-scope=rag-eval`, so L0 query hints, the filtered attempt, and +the unfiltered retry stay inside the eval corpus while the retry still skips the +L0 category filter. + +Frontmatter is used for DB metadata, L0 hints, and vector metadata. It is +stripped before document chunking so embedding content represents the Markdown +body, not the YAML control plane. This is important for fallback eval: a decoy +document may intentionally match L0 keywords, but its body should remain low +quality evidence so the retry path can be exercised. diff --git a/mvp/architecture/retrieval-observability.md b/mvp/architecture/retrieval-observability.md index d97a03b..d06f088 100644 --- a/mvp/architecture/retrieval-observability.md +++ b/mvp/architecture/retrieval-observability.md @@ -30,7 +30,7 @@ flowchart TD L1 --> Mode{"retrieval.vector-store.mode"} Mode -->|auto| Spring["Spring AI VectorStore"] Spring -->|failure| SDK["Milvus SDK fallback"] - Mode -->|spring-ai| Spring + Mode -->|spring / spring-ai| Spring Mode -->|sdk| SDK Spring --> Candidates["L1 candidates"] @@ -89,7 +89,7 @@ L1 通过 `VectorSearchService` 调度,支持三种模式: | 模式 | 行为 | 用途 | |---|---|---| | `auto` | 优先 Spring AI VectorStore,失败 fallback 到 SDK | 默认运行模式 | -| `spring-ai` | 只走 Spring AI VectorStore | 验证框架路径 | +| `spring` / `spring-ai` | 只走 Spring AI VectorStore | 验证框架路径 | | `sdk` | 只走 Milvus SDK | 对比旧链路或临时回退 | ### Spring AI VectorStore 路径 diff --git a/scripts/eval_rag_retrieval.py b/scripts/eval_rag_retrieval.py index c6c1e95..e9ebf7e 100644 --- a/scripts/eval_rag_retrieval.py +++ b/scripts/eval_rag_retrieval.py @@ -2,7 +2,8 @@ """Offline evaluator for RAG retrieval golden cases. The evaluator reads fixed golden cases and saved retrieval fixtures. It does not -call the running application or any external service. +call the running application or any external service. Fixtures must use the +modular `LookupResult` shape produced by lookup_knowledge. """ from __future__ import annotations @@ -19,6 +20,15 @@ DEFAULT_CASES = Path("eval/rag-retrieval/cases/golden-cases.json") DEFAULT_FIXTURES = Path("eval/rag-retrieval/fixtures") DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline.json") DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/baseline.md") +DEFAULT_DIFF_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline-diff.json") +DEFAULT_DIFF_MD_REPORT = Path("eval/rag-retrieval/reports/baseline-diff.md") + +HIT_LEVEL_RANK = { + "miss": 0, + "weak": 1, + "medium": 2, + "strong": 3, +} @dataclass @@ -34,12 +44,16 @@ class Candidate: @classmethod def from_json(cls, raw: dict[str, Any], fallback_rank: int) -> "Candidate": return cls( - rank=int(raw.get("rank") or fallback_rank), - doc_id=str(raw.get("docId") or raw.get("id") or ""), + rank=int(raw.get("rank") or raw.get("finalRank") or fallback_rank), + doc_id=str(raw.get("docId") or raw.get("source") or raw.get("id") or ""), title=str(raw.get("title") or ""), breadcrumb=str(raw.get("breadcrumb") or ""), - content=str(raw.get("content") or ""), - score=_optional_float(raw.get("score")), + content=str(raw.get("content") or raw.get("contentPreview") or ""), + score=_optional_float( + raw.get("score") + if raw.get("score") is not None + else raw.get("finalScore") + ), retrieval_layer=( str(raw.get("retrievalLayer")) if raw.get("retrievalLayer") is not None @@ -57,6 +71,18 @@ class Candidate: return f"{self.rank}:{label}" +@dataclass +class NormalizedFixture: + data_shape: str + candidates: list[Candidate] + selected_attempt: str | None + fallback_reason: str | None + evidence_status: str | None + included_sources: list[str] + omitted_sources: list[str] + rerank_top_source: str | None + + def _optional_float(value: Any) -> float | None: if value is None: return None @@ -88,10 +114,76 @@ def normalize_terms(values: list[Any]) -> list[str]: return [str(value).lower() for value in values if str(value).strip()] +def normalize_sources(values: list[Any]) -> list[str]: + return [str(value) for value in values if str(value).strip()] + + +def get_lookup_result(fixture: dict[str, Any]) -> dict[str, Any] | None: + lookup = fixture.get("lookupResult") + if isinstance(lookup, dict): + return lookup + if "evidenceBlocks" in fixture or "retrievalTrace" in fixture: + return fixture + return None + + +def normalize_fixture(fixture: dict[str, Any], top_k: int) -> NormalizedFixture: + lookup_result = get_lookup_result(fixture) + if lookup_result is None: + return NormalizedFixture( + data_shape="invalid", + candidates=[], + selected_attempt=None, + fallback_reason=None, + evidence_status=None, + included_sources=[], + omitted_sources=[], + rerank_top_source=None, + ) + + raw_blocks = lookup_result.get("evidenceBlocks") or [] + candidates = [ + Candidate.from_json(raw, index + 1) + for index, raw in enumerate(raw_blocks[:top_k]) + if isinstance(raw, dict) + ] + retrieval_trace = lookup_result.get("retrievalTrace") or {} + context_pack = lookup_result.get("contextPack") or {} + rerank_trace = lookup_result.get("rerankTrace") or {} + rerank_items = [ + item for item in rerank_trace.get("items", []) + if isinstance(item, dict) + ] + rerank_items.sort(key=lambda item: int(item.get("finalRank") or 999999)) + return NormalizedFixture( + data_shape="lookupResult", + candidates=candidates, + selected_attempt=optional_string(retrieval_trace.get("selectedAttempt")), + fallback_reason=optional_string(retrieval_trace.get("fallbackReason")), + evidence_status=optional_string(retrieval_trace.get("evidenceStatus")), + included_sources=normalize_sources(context_pack.get("includedSources") or []), + omitted_sources=normalize_sources(context_pack.get("omittedSources") or []), + rerank_top_source=( + optional_string(rerank_items[0].get("source")) + if rerank_items + else None + ), + ) + + +def optional_string(value: Any) -> str | None: + if value is None: + return None + text = str(value) + return text if text else None + + def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[str, Any]: case_id = str(case["caseId"]) fixture_path = fixture_dir / f"{case_id}.json" expected_doc_ids = normalize_terms(case.get("expectedDocIds", [])) + expected_sources = normalize_terms(case.get("expectedSources", [])) + expected_documents = expected_doc_ids or expected_sources expected_breadcrumbs = normalize_terms(case.get("expectedBreadcrumbs", [])) expected_keywords = normalize_terms(case.get("expectedKeywords", [])) @@ -100,25 +192,30 @@ def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[s "caseId": case_id, "scenario": case.get("scenario"), "query": case.get("query"), + "dataShape": None, "hitLevel": "miss", "passed": False, "firstExpectedRank": None, "topCandidates": [], + "matchedKeywords": [], + "breadcrumbMatched": False, + "selectedAttempt": None, + "fallbackReason": None, + "evidenceStatus": None, + "includedSources": [], + "omittedSources": [], + "rerankTopSource": None, "failedChecks": [f"missing fixture: {fixture_path.as_posix()}"], } - fixture = load_json(fixture_path) - raw_candidates = fixture.get("candidates", []) - candidates = [ - Candidate.from_json(raw, index + 1) - for index, raw in enumerate(raw_candidates[:top_k]) - ] + fixture = normalize_fixture(load_json(fixture_path), top_k) + candidates = fixture.candidates first_expected = None expected_doc_candidate = None for candidate in candidates: candidate_doc = candidate.doc_id.lower() - if any(expected == candidate_doc for expected in expected_doc_ids): + if any(expected == candidate_doc for expected in expected_documents): first_expected = candidate.rank expected_doc_candidate = candidate break @@ -149,6 +246,8 @@ def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[s if expected_keywords and not keyword_matches: failed_checks.append("expected evidence keywords not found") + failed_checks.extend(check_modular_contract(case, fixture)) + if expected_doc_candidate is not None and ( breadcrumb_match or bool(keyword_matches) ): @@ -164,16 +263,119 @@ def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[s "caseId": case_id, "scenario": case.get("scenario"), "query": case.get("query"), + "dataShape": fixture.data_shape, "hitLevel": hit_level, - "passed": hit_level in {"strong", "medium"}, + "passed": hit_level in {"strong", "medium"} and not failed_checks, "firstExpectedRank": first_expected, "topCandidates": [candidate.label() for candidate in candidates], "matchedKeywords": keyword_matches, "breadcrumbMatched": breadcrumb_match, + "selectedAttempt": fixture.selected_attempt, + "fallbackReason": fixture.fallback_reason, + "evidenceStatus": fixture.evidence_status, + "includedSources": fixture.included_sources, + "omittedSources": fixture.omitted_sources, + "rerankTopSource": fixture.rerank_top_source, "failedChecks": failed_checks, } +def check_modular_contract(case: dict[str, Any], fixture: NormalizedFixture) -> list[str]: + failed: list[str] = [] + if fixture.data_shape != "lookupResult": + failed.append("fixture must use lookupResult shape") + compare_expected( + failed, + case, + "expectedSelectedAttempt", + fixture.selected_attempt, + "selected attempt mismatch", + ) + if "expectedFallbackReasons" in case: + compare_expected_any( + failed, + case, + "expectedFallbackReasons", + fixture.fallback_reason, + "fallback reason mismatch", + ) + else: + compare_expected( + failed, + case, + "expectedFallbackReason", + fixture.fallback_reason, + "fallback reason mismatch", + ) + compare_expected( + failed, + case, + "expectedEvidenceStatus", + fixture.evidence_status, + "evidence status mismatch", + ) + compare_expected( + failed, + case, + "expectedRerankTopSource", + fixture.rerank_top_source, + "rerank top source mismatch", + ) + + expected_context_sources = normalize_sources(case.get("expectedContextSources", [])) + if expected_context_sources: + included = set(fixture.included_sources) + missing = [ + source for source in expected_context_sources + if source not in included + ] + if missing: + failed.append("expected context sources missing: " + ", ".join(missing)) + return failed + + +def compare_expected( + failed: list[str], + case: dict[str, Any], + field: str, + actual: str | None, + message: str, +) -> None: + if field not in case: + return + expected = case.get(field) + if expected is None: + if actual is not None: + failed.append(f"{message}: expected , got {actual}") + return + if str(expected) != str(actual): + failed.append(f"{message}: expected {expected}, got {actual or ''}") + + +def compare_expected_any( + failed: list[str], + case: dict[str, Any], + field: str, + actual: str | None, + message: str, +) -> None: + expected_values = case.get(field) + if not isinstance(expected_values, list): + failed.append(f"{field} must be a list") + return + normalized_expected = [ + None if value is None else str(value) + for value in expected_values + ] + normalized_actual = None if actual is None else str(actual) + if normalized_actual not in normalized_expected: + expected_text = ", ".join( + "" if value is None else value + for value in normalized_expected + ) + failed.append(f"{message}: expected one of [{expected_text}], got {actual or ''}") + + def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]: total = len(results) counts = { @@ -187,15 +389,23 @@ def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]: for item in results if item.get("firstExpectedRank") is not None ] - passed = counts["strong"] + counts["medium"] + retrieved = counts["strong"] + counts["medium"] + passed = sum(1 for item in results if item["passed"]) + lookup_result_cases = sum( + 1 for item in results if item.get("dataShape") == "lookupResult" + ) return { "caseCount": total, "topK": top_k, + "passedCount": passed, + "failedCount": total - passed, + "passRate": round(passed / total, 4) if total else 0, + "lookupResultCaseCount": lookup_result_cases, "strongHitCount": counts["strong"], "mediumHitCount": counts["medium"], "weakHitCount": counts["weak"], "missCount": counts["miss"], - "recallAtK": round(passed / total, 4) if total else 0, + "recallAtK": round(retrieved / total, 4) if total else 0, "strongHitRate": round(counts["strong"] / total, 4) if total else 0, "averageFirstHitRank": ( round(sum(expected_ranks) / len(expected_ranks), 4) @@ -218,6 +428,10 @@ def render_markdown(report: dict[str, Any]) -> str: "|---|---:|", f"| Cases | {metrics['caseCount']} |", f"| Top K | {metrics['topK']} |", + f"| Passed | {metrics['passedCount']} |", + f"| Failed | {metrics['failedCount']} |", + f"| Pass rate | {metrics['passRate']} |", + f"| LookupResult fixtures | {metrics['lookupResultCaseCount']} |", f"| Recall@K | {metrics['recallAtK']} |", f"| Strong hit rate | {metrics['strongHitRate']} |", f"| Strong hits | {metrics['strongHitCount']} |", @@ -228,18 +442,22 @@ def render_markdown(report: dict[str, Any]) -> str: "", "## Cases", "", - "| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |", - "|---|---|---|---:|---|---|", + "| Case | Scenario | Pass | Hit | Attempt | Fallback | Evidence | First Expected Rank | Top Candidates | Failed Checks |", + "|---|---|---|---|---|---|---|---:|---|---|", ] for item in report["results"]: failed = "
".join(item["failedChecks"]) if item["failedChecks"] else "" top = "
".join(item["topCandidates"]) first_rank = item["firstExpectedRank"] lines.append( - "| {case} | {scenario} | {hit} | {rank} | {top} | {failed} |".format( + "| {case} | {scenario} | {passed} | {hit} | {attempt} | {fallback} | {evidence} | {rank} | {top} | {failed} |".format( case=item["caseId"], scenario=item.get("scenario") or "", + passed=str(item["passed"]).lower(), hit=item["hitLevel"], + attempt=item.get("selectedAttempt") or "", + fallback=item.get("fallbackReason") or "", + evidence=item.get("evidenceStatus") or "", rank=first_rank if first_rank is not None else "", top=top, failed=failed, @@ -249,12 +467,255 @@ def render_markdown(report: dict[str, Any]) -> str: return "\n".join(lines) +def compare_reports(baseline: dict[str, Any], current: dict[str, Any]) -> dict[str, Any]: + items: list[dict[str, Any]] = [] + compare_metric(items, "aggregate", None, "passRate", baseline, current, higher_is_better=True) + compare_metric(items, "aggregate", None, "recallAtK", baseline, current, higher_is_better=True) + compare_metric(items, "aggregate", None, "strongHitRate", baseline, current, higher_is_better=True) + compare_metric(items, "aggregate", None, "missCount", baseline, current, higher_is_better=False) + compare_cases(items, baseline.get("results") or [], current.get("results") or []) + + regression_count = count_type(items, "REGRESSION") + improvement_count = count_type(items, "IMPROVEMENT") + changed_count = count_type(items, "CHANGED") + return { + "generatedAt": datetime.now(timezone.utc).isoformat(), + "baselineReport": baseline.get("caseFile"), + "currentReport": current.get("caseFile"), + "baselineCaseCount": get_aggregate_value(baseline, "caseCount"), + "currentCaseCount": get_aggregate_value(current, "caseCount"), + "baselinePassRate": get_aggregate_value(baseline, "passRate"), + "currentPassRate": get_aggregate_value(current, "passRate"), + "baselineRecallAtK": get_aggregate_value(baseline, "recallAtK"), + "currentRecallAtK": get_aggregate_value(current, "recallAtK"), + "regressionCount": regression_count, + "improvementCount": improvement_count, + "changedCount": changed_count, + "hasRegression": regression_count > 0, + "items": items, + } + + +def compare_metric( + items: list[dict[str, Any]], + scope: str, + case_id: str | None, + metric: str, + baseline: dict[str, Any], + current: dict[str, Any], + higher_is_better: bool, +) -> None: + baseline_value = get_aggregate_value(baseline, metric) + current_value = get_aggregate_value(current, metric) + if baseline_value == current_value: + return + if not isinstance(baseline_value, (int, float)) or not isinstance(current_value, (int, float)): + items.append(diff_item("CHANGED", scope, case_id, metric, baseline_value, current_value, None)) + return + delta = float(current_value) - float(baseline_value) + items.append(diff_item(classify_delta(delta, higher_is_better), scope, case_id, metric, + baseline_value, current_value, delta)) + + +def compare_cases( + items: list[dict[str, Any]], + baseline_results: list[dict[str, Any]], + current_results: list[dict[str, Any]], +) -> None: + baseline_by_id = by_case_id(baseline_results) + current_by_id = by_case_id(current_results) + case_ids = sorted(set(baseline_by_id) | set(current_by_id)) + for case_id in case_ids: + baseline = baseline_by_id.get(case_id) + current = current_by_id.get(case_id) + if baseline is None: + items.append(diff_item("CHANGED", "case", case_id, "casePresence", "missing", "present", None)) + continue + if current is None: + items.append(diff_item("REGRESSION", "case", case_id, "casePresence", "present", "missing", None)) + continue + compare_case_bool(items, case_id, "passed", baseline, current, higher_is_better=True) + compare_hit_level(items, case_id, baseline, current) + compare_case_rank(items, case_id, baseline, current) + compare_case_value(items, case_id, "selectedAttempt", baseline, current) + compare_case_value(items, case_id, "fallbackReason", baseline, current) + compare_case_value(items, case_id, "evidenceStatus", baseline, current) + compare_case_value(items, case_id, "rerankTopSource", baseline, current) + + +def compare_case_bool( + items: list[dict[str, Any]], + case_id: str, + metric: str, + baseline: dict[str, Any], + current: dict[str, Any], + higher_is_better: bool, +) -> None: + baseline_value = bool(baseline.get(metric)) + current_value = bool(current.get(metric)) + if baseline_value == current_value: + return + delta = int(current_value) - int(baseline_value) + items.append(diff_item(classify_delta(delta, higher_is_better), "case", case_id, metric, + baseline_value, current_value, float(delta))) + + +def compare_hit_level( + items: list[dict[str, Any]], + case_id: str, + baseline: dict[str, Any], + current: dict[str, Any], +) -> None: + baseline_value = baseline.get("hitLevel") + current_value = current.get("hitLevel") + if baseline_value == current_value: + return + delta = HIT_LEVEL_RANK.get(str(current_value), 0) - HIT_LEVEL_RANK.get(str(baseline_value), 0) + items.append(diff_item(classify_delta(delta, True), "case", case_id, "hitLevel", + baseline_value, current_value, float(delta))) + + +def compare_case_rank( + items: list[dict[str, Any]], + case_id: str, + baseline: dict[str, Any], + current: dict[str, Any], +) -> None: + baseline_value = baseline.get("firstExpectedRank") + current_value = current.get("firstExpectedRank") + if baseline_value == current_value: + return + if baseline_value is None or current_value is None: + change_type = "REGRESSION" if current_value is None else "IMPROVEMENT" + items.append(diff_item(change_type, "case", case_id, "firstExpectedRank", + baseline_value, current_value, None)) + return + delta = int(current_value) - int(baseline_value) + items.append(diff_item(classify_delta(delta, False), "case", case_id, "firstExpectedRank", + baseline_value, current_value, float(delta))) + + +def compare_case_value( + items: list[dict[str, Any]], + case_id: str, + metric: str, + baseline: dict[str, Any], + current: dict[str, Any], +) -> None: + baseline_value = baseline.get(metric) + current_value = current.get(metric) + if baseline_value == current_value: + return + items.append(diff_item("CHANGED", "case", case_id, metric, baseline_value, current_value, None)) + + +def diff_item( + change_type: str, + scope: str, + case_id: str | None, + metric: str, + baseline_value: Any, + current_value: Any, + delta: float | None, +) -> dict[str, Any]: + target = case_id or scope + return { + "type": change_type, + "scope": scope, + "caseId": case_id, + "metric": metric, + "baselineValue": value_label(baseline_value), + "currentValue": value_label(current_value), + "delta": delta, + "message": f"{target} {metric} changed", + } + + +def render_diff_markdown(report: dict[str, Any]) -> str: + lines = [ + "# RAG Retrieval Baseline Diff", + "", + f"Generated at: `{report['generatedAt']}`", + "", + "## Summary", + "", + "| Metric | Value |", + "|---|---:|", + f"| Baseline cases | {report['baselineCaseCount']} |", + f"| Current cases | {report['currentCaseCount']} |", + f"| Baseline pass rate | {report['baselinePassRate']} |", + f"| Current pass rate | {report['currentPassRate']} |", + f"| Baseline recall@K | {report['baselineRecallAtK']} |", + f"| Current recall@K | {report['currentRecallAtK']} |", + f"| Regressions | {report['regressionCount']} |", + f"| Improvements | {report['improvementCount']} |", + f"| Changed | {report['changedCount']} |", + "", + "## Items", + "", + "| Type | Scope | Case | Metric | Baseline | Current | Delta | Message |", + "|---|---|---|---|---|---|---:|---|", + ] + for item in report["items"]: + delta = "" if item.get("delta") is None else item["delta"] + lines.append( + "| {type} | {scope} | {case} | {metric} | {baseline} | {current} | {delta} | {message} |".format( + type=item["type"], + scope=item["scope"], + case=item.get("caseId") or "", + metric=item["metric"], + baseline=item["baselineValue"], + current=item["currentValue"], + delta=delta, + message=item["message"], + ) + ) + lines.append("") + return "\n".join(lines) + + +def get_aggregate_value(report: dict[str, Any], metric: str) -> Any: + return (report.get("aggregate") or {}).get(metric) + + +def by_case_id(results: list[dict[str, Any]]) -> dict[str, dict[str, Any]]: + return { + str(item.get("caseId")): item + for item in sorted(results, key=lambda item: str(item.get("caseId"))) + } + + +def classify_delta(delta: float, higher_is_better: bool) -> str: + if delta == 0.0: + return "CHANGED" + improved = delta > 0 if higher_is_better else delta < 0 + return "IMPROVEMENT" if improved else "REGRESSION" + + +def count_type(items: list[dict[str, Any]], change_type: str) -> int: + return sum(1 for item in items if item["type"] == change_type) + + +def value_label(value: Any) -> str: + if value is None: + return "-" + return str(value) + + def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--cases", type=Path, default=DEFAULT_CASES) parser.add_argument("--fixtures", type=Path, default=DEFAULT_FIXTURES) parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT) parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT) + parser.add_argument( + "--compare-to", + type=Path, + default=None, + help="Optional baseline report to diff against the newly generated report.", + ) + parser.add_argument("--diff-json-report", type=Path, default=DEFAULT_DIFF_JSON_REPORT) + parser.add_argument("--diff-markdown-report", type=Path, default=DEFAULT_DIFF_MD_REPORT) return parser.parse_args() @@ -274,16 +735,32 @@ def main() -> int: write_json(args.json_report, report) write_text(args.markdown_report, render_markdown(report)) - failed = [item for item in results if item["hitLevel"] == "miss"] + failed = [item for item in results if not item["passed"]] + diff_failed = False + if args.compare_to is not None: + baseline = load_json(args.compare_to) + diff = compare_reports(baseline, report) + write_json(args.diff_json_report, diff) + write_text(args.diff_markdown_report, render_diff_markdown(diff)) + diff_failed = bool(diff["hasRegression"]) + print( + "Diffed against {baseline}: regressions={regressions}, changes={changes}".format( + baseline=args.compare_to.as_posix(), + regressions=diff["regressionCount"], + changes=diff["changedCount"], + ) + ) + print( - "Evaluated {total} cases: recall@{top_k}={recall}, misses={misses}".format( + "Evaluated {total} cases: passRate={pass_rate}, recall@{top_k}={recall}, failed={failed}".format( total=len(results), + pass_rate=report["aggregate"]["passRate"], top_k=top_k, recall=report["aggregate"]["recallAtK"], - misses=len(failed), + failed=len(failed), ) ) - return 1 if failed else 0 + return 1 if failed or diff_failed else 0 if __name__ == "__main__": diff --git a/scripts/generate_rag_lookup_snapshots.ps1 b/scripts/generate_rag_lookup_snapshots.ps1 new file mode 100644 index 0000000..8e84fd8 --- /dev/null +++ b/scripts/generate_rag_lookup_snapshots.ps1 @@ -0,0 +1,34 @@ +param( + [string]$Cases = "eval\rag-retrieval\cases\golden-cases.json", + [string]$Fixtures = "eval\rag-retrieval\fixtures", + [string]$RetrievedAt = "", + [string]$KbScope = "rag-eval", + [string]$VectorStoreMode = "spring", + [switch]$SkipEval +) + +$ErrorActionPreference = "Stop" + +$mavenArgs = @( + "-q", + "-Dtest=RagLookupSnapshotGeneratorTest", + "-Drag.snapshot.enabled=true", + "-Drag.snapshot.cases=$Cases", + "-Drag.snapshot.fixtures=$Fixtures", + "-Dretrieval.kb-scope=$KbScope", + "-Dretrieval.vector-store.mode=$VectorStoreMode" +) + +if ($RetrievedAt -ne "") { + $mavenArgs += "-Drag.snapshot.retrievedAt=$RetrievedAt" +} + +$mavenArgs += "test" + +Write-Host "Generating RAG lookupResult fixtures from real LookupKnowledgeTool..." +& mvn @mavenArgs + +if (-not $SkipEval) { + Write-Host "Running offline RAG retrieval baseline..." + & python scripts\eval_rag_retrieval.py --cases $Cases --fixtures $Fixtures +} diff --git a/scripts/prepare_rag_eval_seed.ps1 b/scripts/prepare_rag_eval_seed.ps1 new file mode 100644 index 0000000..b407658 --- /dev/null +++ b/scripts/prepare_rag_eval_seed.ps1 @@ -0,0 +1,18 @@ +param( + [string]$SeedDocs = "eval\rag-retrieval\seed-docs", + [string]$KbScope = "rag-eval" +) + +$ErrorActionPreference = "Stop" + +$mavenArgs = @( + "-q", + "-Dtest=RagEvalSeedImporterTest", + "-Drag.seed.enabled=true", + "-Drag.seed.docs=$SeedDocs", + "-Dretrieval.kb-scope=$KbScope", + "test" +) + +Write-Host "Importing RAG eval seed docs through DocumentManagementService..." +& mvn @mavenArgs diff --git a/src/main/java/com/superbiz/agent/dto/Frontmatter.java b/src/main/java/com/superbiz/agent/dto/Frontmatter.java index 3d02554..5e97045 100644 --- a/src/main/java/com/superbiz/agent/dto/Frontmatter.java +++ b/src/main/java/com/superbiz/agent/dto/Frontmatter.java @@ -1,5 +1,6 @@ package com.superbiz.agent.dto; +import com.fasterxml.jackson.annotation.JsonProperty; import lombok.AllArgsConstructor; import lombok.Builder; import lombok.Data; @@ -39,6 +40,13 @@ public class Frontmatter { */ private String category; + private String source; + + private String breadcrumb; + + @JsonProperty("kb_scope") + private String kbScope; + /** * 章节锚点(预留字段,MVP 不使用) * Key: 章节标题,Value: 章节 Markdown 标题 diff --git a/src/main/java/com/superbiz/agent/dto/KnowledgeEntry.java b/src/main/java/com/superbiz/agent/dto/KnowledgeEntry.java index 2b89cf7..5335a73 100644 --- a/src/main/java/com/superbiz/agent/dto/KnowledgeEntry.java +++ b/src/main/java/com/superbiz/agent/dto/KnowledgeEntry.java @@ -39,6 +39,8 @@ public class KnowledgeEntry { */ private String category; + private String kbScope; + /** * 章节锚点(预留字段,MVP 不使用) */ diff --git a/src/main/java/com/superbiz/agent/service/DocumentManagementService.java b/src/main/java/com/superbiz/agent/service/DocumentManagementService.java index da5dbda..7ad1ecb 100644 --- a/src/main/java/com/superbiz/agent/service/DocumentManagementService.java +++ b/src/main/java/com/superbiz/agent/service/DocumentManagementService.java @@ -126,11 +126,13 @@ public class DocumentManagementService { // 5. 解析 frontmatter long frontmatterStart = System.currentTimeMillis(); Frontmatter frontmatter = null; + String bodyText = text; if (frontmatterParser.hasFrontmatter(text)) { frontmatter = frontmatterParser.parse(text); if (frontmatter != null) { // LLM 补全 covers / whenToRetrieve(已有值则跳过) - documentFieldEnricher.enrich(frontmatter, text, category); + bodyText = frontmatterParser.stripFrontmatter(text); + documentFieldEnricher.enrich(frontmatter, bodyText, category); log.info("解析到frontmatter: title={}, keywords={}, time={}ms", frontmatter.getTitle(), frontmatter.getKeywords(), System.currentTimeMillis() - frontmatterStart); } else { @@ -142,7 +144,7 @@ public class DocumentManagementService { // 6. 分块 long chunkStart = System.currentTimeMillis(); - List chunks = documentChunkService.chunkDocument(text, fileName); + List chunks = documentChunkService.chunkDocument(bodyText, fileName); if (chunks.isEmpty()) { throw new DocumentProcessException(fileName, "upload", "文档分块失败"); } @@ -150,7 +152,7 @@ public class DocumentManagementService { fileName, chunks.size(), System.currentTimeMillis() - chunkStart); // 7. 创建文档元数据 - String docId = UUID.randomUUID().toString(); + String docId = resolveDocumentId(frontmatter); String metadataJson = null; if (frontmatter != null) { try { @@ -181,7 +183,7 @@ public class DocumentManagementService { // 8. 向量化并索引 try { long vectorStart = System.currentTimeMillis(); - vectorIndexService.indexDocumentChunks(docId, chunks, category); + vectorIndexService.indexDocumentChunks(docId, chunks, category, frontmatter); document.setStatus("INDEXED"); document.setIndexedAt(LocalDateTime.now()); apiDocumentRepository.save(document); @@ -203,6 +205,7 @@ public class DocumentManagementService { .keywords(frontmatter.getKeywords()) .summary(frontmatter.getSummary()) .category(category) + .kbScope(frontmatter.getKbScope()) .sections(frontmatter.getSections()) .covers(frontmatter.getCovers()) .whenToRetrieve(frontmatter.getWhenToRetrieve()) @@ -311,6 +314,16 @@ public class DocumentManagementService { } } + private String resolveDocumentId(Frontmatter frontmatter) { + if (frontmatter != null && frontmatter.getSource() != null) { + String source = frontmatter.getSource().trim(); + if (!source.isEmpty() && source.length() <= 64) { + return source; + } + } + return UUID.randomUUID().toString(); + } + /** * 根据 docId 查询文档 */ diff --git a/src/main/java/com/superbiz/agent/service/FrontmatterParser.java b/src/main/java/com/superbiz/agent/service/FrontmatterParser.java index 9efcad6..350e921 100644 --- a/src/main/java/com/superbiz/agent/service/FrontmatterParser.java +++ b/src/main/java/com/superbiz/agent/service/FrontmatterParser.java @@ -62,6 +62,9 @@ public class FrontmatterParser { .keywords((java.util.List) map.get("keywords")) .summary((String) map.get("summary")) .category((String) map.get("category")) + .source((String) map.get("source")) + .breadcrumb((String) map.get("breadcrumb")) + .kbScope(firstString(map, "kb_scope", "kbScope")) .sections((Map) map.get("sections")) .version((String) map.get("version")) .author((String) map.get("author")) @@ -87,6 +90,35 @@ public class FrontmatterParser { } } + public String stripFrontmatter(String content) { + if (!hasFrontmatter(content)) { + return content; + } + + String trimmed = content.trim(); + int secondDelimiter = trimmed.indexOf("\n---", 3); + int delimiterLength = 4; + if (secondDelimiter == -1) { + secondDelimiter = trimmed.indexOf("\r\n---", 3); + delimiterLength = 5; + } + if (secondDelimiter == -1) { + return content; + } + + int bodyStart = secondDelimiter + delimiterLength; + if (bodyStart < trimmed.length()) { + char next = trimmed.charAt(bodyStart); + if (next == '\r') { + bodyStart++; + } + if (bodyStart < trimmed.length() && trimmed.charAt(bodyStart) == '\n') { + bodyStart++; + } + } + return trimmed.substring(Math.min(bodyStart, trimmed.length())).stripLeading(); + } + /** * 提取 frontmatter 文本(两个 --- 之间的内容) * @@ -115,4 +147,14 @@ public class FrontmatterParser { // 提取 frontmatter(不包含 --- 标记) return content.substring(3, secondDelimiter).trim(); } + + private String firstString(Map map, String... keys) { + for (String key : keys) { + Object value = map.get(key); + if (value instanceof String text && !text.isBlank()) { + return text; + } + } + return null; + } } diff --git a/src/main/java/com/superbiz/agent/service/KnowledgeIndexService.java b/src/main/java/com/superbiz/agent/service/KnowledgeIndexService.java index d702d9a..db347b8 100644 --- a/src/main/java/com/superbiz/agent/service/KnowledgeIndexService.java +++ b/src/main/java/com/superbiz/agent/service/KnowledgeIndexService.java @@ -36,6 +36,9 @@ public class KnowledgeIndexService { @Value("${knowledge.base-path:knowledge_base}") private String knowledgeBasePath; + @Value("${retrieval.kb-scope:}") + private String kbScope = ""; + @Autowired private ApiDocumentRepository apiDocumentRepository; @@ -114,6 +117,7 @@ public class KnowledgeIndexService { .keywords(frontmatter.getKeywords()) .summary(frontmatter.getSummary()) .category(frontmatter.getCategory()) + .kbScope(frontmatter.getKbScope()) .covers(frontmatter.getCovers()) .whenToRetrieve(frontmatter.getWhenToRetrieve()) .build(); @@ -144,6 +148,9 @@ public class KnowledgeIndexService { Set titles = new LinkedHashSet<>(); for (KnowledgeEntry entry : knowledgeIndex) { + if (!matchesConfiguredScope(entry)) { + continue; + } List entryMatchedKeywords = matchedKeywords(entry, queryLower); if (entryMatchedKeywords.isEmpty()) { continue; @@ -178,6 +185,21 @@ public class KnowledgeIndexService { return !matchedKeywords(entry, query).isEmpty(); } + private boolean matchesConfiguredScope(KnowledgeEntry entry) { + String scope = trimToNull(kbScope); + if (scope == null) { + return true; + } + return scope.equals(trimToNull(entry.getKbScope())); + } + + private String trimToNull(String value) { + if (value == null || value.isBlank()) { + return null; + } + return value.trim(); + } + private List matchedKeywords(KnowledgeEntry entry, String query) { if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) { return List.of(); diff --git a/src/main/java/com/superbiz/agent/service/SpringAiVectorStoreSidecarService.java b/src/main/java/com/superbiz/agent/service/SpringAiVectorStoreSidecarService.java index 362976b..d8911ed 100644 --- a/src/main/java/com/superbiz/agent/service/SpringAiVectorStoreSidecarService.java +++ b/src/main/java/com/superbiz/agent/service/SpringAiVectorStoreSidecarService.java @@ -8,6 +8,7 @@ import org.springframework.ai.document.Document; import org.springframework.ai.vectorstore.SearchRequest; import org.springframework.ai.vectorstore.VectorStore; import org.springframework.beans.factory.ObjectProvider; +import org.springframework.beans.factory.annotation.Value; import org.springframework.stereotype.Service; import java.util.ArrayList; @@ -21,6 +22,9 @@ public class SpringAiVectorStoreSidecarService { private final ObjectProvider vectorStoreProvider; private final RetrievalResultNormalizer normalizer; + @Value("${retrieval.kb-scope:}") + private String kbScope = ""; + public SpringAiVectorStoreSidecarService(RagSidecarProperties properties, ObjectProvider vectorStoreProvider, RetrievalResultNormalizer normalizer) { @@ -44,8 +48,9 @@ public class SpringAiVectorStoreSidecarService { .query(query) .topK(topK) .similarityThresholdAll(); - if (category != null && !category.isBlank()) { - builder.filterExpression("category == '" + escapeFilterValue(category) + "'"); + String filterExpression = buildFilterExpression(category); + if (filterExpression != null) { + builder.filterExpression(filterExpression); } List documents = vectorStore.similaritySearch(builder.build()); @@ -78,4 +83,24 @@ public class SpringAiVectorStoreSidecarService { private String escapeFilterValue(String value) { return value.replace("'", "\\'"); } + + String buildFilterExpression(String category) { + List parts = new ArrayList<>(); + String categoryFilter = trimToNull(category); + if (categoryFilter != null) { + parts.add("category == '" + escapeFilterValue(categoryFilter) + "'"); + } + String scopeFilter = trimToNull(kbScope); + if (scopeFilter != null) { + parts.add("kb_scope == '" + escapeFilterValue(scopeFilter) + "'"); + } + return parts.isEmpty() ? null : String.join(" && ", parts); + } + + private String trimToNull(String value) { + if (value == null || value.isBlank()) { + return null; + } + return value.trim(); + } } diff --git a/src/main/java/com/superbiz/agent/service/VectorIndexService.java b/src/main/java/com/superbiz/agent/service/VectorIndexService.java index 447dad0..6ab8355 100644 --- a/src/main/java/com/superbiz/agent/service/VectorIndexService.java +++ b/src/main/java/com/superbiz/agent/service/VectorIndexService.java @@ -11,6 +11,7 @@ import lombok.Getter; import lombok.Setter; import com.superbiz.agent.constant.MilvusConstants; import com.superbiz.agent.dto.DocumentChunk; +import com.superbiz.agent.dto.Frontmatter; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.springframework.beans.factory.annotation.Autowired; @@ -176,6 +177,10 @@ public class VectorIndexService { * @throws Exception 索引失败时抛出异常 */ public void indexDocumentChunks(String docId, List chunks, String category) throws Exception { + indexDocumentChunks(docId, chunks, category, null); + } + + public void indexDocumentChunks(String docId, List chunks, String category, Frontmatter frontmatter) throws Exception { if (chunks == null || chunks.isEmpty()) { throw new IllegalArgumentException("文档分块列表为空"); } @@ -194,7 +199,7 @@ public class VectorIndexService { List vector = embeddingService.generateEmbedding(buildEmbeddingText(chunk)); // 构建元数据(使用 docId 和 category) - Map metadata = buildDocumentMetadata(docId, chunk, chunks.size(), category); + Map metadata = buildDocumentMetadata(docId, chunk, chunks.size(), category, frontmatter); // 插入到 Milvus insertToMilvus(chunk.getContent(), vector, metadata, chunk.getChunkIndex()); @@ -253,30 +258,47 @@ public class VectorIndexService { /** * 构建文档元数据(用于上传文档) */ - private Map buildDocumentMetadata(String docId, DocumentChunk chunk, int totalChunks, String category) { + static Map buildDocumentMetadata(String docId, DocumentChunk chunk, int totalChunks, String category) { + return buildDocumentMetadata(docId, chunk, totalChunks, category, null); + } + + static Map buildDocumentMetadata(String docId, + DocumentChunk chunk, + int totalChunks, + String category, + Frontmatter frontmatter) { Map metadata = new HashMap<>(); // 文档标识 + String source = firstNonBlank(frontmatter != null ? frontmatter.getSource() : null, "upload:" + docId); metadata.put("docId", docId); - metadata.put("_source", "upload:" + docId); // 区分文件索引和上传文档 + metadata.put("_source", source); // 区分文件索引和上传文档 + metadata.put("source", source); // 分片信息 metadata.put("chunkIndex", chunk.getChunkIndex()); metadata.put("totalChunks", totalChunks); // 标题信息 - if (chunk.getTitle() != null && !chunk.getTitle().isEmpty()) { - metadata.put("title", chunk.getTitle()); + String title = firstNonBlank(chunk.getTitle(), frontmatter != null ? frontmatter.getTitle() : null); + if (title != null) { + metadata.put("title", title); } // 面包屑导航(完整标题层级路径) - if (chunk.getBreadcrumb() != null && !chunk.getBreadcrumb().isEmpty()) { - metadata.put("breadcrumb", chunk.getBreadcrumb()); + String breadcrumb = firstNonBlank(frontmatter != null ? frontmatter.getBreadcrumb() : null, chunk.getBreadcrumb()); + if (breadcrumb != null) { + metadata.put("breadcrumb", breadcrumb); } // 文档类别 metadata.put("category", category != null && !category.isBlank() ? category : "upload"); + String kbScope = trimToNull(frontmatter != null ? frontmatter.getKbScope() : null); + if (kbScope != null) { + metadata.put("kb_scope", kbScope); + } + return metadata; } @@ -304,6 +326,23 @@ public class VectorIndexService { return value == null ? "" : value.trim(); } + private static String trimToNull(String value) { + if (value == null || value.isBlank()) { + return null; + } + return value.trim(); + } + + private static String firstNonBlank(String... values) { + for (String value : values) { + String trimmed = trimToNull(value); + if (trimmed != null) { + return trimmed; + } + } + return null; + } + /** * 删除文件的旧数据(根据 metadata._source) */ diff --git a/src/main/java/com/superbiz/agent/service/VectorSearchService.java b/src/main/java/com/superbiz/agent/service/VectorSearchService.java index 4b49a49..8e4bdba 100644 --- a/src/main/java/com/superbiz/agent/service/VectorSearchService.java +++ b/src/main/java/com/superbiz/agent/service/VectorSearchService.java @@ -54,6 +54,9 @@ public class VectorSearchService { @Value("${retrieval.normalization.max-l2-distance:2.0}") private double maxL2Distance = 2.0; + @Value("${retrieval.kb-scope:}") + private String kbScope = ""; + public List searchSimilarDocuments(String query, int topK) { return searchSimilarDocuments(query, topK, null); } @@ -62,7 +65,7 @@ public class VectorSearchService { String mode = vectorStoreMode == null ? "auto" : vectorStoreMode.trim().toLowerCase(); return switch (mode) { case "sdk" -> searchSimilarDocumentsWithSdk(query, topK, category); - case "spring-ai" -> searchSimilarDocumentsWithVectorStore(query, topK, category); + case "spring", "spring-ai" -> searchSimilarDocumentsWithVectorStore(query, topK, category); case "auto" -> searchWithAutoFallback(query, topK, category); default -> { logger.warn("Unknown retrieval.vector-store.mode={}, using auto mode", vectorStoreMode); @@ -86,15 +89,16 @@ public class VectorSearchService { throw new IllegalStateException("Spring AI VectorStore bean is unavailable"); } - logger.info("Starting Spring AI VectorStore search: query={}, topK={}, category={}", query, topK, category); + logger.info("Starting Spring AI VectorStore search: query={}, topK={}, category={}, kbScope={}", + query, topK, category, effectiveKbScope()); SearchRequest.Builder builder = SearchRequest.builder() .query(query) .topK(topK) .similarityThresholdAll(); - if (category != null && !category.trim().isEmpty()) { - String filterExpression = "category == '" + escapeFilterValue(category.trim()) + "'"; + String filterExpression = buildSpringAiFilterExpression(category); + if (filterExpression != null) { builder.filterExpression(filterExpression); - logger.info("Spring AI VectorStore category filter: {}", filterExpression); + logger.info("Spring AI VectorStore metadata filter: {}", filterExpression); } List documents = vectorStore.similaritySearch(builder.build()); @@ -115,7 +119,8 @@ public class VectorSearchService { List searchSimilarDocumentsWithSdk(String query, int topK, String category) { try { - logger.info("Starting Milvus SDK search: query={}, topK={}, category={}", query, topK, category); + logger.info("Starting Milvus SDK search: query={}, topK={}, category={}, kbScope={}", + query, topK, category, effectiveKbScope()); List queryVector = embeddingService.generateQueryVector(query); logger.debug("Query vector generated, dimension={}", queryVector.size()); @@ -129,10 +134,10 @@ public class VectorSearchService { .withOutFields(List.of("id", "content", "metadata")) .withParams("{\"nprobe\":10}"); - if (category != null && !category.trim().isEmpty()) { - String expr = String.format("metadata[\"category\"] == \"%s\"", category); + String expr = buildSdkFilterExpression(category); + if (expr != null) { searchParamBuilder.withExpr(expr); - logger.info("Milvus SDK category filter: {}", expr); + logger.info("Milvus SDK metadata filter: {}", expr); } R searchResponse = milvusClient.search(searchParamBuilder.build()); @@ -215,6 +220,47 @@ public class VectorSearchService { return value.replace("'", "\\'"); } + String buildSpringAiFilterExpression(String category) { + List parts = new ArrayList<>(); + String categoryFilter = trimToNull(category); + if (categoryFilter != null) { + parts.add("category == '" + escapeFilterValue(categoryFilter) + "'"); + } + String scopeFilter = effectiveKbScope(); + if (scopeFilter != null) { + parts.add("kb_scope == '" + escapeFilterValue(scopeFilter) + "'"); + } + return parts.isEmpty() ? null : String.join(" && ", parts); + } + + String buildSdkFilterExpression(String category) { + List parts = new ArrayList<>(); + String categoryFilter = trimToNull(category); + if (categoryFilter != null) { + parts.add("metadata[\"category\"] == \"" + escapeMilvusString(categoryFilter) + "\""); + } + String scopeFilter = effectiveKbScope(); + if (scopeFilter != null) { + parts.add("metadata[\"kb_scope\"] == \"" + escapeMilvusString(scopeFilter) + "\""); + } + return parts.isEmpty() ? null : String.join(" && ", parts); + } + + private String effectiveKbScope() { + return trimToNull(kbScope); + } + + private String trimToNull(String value) { + if (value == null || value.isBlank()) { + return null; + } + return value.trim(); + } + + private String escapeMilvusString(String value) { + return value.replace("\\", "\\\\").replace("\"", "\\\""); + } + @Setter @Getter public static class SearchResult { diff --git a/src/main/resources/application.yml b/src/main/resources/application.yml index 5ae16ca..4a02b3a 100644 --- a/src/main/resources/application.yml +++ b/src/main/resources/application.yml @@ -157,6 +157,7 @@ rag: # 检索归一化配置 retrieval: + kb-scope: "" # empty means search all legacy documents; use rag-eval for eval seed docs vector-store: mode: auto # auto | spring-ai | sdk normalization: diff --git a/src/test/java/com/superbiz/agent/eval/RagEvalSeedImporterTest.java b/src/test/java/com/superbiz/agent/eval/RagEvalSeedImporterTest.java new file mode 100644 index 0000000..6f9e0e4 --- /dev/null +++ b/src/test/java/com/superbiz/agent/eval/RagEvalSeedImporterTest.java @@ -0,0 +1,93 @@ +package com.superbiz.agent.eval; + +import com.superbiz.agent.Main; +import com.superbiz.agent.domain.entity.ApiDocument; +import com.superbiz.agent.dto.DocumentUploadRequest; +import com.superbiz.agent.dto.Frontmatter; +import com.superbiz.agent.repository.ApiDocumentRepository; +import com.superbiz.agent.service.DocumentManagementService; +import com.superbiz.agent.service.FrontmatterParser; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.condition.EnabledIfSystemProperty; +import org.springframework.beans.factory.annotation.Autowired; +import org.springframework.boot.test.context.SpringBootTest; +import org.springframework.mock.web.MockMultipartFile; + +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.List; + +import static org.junit.jupiter.api.Assertions.assertFalse; + +/** + * Imports canonical RAG eval documents through the real document pipeline. + * + *

Disabled by default because it writes DB rows, local knowledge files, and + * vector index records in the configured runtime environment.

+ */ +@SpringBootTest( + classes = Main.class, + webEnvironment = SpringBootTest.WebEnvironment.NONE, + properties = "spring.main.web-application-type=none" +) +@EnabledIfSystemProperty(named = "rag.seed.enabled", matches = "true") +class RagEvalSeedImporterTest { + + private static final Path DEFAULT_SEED_DOCS = Path.of("eval/rag-retrieval/seed-docs"); + + @Autowired + private DocumentManagementService documentManagementService; + + @Autowired + private FrontmatterParser frontmatterParser; + + @Autowired + private ApiDocumentRepository apiDocumentRepository; + + @Test + void importSeedDocuments() throws Exception { + Path seedDir = Path.of(System.getProperty("rag.seed.docs", DEFAULT_SEED_DOCS.toString())); + List docs; + try (var stream = Files.list(seedDir)) { + docs = stream + .filter(path -> path.getFileName().toString().endsWith(".md")) + .sorted() + .toList(); + } + assertFalse(docs.isEmpty(), "seed docs directory must contain markdown files"); + + for (Path docPath : docs) { + String content = Files.readString(docPath, StandardCharsets.UTF_8); + Frontmatter frontmatter = frontmatterParser.parse(content); + if (frontmatter == null || frontmatter.getSource() == null || frontmatter.getSource().isBlank()) { + throw new IllegalArgumentException("seed doc must include frontmatter source: " + docPath); + } + + apiDocumentRepository.findByDocId(frontmatter.getSource().trim()) + .map(ApiDocument::getDocId) + .ifPresent(documentManagementService::deleteDocument); + + String fileName = docPath.getFileName().toString(); + MockMultipartFile file = new MockMultipartFile( + "file", + fileName, + "text/markdown", + content.getBytes(StandardCharsets.UTF_8) + ); + DocumentUploadRequest request = DocumentUploadRequest.builder() + .file(file) + .category(resolveCategory(frontmatter)) + .build(); + + documentManagementService.uploadDocument(request); + } + } + + private String resolveCategory(Frontmatter frontmatter) { + if (frontmatter.getCategory() != null && !frontmatter.getCategory().isBlank()) { + return frontmatter.getCategory().trim(); + } + return "rag-eval"; + } +} diff --git a/src/test/java/com/superbiz/agent/eval/RagLookupSnapshotGeneratorTest.java b/src/test/java/com/superbiz/agent/eval/RagLookupSnapshotGeneratorTest.java new file mode 100644 index 0000000..30a37c2 --- /dev/null +++ b/src/test/java/com/superbiz/agent/eval/RagLookupSnapshotGeneratorTest.java @@ -0,0 +1,85 @@ +package com.superbiz.agent.eval; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.fasterxml.jackson.databind.node.ObjectNode; +import com.superbiz.agent.Main; +import com.superbiz.agent.dto.LookupResult; +import com.superbiz.agent.tool.LookupKnowledgeTool; +import com.superbiz.agent.util.SessionContextHolder; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.condition.EnabledIfSystemProperty; +import org.springframework.beans.factory.annotation.Autowired; +import org.springframework.boot.test.context.SpringBootTest; + +import java.nio.file.Files; +import java.nio.file.Path; +import java.time.Instant; + +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Generates RAG retrieval fixtures from the real LookupKnowledgeTool bean. + * + *

This class is disabled by default because it writes repository files and + * depends on the configured runtime retrieval stack.

+ */ +@SpringBootTest( + classes = Main.class, + webEnvironment = SpringBootTest.WebEnvironment.NONE, + properties = "spring.main.web-application-type=none" +) +@EnabledIfSystemProperty(named = "rag.snapshot.enabled", matches = "true") +class RagLookupSnapshotGeneratorTest { + + private static final Path DEFAULT_CASES = Path.of("eval/rag-retrieval/cases/golden-cases.json"); + private static final Path DEFAULT_FIXTURES = Path.of("eval/rag-retrieval/fixtures"); + + @Autowired + private LookupKnowledgeTool lookupKnowledgeTool; + + @Autowired + private ObjectMapper objectMapper; + + @Test + void generateLookupResultFixtures() throws Exception { + Path casesPath = Path.of(System.getProperty("rag.snapshot.cases", DEFAULT_CASES.toString())); + Path fixturesDir = Path.of(System.getProperty("rag.snapshot.fixtures", DEFAULT_FIXTURES.toString())); + String retrievedAt = System.getProperty("rag.snapshot.retrievedAt", Instant.now().toString()); + + JsonNode root = objectMapper.readTree(casesPath.toFile()); + JsonNode cases = root.path("cases"); + assertTrue(cases.isArray(), "golden cases file must contain a cases array"); + + Files.createDirectories(fixturesDir); + for (JsonNode testCase : cases) { + String caseId = requiredText(testCase, "caseId"); + String query = requiredText(testCase, "query"); + + SessionContextHolder.clear(); + LookupResult lookupResult; + try { + lookupResult = lookupKnowledgeTool.lookupKnowledge(query); + } finally { + SessionContextHolder.clear(); + } + + ObjectNode fixture = objectMapper.createObjectNode(); + fixture.put("caseId", caseId); + fixture.put("query", query); + fixture.put("retrievedAt", retrievedAt); + fixture.set("lookupResult", objectMapper.valueToTree(lookupResult)); + + Path output = fixturesDir.resolve(caseId + ".json"); + objectMapper.writerWithDefaultPrettyPrinter().writeValue(output.toFile(), fixture); + } + } + + private String requiredText(JsonNode node, String fieldName) { + JsonNode value = node.get(fieldName); + if (value == null || value.asText().isBlank()) { + throw new IllegalArgumentException("golden case is missing required field: " + fieldName); + } + return value.asText(); + } +} diff --git a/src/test/java/com/superbiz/agent/service/DocumentManagementServiceTest.java b/src/test/java/com/superbiz/agent/service/DocumentManagementServiceTest.java index 584332d..f3c44b6 100644 --- a/src/test/java/com/superbiz/agent/service/DocumentManagementServiceTest.java +++ b/src/test/java/com/superbiz/agent/service/DocumentManagementServiceTest.java @@ -1,5 +1,6 @@ package com.superbiz.agent.service; +import com.superbiz.agent.dto.Frontmatter; import org.junit.jupiter.api.Test; import org.junit.jupiter.api.io.TempDir; import org.springframework.mock.web.MockMultipartFile; @@ -38,4 +39,16 @@ class DocumentManagementServiceTest { assertEquals("payment/runbook.md", storedPath); assertTrue(Files.exists(tempDir.resolve("payment").resolve("runbook.md"))); } + + @Test + void resolveDocumentIdUsesFrontmatterSourceWhenItFitsDatabaseColumn() { + DocumentManagementService service = new DocumentManagementService(); + Frontmatter frontmatter = Frontmatter.builder() + .source("mysql-connection-pool") + .build(); + + String docId = ReflectionTestUtils.invokeMethod(service, "resolveDocumentId", frontmatter); + + assertEquals("mysql-connection-pool", docId); + } } diff --git a/src/test/java/com/superbiz/agent/service/FrontmatterParserTest.java b/src/test/java/com/superbiz/agent/service/FrontmatterParserTest.java index b4fb156..fac2fe7 100644 --- a/src/test/java/com/superbiz/agent/service/FrontmatterParserTest.java +++ b/src/test/java/com/superbiz/agent/service/FrontmatterParserTest.java @@ -146,4 +146,48 @@ class FrontmatterParserTest { assertEquals("1.0.0", result.getVersion()); assertEquals("Test Author", result.getAuthor()); } + + @Test + void testParse_withRetrievalMetadata() { + String content = """ + --- + title: MySQL Connection Pool + keywords: [connection pool, HikariCP] + summary: Diagnose exhausted MySQL connection pools + category: database + source: mysql-connection-pool + breadcrumb: Database > MySQL > Connection Pool + kb_scope: rag-eval + --- + Content + """; + + Frontmatter result = parser.parse(content); + + assertNotNull(result); + assertEquals("mysql-connection-pool", result.getSource()); + assertEquals("Database > MySQL > Connection Pool", result.getBreadcrumb()); + assertEquals("rag-eval", result.getKbScope()); + } + + @Test + void testStripFrontmatter_returnsMarkdownBodyOnly() { + String content = """ + --- + title: Test + keywords: [frontmatter-only] + summary: Summary + --- + + # Body + + Body content + """; + + String body = parser.stripFrontmatter(content); + + assertFalse(body.contains("frontmatter-only")); + assertTrue(body.startsWith("# Body")); + assertTrue(body.contains("Body content")); + } } diff --git a/src/test/java/com/superbiz/agent/service/KnowledgeIndexServiceTest.java b/src/test/java/com/superbiz/agent/service/KnowledgeIndexServiceTest.java index 71c90fe..1943926 100644 --- a/src/test/java/com/superbiz/agent/service/KnowledgeIndexServiceTest.java +++ b/src/test/java/com/superbiz/agent/service/KnowledgeIndexServiceTest.java @@ -138,6 +138,49 @@ class KnowledgeIndexServiceTest { assertNull(hint.singleDomainOrNull()); } + @Test + void testAnalyzeQuery_filtersByConfiguredKbScope() { + ReflectionTestUtils.setField(service, "kbScope", "rag-eval"); + service.addToIndex(KnowledgeEntry.builder() + .filePath("legacy.md") + .keywords(List.of("timeout")) + .category("legacy") + .build()); + service.addToIndex(KnowledgeEntry.builder() + .filePath("eval.md") + .keywords(List.of("timeout")) + .category("eval") + .kbScope("rag-eval") + .build()); + + KnowledgeIndexService.L0Hint hint = service.analyzeQuery("timeout"); + + assertEquals(1, hint.matches().size()); + assertEquals(List.of("eval"), hint.domains()); + assertEquals("eval", hint.singleDomainOrNull()); + } + + @Test + void testAnalyzeQuery_keepsLegacyEntriesWhenNoScopeConfigured() { + ReflectionTestUtils.setField(service, "kbScope", ""); + service.addToIndex(KnowledgeEntry.builder() + .filePath("legacy.md") + .keywords(List.of("timeout")) + .category("legacy") + .build()); + service.addToIndex(KnowledgeEntry.builder() + .filePath("eval.md") + .keywords(List.of("timeout")) + .category("eval") + .kbScope("rag-eval") + .build()); + + KnowledgeIndexService.L0Hint hint = service.analyzeQuery("timeout"); + + assertEquals(2, hint.matches().size()); + assertNull(hint.singleDomainOrNull()); + } + @Test void testExactMatch_noMatch() { KnowledgeEntry entry = KnowledgeEntry.builder() diff --git a/src/test/java/com/superbiz/agent/service/VectorIndexServiceTest.java b/src/test/java/com/superbiz/agent/service/VectorIndexServiceTest.java index 2957774..06627da 100644 --- a/src/test/java/com/superbiz/agent/service/VectorIndexServiceTest.java +++ b/src/test/java/com/superbiz/agent/service/VectorIndexServiceTest.java @@ -1,8 +1,11 @@ package com.superbiz.agent.service; import com.superbiz.agent.dto.DocumentChunk; +import com.superbiz.agent.dto.Frontmatter; import org.junit.jupiter.api.Test; +import java.util.Map; + import static org.junit.jupiter.api.Assertions.assertEquals; class VectorIndexServiceTest { @@ -34,4 +37,35 @@ class VectorIndexServiceTest { assertEquals("Plain chunk content.", VectorIndexService.buildEmbeddingText(chunk)); } + + @Test + void buildDocumentMetadataUsesFrontmatterRetrievalFields() { + DocumentChunk chunk = DocumentChunk.builder() + .chunkIndex(0) + .title("Chunk Title") + .breadcrumb("Chunk > Path") + .content("content") + .build(); + Frontmatter frontmatter = Frontmatter.builder() + .title("Document Title") + .source("mysql-connection-pool") + .breadcrumb("Database > MySQL > Connection Pool") + .kbScope("rag-eval") + .build(); + + Map metadata = VectorIndexService.buildDocumentMetadata( + "mysql-connection-pool", + chunk, + 2, + "database", + frontmatter + ); + + assertEquals("mysql-connection-pool", metadata.get("docId")); + assertEquals("mysql-connection-pool", metadata.get("_source")); + assertEquals("mysql-connection-pool", metadata.get("source")); + assertEquals("database", metadata.get("category")); + assertEquals("rag-eval", metadata.get("kb_scope")); + assertEquals("Database > MySQL > Connection Pool", metadata.get("breadcrumb")); + } } diff --git a/src/test/java/com/superbiz/agent/service/VectorSearchServiceTest.java b/src/test/java/com/superbiz/agent/service/VectorSearchServiceTest.java index 0fb7cc6..5a5cc5f 100644 --- a/src/test/java/com/superbiz/agent/service/VectorSearchServiceTest.java +++ b/src/test/java/com/superbiz/agent/service/VectorSearchServiceTest.java @@ -13,6 +13,7 @@ import java.util.List; import java.util.Map; import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNull; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.mockito.ArgumentMatchers.any; import static org.mockito.ArgumentMatchers.eq; @@ -151,6 +152,45 @@ class VectorSearchServiceTest { assertTrue(request.toString().contains("api")); } + @Test + void springModeUsesVectorStoreAlias() { + VectorStore vectorStore = mock(VectorStore.class); + ObjectProvider provider = mock(ObjectProvider.class); + when(provider.getIfAvailable()).thenReturn(vectorStore); + when(vectorStore.similaritySearch(any(SearchRequest.class))).thenReturn(List.of()); + + VectorSearchService service = new VectorSearchService(); + setMode(service, "spring"); + setVectorStore(service, provider); + + service.searchSimilarDocuments("query", 5, "api"); + + verify(vectorStore).similaritySearch(any(SearchRequest.class)); + } + + @Test + void defaultScopeDoesNotAddMetadataFilter() { + VectorSearchService service = new VectorSearchService(); + setKbScope(service, ""); + + assertNull(service.buildSpringAiFilterExpression(null)); + assertNull(service.buildSdkFilterExpression(null)); + assertEquals("category == 'api'", service.buildSpringAiFilterExpression("api")); + assertEquals("metadata[\"category\"] == \"api\"", service.buildSdkFilterExpression("api")); + } + + @Test + void configuredScopeCombinesWithCategoryFilter() { + VectorSearchService service = new VectorSearchService(); + setKbScope(service, "rag-eval"); + + assertEquals("kb_scope == 'rag-eval'", service.buildSpringAiFilterExpression(null)); + assertEquals("category == 'api' && kb_scope == 'rag-eval'", + service.buildSpringAiFilterExpression("api")); + assertEquals("metadata[\"category\"] == \"api\" && metadata[\"kb_scope\"] == \"rag-eval\"", + service.buildSdkFilterExpression("api")); + } + private static void setMode(VectorSearchService service, String mode) { ReflectionTestUtils.setField(service, "vectorStoreMode", mode); } @@ -161,6 +201,10 @@ class VectorSearchServiceTest { ReflectionTestUtils.setField(service, "maxL2Distance", 2.0); } + private static void setKbScope(VectorSearchService service, String kbScope) { + ReflectionTestUtils.setField(service, "kbScope", kbScope); + } + private static VectorSearchService.SearchResult result(String id, float score) { VectorSearchService.SearchResult result = new VectorSearchService.SearchResult(); result.setId(id);