Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ed267d753d | ||
|
|
2658742119 | ||
|
|
72a3dbf8c5 | ||
|
|
674dd27a48 | ||
|
|
c7e2fc2ee2 | ||
|
|
1bfe1a17b4 | ||
|
|
9dd6823fe7 | ||
|
|
9376448804 | ||
|
|
f2bae0382c | ||
|
|
5c71f5fc79 | ||
|
|
b9ec07de57 | ||
|
|
5197712719 | ||
|
|
4a94c14feb | ||
|
|
9a2a44d1b5 |
@@ -0,0 +1,86 @@
|
||||
# RAG Retrieval Baseline
|
||||
|
||||
This directory contains the offline retrieval baseline for the RAG refactor.
|
||||
|
||||
The baseline is intentionally narrower than full diagnosis evaluation. It checks
|
||||
whether fixed retrieval queries can recover expected documents, breadcrumbs, and
|
||||
evidence keywords before changing L0 behavior, query augmentation, evidence
|
||||
post-processing, or Spring AI VectorStore integration.
|
||||
|
||||
## Layout
|
||||
|
||||
```text
|
||||
eval/rag-retrieval/
|
||||
cases/golden-cases.json Fixed retrieval golden cases
|
||||
fixtures/*.json Saved retrieval candidates for each case
|
||||
reports/baseline.json Machine-readable baseline report
|
||||
reports/baseline.md Human-readable baseline report
|
||||
reports/live-post-reindex.* Optional live acceptance reports
|
||||
```
|
||||
|
||||
## Run
|
||||
|
||||
From the repository root:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_retrieval.py
|
||||
```
|
||||
|
||||
Custom paths are also supported:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_retrieval.py \
|
||||
--cases eval/rag-retrieval/cases/golden-cases.json \
|
||||
--fixtures eval/rag-retrieval/fixtures \
|
||||
--json-report eval/rag-retrieval/reports/baseline.json \
|
||||
--markdown-report eval/rag-retrieval/reports/baseline.md
|
||||
```
|
||||
|
||||
## Hit Levels
|
||||
|
||||
- `strong`: expected document is found and breadcrumb or evidence keyword coverage is satisfied.
|
||||
- `medium`: expected document is found, but breadcrumb or keyword coverage is incomplete.
|
||||
- `weak`: expected evidence keyword is found, but expected document is missing.
|
||||
- `miss`: expected document and expected evidence are not found.
|
||||
|
||||
`Recall@K` counts `strong` and `medium` as retrieved.
|
||||
|
||||
## Scope
|
||||
|
||||
This baseline runs fully offline and does not call MySQL, Redis, Milvus, an LLM,
|
||||
or the Spring Boot application. It is a regression harness for retrieval behavior,
|
||||
not a claim that live production retrieval accuracy is complete.
|
||||
|
||||
## Live Post-Reindex Acceptance
|
||||
|
||||
When embedding input changes, existing vectors do not update by themselves. For
|
||||
example, after adding `title` and `breadcrumb` to the embedding text, the live
|
||||
Milvus/Zilliz collection must be reindexed before retrieval can reflect that new
|
||||
semantic signal.
|
||||
|
||||
Use this optional live acceptance flow after the application is running and the
|
||||
knowledge base has been reindexed:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_live_acceptance.py
|
||||
```
|
||||
|
||||
Custom service URL and output paths are supported:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_live_acceptance.py \
|
||||
--base-url http://127.0.0.1:9900 \
|
||||
--json-report eval/rag-retrieval/reports/live-post-reindex.json \
|
||||
--markdown-report eval/rag-retrieval/reports/live-post-reindex.md
|
||||
```
|
||||
|
||||
The script calls:
|
||||
|
||||
```text
|
||||
GET /api/search/similar
|
||||
```
|
||||
|
||||
It writes JSON and Markdown reports with query, topK, result count, top
|
||||
candidates, breadcrumb, score labels, and raw response fields. This is a live
|
||||
smoke check for environment readiness and post-reindex behavior; it does not
|
||||
replace the deterministic offline baseline above.
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"version": 1,
|
||||
"description": "Offline golden retrieval cases for RAG refactor baseline.",
|
||||
"topK": 5,
|
||||
"cases": [
|
||||
{
|
||||
"caseId": "chat-mysql-connection-pool",
|
||||
"scenario": "chat",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"expectedDocIds": ["mysql-connection-pool"],
|
||||
"expectedBreadcrumbs": ["Database > MySQL > Connection Pool"],
|
||||
"expectedKeywords": ["connection pool", "max_connections", "HikariCP"],
|
||||
"notes": "Covers precise database troubleshooting retrieval."
|
||||
},
|
||||
{
|
||||
"caseId": "chat-diagnosis-flow",
|
||||
"scenario": "chat",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"expectedDocIds": ["incident-diagnosis-flow"],
|
||||
"expectedBreadcrumbs": ["AIOps > Diagnosis Flow"],
|
||||
"expectedKeywords": ["collect evidence", "verify", "remediation"],
|
||||
"notes": "Covers process-style knowledge where breadcrumb matters."
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-payment-latency-alert",
|
||||
"scenario": "aiops",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"expectedDocIds": ["payment-service-latency"],
|
||||
"expectedBreadcrumbs": ["AIOps > Service Alerts > Payment Latency"],
|
||||
"expectedKeywords": ["p95 latency", "payment-service", "downstream dependency"],
|
||||
"notes": "Covers alert payload terms that should become retrieval hints."
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-prometheus-alert-scope",
|
||||
"scenario": "aiops",
|
||||
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
||||
"expectedDocIds": ["aiops-alert-scope-control"],
|
||||
"expectedBreadcrumbs": ["AIOps > Alert Scope Control"],
|
||||
"expectedKeywords": ["payload", "unrelated active alerts", "scope"],
|
||||
"notes": "Covers scoped alert diagnosis behavior."
|
||||
},
|
||||
{
|
||||
"caseId": "chat-rag-chunk-context",
|
||||
"scenario": "chat",
|
||||
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"expectedDocIds": ["rag-chunk-context-reconstruction"],
|
||||
"expectedBreadcrumbs": ["RAG > Chunking > Context Reconstruction"],
|
||||
"expectedKeywords": ["neighbor chunk", "same section", "breadcrumb"],
|
||||
"notes": "Covers the known RAG refactor issue around context reconstruction."
|
||||
},
|
||||
{
|
||||
"caseId": "chat-l0-domain-hint",
|
||||
"scenario": "chat",
|
||||
"query": "Should L0 keyword matching decide the final retrieval result?",
|
||||
"expectedDocIds": ["rag-l0-domain-entity-hint"],
|
||||
"expectedBreadcrumbs": ["RAG > L0 > Domain Entity Hint"],
|
||||
"expectedKeywords": ["domain detector", "entity extractor", "metadata filter"],
|
||||
"notes": "Covers the target L0 role after refactor."
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "aiops-payment-latency-alert",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "payment-service-latency",
|
||||
"title": "Payment Service Latency Alert Playbook",
|
||||
"breadcrumb": "AIOps > Service Alerts > Payment Latency",
|
||||
"content": "For payment-service p95 latency alerts, check downstream dependency latency, thread pool saturation, gateway retries, and recent deployment changes.",
|
||||
"score": 0.84,
|
||||
"retrievalLayer": "L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "mysql-connection-pool",
|
||||
"title": "MySQL Connection Pool Troubleshooting",
|
||||
"breadcrumb": "Database > MySQL > Connection Pool",
|
||||
"content": "Database connection pool saturation can increase payment latency when checkout paths wait for connections.",
|
||||
"score": 0.68,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
{
|
||||
"caseId": "aiops-prometheus-alert-scope",
|
||||
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "aiops-alert-scope-control",
|
||||
"title": "AIOps Alert Scope Control",
|
||||
"breadcrumb": "AIOps > Alert Scope Control",
|
||||
"content": "When payload mode is active, queryPrometheusAlerts can verify the supplied alert, but unrelated active alerts must remain scoped context and should not become full diagnoses.",
|
||||
"score": 0.9,
|
||||
"retrievalLayer": "L0+L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-diagnosis-flow",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "incident-diagnosis-flow",
|
||||
"title": "Incident Diagnosis Flow",
|
||||
"breadcrumb": "AIOps > Diagnosis Flow",
|
||||
"content": "The standard flow is to collect evidence, identify the suspected fault domain, verify the hypothesis, apply remediation, and confirm recovery.",
|
||||
"score": 0.82,
|
||||
"retrievalLayer": "L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "rag-chunk-context-reconstruction",
|
||||
"title": "RAG Chunk Context Reconstruction",
|
||||
"breadcrumb": "RAG > Chunking > Context Reconstruction",
|
||||
"content": "Long sections may require neighbor chunk expansion and breadcrumb-aware packing.",
|
||||
"score": 0.55,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-l0-domain-hint",
|
||||
"query": "Should L0 keyword matching decide the final retrieval result?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "rag-l0-domain-entity-hint",
|
||||
"title": "RAG L0 Domain Entity Hint",
|
||||
"breadcrumb": "RAG > L0 > Domain Entity Hint",
|
||||
"content": "L0 should be retained as a domain detector, entity extractor, metadata filter generator, and explainability signal, not as the final retrieval decision.",
|
||||
"score": 0.88,
|
||||
"retrievalLayer": "L0"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "rag-l0-l1-fusion-ranking",
|
||||
"title": "RAG L0 L1 Fusion Ranking",
|
||||
"breadcrumb": "RAG > Ranking > Fusion",
|
||||
"content": "L0 and L1 candidates should eventually be fused rather than handled as an early-return branch.",
|
||||
"score": 0.75,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-mysql-connection-pool",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "mysql-connection-pool",
|
||||
"title": "MySQL Connection Pool Troubleshooting",
|
||||
"breadcrumb": "Database > MySQL > Connection Pool",
|
||||
"content": "When the connection pool is exhausted, inspect HikariCP active connections, max_connections, slow SQL, leak detection, and database wait events.",
|
||||
"score": 0.86,
|
||||
"retrievalLayer": "L0+L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "incident-diagnosis-flow",
|
||||
"title": "Incident Diagnosis Flow",
|
||||
"breadcrumb": "AIOps > Diagnosis Flow",
|
||||
"content": "Collect evidence, compare metrics and logs, then verify remediation before closing the incident.",
|
||||
"score": 0.61,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"caseId": "chat-rag-chunk-context",
|
||||
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"retrievedAt": "2026-07-05T00:00:00Z",
|
||||
"candidates": [
|
||||
{
|
||||
"rank": 1,
|
||||
"docId": "rag-chunk-context-reconstruction",
|
||||
"title": "RAG Chunk Context Reconstruction",
|
||||
"breadcrumb": "RAG > Chunking > Context Reconstruction",
|
||||
"content": "After a chunk hit, expand to neighbor chunk candidates from the same section and preserve breadcrumb metadata in the evidence pack.",
|
||||
"score": 0.79,
|
||||
"retrievalLayer": "L1"
|
||||
},
|
||||
{
|
||||
"rank": 2,
|
||||
"docId": "rag-breadcrumb-embedding-gap",
|
||||
"title": "RAG Breadcrumb Embedding Gap",
|
||||
"breadcrumb": "RAG > Embedding > Breadcrumb",
|
||||
"content": "Embedding title and breadcrumb with content helps recover section semantics.",
|
||||
"score": 0.72,
|
||||
"retrievalLayer": "L1"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1 @@
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
{
|
||||
"generatedAt": "2026-07-04T17:59:52.172759+00:00",
|
||||
"caseFile": "eval/rag-retrieval/cases/golden-cases.json",
|
||||
"fixtureDir": "eval/rag-retrieval/fixtures",
|
||||
"aggregate": {
|
||||
"caseCount": 6,
|
||||
"topK": 5,
|
||||
"strongHitCount": 6,
|
||||
"mediumHitCount": 0,
|
||||
"weakHitCount": 0,
|
||||
"missCount": 0,
|
||||
"recallAtK": 1.0,
|
||||
"strongHitRate": 1.0,
|
||||
"averageFirstHitRank": 1.0
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"caseId": "chat-mysql-connection-pool",
|
||||
"scenario": "chat",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:mysql-connection-pool",
|
||||
"2:incident-diagnosis-flow"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"connection pool",
|
||||
"max_connections",
|
||||
"hikaricp"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "chat-diagnosis-flow",
|
||||
"scenario": "chat",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:incident-diagnosis-flow",
|
||||
"2:rag-chunk-context-reconstruction"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"collect evidence",
|
||||
"verify",
|
||||
"remediation"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-payment-latency-alert",
|
||||
"scenario": "aiops",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:payment-service-latency",
|
||||
"2:mysql-connection-pool"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"p95 latency",
|
||||
"payment-service",
|
||||
"downstream dependency"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-prometheus-alert-scope",
|
||||
"scenario": "aiops",
|
||||
"query": "When an AIOps request already includes alert payload, should the agent diagnose unrelated active alerts?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:aiops-alert-scope-control"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"payload",
|
||||
"unrelated active alerts",
|
||||
"scope"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "chat-rag-chunk-context",
|
||||
"scenario": "chat",
|
||||
"query": "If a long section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:rag-chunk-context-reconstruction",
|
||||
"2:rag-breadcrumb-embedding-gap"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"neighbor chunk",
|
||||
"same section",
|
||||
"breadcrumb"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
},
|
||||
{
|
||||
"caseId": "chat-l0-domain-hint",
|
||||
"scenario": "chat",
|
||||
"query": "Should L0 keyword matching decide the final retrieval result?",
|
||||
"hitLevel": "strong",
|
||||
"passed": true,
|
||||
"firstExpectedRank": 1,
|
||||
"topCandidates": [
|
||||
"1:rag-l0-domain-entity-hint",
|
||||
"2:rag-l0-l1-fusion-ranking"
|
||||
],
|
||||
"matchedKeywords": [
|
||||
"domain detector",
|
||||
"entity extractor",
|
||||
"metadata filter"
|
||||
],
|
||||
"breadcrumbMatched": true,
|
||||
"failedChecks": []
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
# RAG Retrieval Baseline
|
||||
|
||||
Generated at: `2026-07-04T17:59:52.172759+00:00`
|
||||
|
||||
## Aggregate
|
||||
|
||||
| Metric | Value |
|
||||
|---|---:|
|
||||
| Cases | 6 |
|
||||
| Top K | 5 |
|
||||
| Recall@K | 1.0 |
|
||||
| Strong hit rate | 1.0 |
|
||||
| Strong hits | 6 |
|
||||
| Medium hits | 0 |
|
||||
| Weak hits | 0 |
|
||||
| Misses | 0 |
|
||||
| Average first hit rank | 1.0 |
|
||||
|
||||
## Cases
|
||||
|
||||
| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |
|
||||
|---|---|---|---:|---|---|
|
||||
| chat-mysql-connection-pool | chat | strong | 1 | 1:mysql-connection-pool<br>2:incident-diagnosis-flow | |
|
||||
| chat-diagnosis-flow | chat | strong | 1 | 1:incident-diagnosis-flow<br>2:rag-chunk-context-reconstruction | |
|
||||
| aiops-payment-latency-alert | aiops | strong | 1 | 1:payment-service-latency<br>2:mysql-connection-pool | |
|
||||
| aiops-prometheus-alert-scope | aiops | strong | 1 | 1:aiops-alert-scope-control | |
|
||||
| chat-rag-chunk-context | chat | strong | 1 | 1:rag-chunk-context-reconstruction<br>2:rag-breadcrumb-embedding-gap | |
|
||||
| chat-l0-domain-hint | chat | strong | 1 | 1:rag-l0-domain-entity-hint<br>2:rag-l0-l1-fusion-ranking | |
|
||||
@@ -0,0 +1,53 @@
|
||||
# AIOps Lightweight Verifier
|
||||
|
||||
## What Changed
|
||||
|
||||
AIOps now has a deterministic post-run quality gate.
|
||||
|
||||
After the final AIOps report is persisted, the service evaluates:
|
||||
|
||||
- whether the final report exists and is not trivially short
|
||||
- whether a payload-targeted report mentions the supplied alert and service
|
||||
- whether evidence tools such as `lookup_knowledge`, `query_metrics`, or `query_logs` were persisted
|
||||
|
||||
The result is stored under:
|
||||
|
||||
```text
|
||||
diagnosis_session.self_evaluation.aiops_rule_evaluation
|
||||
```
|
||||
|
||||
The trace API returns this payload through the existing session self-evaluation field.
|
||||
|
||||
## Why Rule-Based First
|
||||
|
||||
This is not a full LLM verifier yet.
|
||||
|
||||
The first AIOps quality risks are concrete and easy to check with rules:
|
||||
|
||||
- Did the report stay focused on the payload?
|
||||
- Did the run use evidence tools?
|
||||
- Did the system produce a usable final report?
|
||||
|
||||
Rule evaluation is stable, cheap, and easy to explain. It also avoids adding another hidden model call to the AIOps flow before the current trace contract is mature.
|
||||
|
||||
## Verdicts
|
||||
|
||||
The evaluator emits:
|
||||
|
||||
```text
|
||||
PASS
|
||||
WARN
|
||||
FAIL
|
||||
```
|
||||
|
||||
`FAIL` is reserved for critical issues such as a missing or too-short report. Missing payload focus terms or missing evidence tools currently produce `WARN`, because valid reports may use slightly different wording or evidence may be unavailable in a mock/demo environment.
|
||||
|
||||
## Interview Answer
|
||||
|
||||
If asked why AIOps has a verifier now:
|
||||
|
||||
> Chat already has an LLM verifier because the user questions are open-ended. For AIOps, I started with a lighter rule-based verifier because the first quality checks are very concrete: payload focus, evidence coverage, and report completeness. The evaluation is persisted into `self_evaluation`, so the trace can show not only what the Agent did, but also whether the output passed basic quality gates.
|
||||
|
||||
If asked why not use the Chat verifier directly:
|
||||
|
||||
> AIOps verification is different from Chat verification. It needs to check alert scope, evidence tool coverage, and whether unrelated active alerts were over-expanded. Reusing the Chat verifier directly would blur those semantics. The rule-based evaluator gives us a stable first quality gate; a later AIOps LLM verifier can build on the same trace contract.
|
||||
@@ -0,0 +1,62 @@
|
||||
# AIOps Query Augmentation
|
||||
|
||||
## What Changed
|
||||
|
||||
Payload-targeted AIOps prompts now include a deterministic recommended knowledge query.
|
||||
|
||||
The query is built from the non-blank payload fields:
|
||||
|
||||
```text
|
||||
alertName service severity description timeRange userRequest
|
||||
```
|
||||
|
||||
Example:
|
||||
|
||||
```text
|
||||
HighCPUUsage payment-service P1 CPU usage is above 80% last_15m
|
||||
```
|
||||
|
||||
## Why This Matters
|
||||
|
||||
AIOps payload fields contain high-value retrieval terms:
|
||||
|
||||
- alert name
|
||||
- service name
|
||||
- severity
|
||||
- symptom description
|
||||
- time range
|
||||
- operator request
|
||||
|
||||
Before this change, the Agent still had to invent its own `lookup_knowledge` query from the full prompt. That can work, but it may omit important terms such as the service name or alert name.
|
||||
|
||||
The new prompt makes the retrieval seed explicit:
|
||||
|
||||
```text
|
||||
Recommended lookup_knowledge query: ...
|
||||
```
|
||||
|
||||
## Design Choice
|
||||
|
||||
This is prompt-level query augmentation, not hidden retrieval.
|
||||
|
||||
I intentionally did not call `lookup_knowledge` automatically before the Agent runs. The project values traceability: tool calls should appear as Agent actions, with their inputs and outputs recorded in `tool_invocation`.
|
||||
|
||||
So the design is:
|
||||
|
||||
```text
|
||||
AIOps payload
|
||||
-> deterministic recommended retrieval query
|
||||
-> Agent prompt
|
||||
-> Agent may call lookup_knowledge explicitly
|
||||
-> tool_invocation records the real retrieval action
|
||||
```
|
||||
|
||||
## Interview Answer
|
||||
|
||||
If asked how AIOps payload improves RAG retrieval:
|
||||
|
||||
> I do not replace the user query with a broad domain. I extract the high-signal alert terms from the payload, such as alertName, service, severity, symptom, and time range, and put them into a compact recommended lookup query. The Agent still calls `lookup_knowledge` explicitly, so the trace remains auditable, but the retrieval query is less dependent on model improvisation.
|
||||
|
||||
If asked why not auto-call retrieval:
|
||||
|
||||
> Auto-calling retrieval would create hidden evidence before the Agent actually decides to use a tool. For this project, explicit tool invocation is more important because the interview story is about observable Agent execution. Prompt-level augmentation gives the Agent a better query seed without changing the trace contract.
|
||||
@@ -0,0 +1,66 @@
|
||||
# RAG Breadcrumb Embedding Acceptance
|
||||
|
||||
## What Changed
|
||||
|
||||
The indexing path now builds embedding text from chunk structure plus content:
|
||||
|
||||
```text
|
||||
Title: {title}
|
||||
Path: {breadcrumb}
|
||||
Content:
|
||||
{content}
|
||||
```
|
||||
|
||||
The stored Milvus `content` field remains the original chunk content. This keeps display and evidence output clean while allowing the vector to carry section-level semantics.
|
||||
|
||||
## Why Reindex Is Required
|
||||
|
||||
Embeddings are materialized at index time. Existing vectors were generated from the previous content-only text, so they cannot benefit from `title` and `breadcrumb` until the knowledge base is reindexed.
|
||||
|
||||
This is the key acceptance point:
|
||||
|
||||
```text
|
||||
code change alone != live retrieval changed
|
||||
code change + reindex + live query report = accepted behavior
|
||||
```
|
||||
|
||||
## How To Validate
|
||||
|
||||
1. Start the Spring Boot application.
|
||||
2. Reindex the knowledge base through the existing indexing path.
|
||||
3. Run:
|
||||
|
||||
```bash
|
||||
python scripts/eval_rag_live_acceptance.py
|
||||
```
|
||||
|
||||
The script writes:
|
||||
|
||||
```text
|
||||
eval/rag-retrieval/reports/live-post-reindex.json
|
||||
eval/rag-retrieval/reports/live-post-reindex.md
|
||||
```
|
||||
|
||||
The default cases cover:
|
||||
|
||||
- RAG chunk context questions where breadcrumb matters.
|
||||
- Diagnosis flow questions where section path matters.
|
||||
- `ERR_TIMEOUT` exact error-code retrieval.
|
||||
- MySQL connection pool troubleshooting.
|
||||
- AIOps payment-service latency alert retrieval.
|
||||
|
||||
## What To Look For
|
||||
|
||||
For breadcrumb-sensitive cases, inspect whether top candidates expose expected `title` and `breadcrumb` values in the report.
|
||||
|
||||
For core troubleshooting cases, check that result counts and top candidates remain stable. The goal is not to prove a full benchmark; it is to prove that reindexing did not obviously break important demo retrieval paths.
|
||||
|
||||
## Interview Answer
|
||||
|
||||
If asked how I verified the breadcrumb embedding change:
|
||||
|
||||
> I separated deterministic regression from live acceptance. The offline fixture baseline still runs without services. But because embedding changes only affect newly indexed vectors, I added a live post-reindex acceptance script. It calls the real `/api/search/similar` endpoint against representative breadcrumb-sensitive, troubleshooting, and AIOps queries, then writes JSON and Markdown reports. This lets me prove both that the code changed and that the live vector collection was refreshed.
|
||||
|
||||
If asked why the script does not reindex automatically:
|
||||
|
||||
> Reindexing mutates the vector store and depends on environment-specific data. I kept mutation explicit and made the script validation-only. That makes failures easier to diagnose: if retrieval does not improve, I can distinguish code changes, reindex state, and runtime retrieval behavior.
|
||||
@@ -0,0 +1,209 @@
|
||||
# RAG Refactor Story
|
||||
|
||||
## The Starting Point
|
||||
|
||||
The original RAG implementation was already usable for the MVP:
|
||||
|
||||
- Documents could be uploaded, chunked, embedded, and written to Milvus/Zilliz.
|
||||
- The Agent could call `lookup_knowledge` as an explicit tool.
|
||||
- AIOps diagnosis could retrieve troubleshooting knowledge during an alert workflow.
|
||||
- Tool invocations were persisted, so the retrieval step was visible in the execution trace.
|
||||
|
||||
But the design had several engineering problems:
|
||||
|
||||
- Retrieval was too SDK-specific. The business code directly owned many Milvus search details.
|
||||
- L0 and L1 responsibilities were blurry. L0 keyword matching could look like a final retrieval decision instead of a hint.
|
||||
- Chunk-level retrieval could lose section context when one section was split into multiple chunks.
|
||||
- Metadata such as `breadcrumb` existed, but it was not fully used in retrieval, filtering, or context reconstruction.
|
||||
- Retrieval quality was mostly checked by manual API calls and logs, not by repeatable cases.
|
||||
|
||||
So the refactor goal was not "replace everything with a framework." The goal was to move generic RAG infrastructure toward Spring AI while keeping the project-specific Agent evidence chain.
|
||||
|
||||
## How I Broke The Problem Down
|
||||
|
||||
I treated this as a staged migration, because RAG touches the Agent tool layer, AIOps diagnosis, vector retrieval, evidence packing, and database traces.
|
||||
|
||||
The first step was to establish a baseline. I added retrieval evaluation cases under `eval/rag-retrieval/` so future changes could be compared against known queries instead of judged only by intuition.
|
||||
|
||||
Then I clarified the retrieval roles:
|
||||
|
||||
```text
|
||||
L0 = domain/entity hint
|
||||
L1 = semantic retrieval
|
||||
postprocess = evidence shaping and trace-friendly output
|
||||
```
|
||||
|
||||
That means L0 is still valuable, but it should not bypass semantic retrieval as the default path. It is better used to extract service names, alert names, error codes, domains, and metadata hints.
|
||||
|
||||
After that, I added evidence postprocessing. The Agent should not just receive raw chunks; it should receive structured evidence with source, title, breadcrumb, score, hit reason, and content. This makes the result easier to inspect and easier to explain in an interview.
|
||||
|
||||
Finally, I integrated Spring AI `VectorStore` as the main read path while preserving the original Milvus SDK implementation as fallback.
|
||||
|
||||
## Current Architecture
|
||||
|
||||
The current retrieval path is:
|
||||
|
||||
```text
|
||||
Agent / API
|
||||
-> lookup_knowledge or /api/search/similar
|
||||
-> L0 domain/entity hint
|
||||
-> VectorSearchService
|
||||
-> Spring AI VectorStore
|
||||
-> Milvus SDK fallback
|
||||
-> evidence postprocess
|
||||
-> tool_invocation trace
|
||||
```
|
||||
|
||||
`VectorSearchService` is still the public retrieval facade. This is deliberate: the Agent tool layer does not need to know whether the underlying retrieval engine is SDK-based or Spring AI-based.
|
||||
|
||||
The supported retrieval modes are:
|
||||
|
||||
```text
|
||||
auto -> try Spring AI VectorStore, fallback to SDK
|
||||
spring-ai -> force Spring AI VectorStore
|
||||
sdk -> force Milvus SDK
|
||||
```
|
||||
|
||||
This keeps the migration reversible and testable.
|
||||
|
||||
## Key Tradeoffs
|
||||
|
||||
### Keep The Explicit Tool
|
||||
|
||||
I did not hide retrieval inside a Spring AI Advisor.
|
||||
|
||||
For this project, `lookup_knowledge` is part of the Agent execution story. It records what query was used, which evidence was retrieved, how relevant it looked, and how it supported diagnosis. If retrieval is hidden inside an advisor, the answer may still work, but the audit trail becomes harder to show.
|
||||
|
||||
### Keep SDK Fallback
|
||||
|
||||
The SDK path is not dead code. It is a safety net during migration.
|
||||
|
||||
This proved useful during live validation. The first VectorStore run pointed at the wrong collection name, but `auto` mode fell back to SDK and still returned results. After the collection was corrected to `biz`, the Spring AI path worked as the main path.
|
||||
|
||||
### Keep L0, But Reduce Its Authority
|
||||
|
||||
L0 is worth keeping because production incidents often contain exact identifiers:
|
||||
|
||||
- error code
|
||||
- alert name
|
||||
- service name
|
||||
- metric name
|
||||
- domain tag
|
||||
|
||||
But L0 should not be the final judge of retrieval quality. Its role is now closer to domain hint, entity extraction, metadata filtering, and explainability signal.
|
||||
|
||||
### Split Score Semantics
|
||||
|
||||
The old SDK path used L2 distance. Spring AI exposes similarity. Treating those as the same number would quietly break relevance normalization.
|
||||
|
||||
So the result separates:
|
||||
|
||||
```text
|
||||
score -> compatibility score used by existing logic
|
||||
rawScore -> raw score from the retrieval implementation
|
||||
scoreLabel -> semantic label for rawScore
|
||||
```
|
||||
|
||||
For SDK:
|
||||
|
||||
```text
|
||||
score = L2 distance
|
||||
rawScore = L2 distance
|
||||
scoreLabel = l2_distance
|
||||
```
|
||||
|
||||
For VectorStore:
|
||||
|
||||
```text
|
||||
score = Milvus metadata.distance when available
|
||||
rawScore = Spring AI similarity
|
||||
scoreLabel = similarity
|
||||
```
|
||||
|
||||
This makes the migration inspectable instead of hiding score changes behind one overloaded field.
|
||||
|
||||
### Do Not Migrate Writes Yet
|
||||
|
||||
Writes and indexing still use the SDK path.
|
||||
|
||||
That is intentional. Migrating reads and writes at the same time would make debugging harder. The read path can be validated first; write-path migration can happen later if Spring AI `VectorStore.add(...)` fits the existing metadata and chunk model.
|
||||
|
||||
## Validation Story
|
||||
|
||||
I validated the refactor at multiple levels.
|
||||
|
||||
Unit tests cover:
|
||||
|
||||
- SDK mode.
|
||||
- Spring AI mode.
|
||||
- `auto` fallback.
|
||||
- category filter behavior.
|
||||
- distance metadata mapping.
|
||||
|
||||
Live API verification used:
|
||||
|
||||
```text
|
||||
GET /api/search/similar?query=ERR_TIMEOUT&topK=3
|
||||
```
|
||||
|
||||
Logs confirmed when the Spring AI VectorStore path was used and when fallback happened.
|
||||
|
||||
Then I compared SDK and VectorStore retrieval quality on representative queries:
|
||||
|
||||
| Query Type | Result |
|
||||
| --- | --- |
|
||||
| exact error code | same top3 |
|
||||
| payment-service timeout | same top3 |
|
||||
| MySQL connection pool | same top3 |
|
||||
| AIOps alert-style query | same top3 |
|
||||
| abstract RAG design query | same top1, VectorStore returned fewer tail results |
|
||||
| category filter | both returned zero because metadata taxonomy did not match |
|
||||
|
||||
The acceptance decision was that Spring AI VectorStore is good enough for the current MVP read path, with SDK fallback preserved.
|
||||
|
||||
## Known Gaps
|
||||
|
||||
The refactor improved the architecture, but it did not solve every retrieval-quality problem.
|
||||
|
||||
Known gaps:
|
||||
|
||||
- Metadata taxonomy still needs cleanup, for example `database` vs `infrastructure`.
|
||||
- Abstract design questions may need query rewriting or better indexed interview/devflow documents.
|
||||
- Chunk context reconstruction is still limited when one logical section spans multiple chunks.
|
||||
- `breadcrumb` now participates in embedding text, but it can still be used more strongly in context expansion, rerank, and evidence packing.
|
||||
- Rerank, RRF, BM25, and hybrid retrieval are not implemented yet.
|
||||
- Indexing writes still use SDK.
|
||||
|
||||
These are good follow-up issues because they are retrieval-quality improvements, not blockers for the VectorStore migration.
|
||||
|
||||
## How I Present This In An Interview
|
||||
|
||||
My short version would be:
|
||||
|
||||
> This RAG system started as a self-built MVP around Milvus SDK retrieval. It worked, but too much infrastructure logic lived in business code, and L0/L1 responsibilities were unclear. I refactored it in stages: first I added baseline retrieval cases, then made L0 a domain/entity hint instead of a final decision layer, then added evidence postprocessing, and finally moved the main read path to Spring AI VectorStore with SDK fallback. I kept `lookup_knowledge` as an explicit Agent tool because the project values traceability: the interviewer can see when retrieval happened, what evidence was found, and how it supported the diagnosis. The result is closer to standard Spring AI RAG while still preserving business-specific observability.
|
||||
|
||||
If asked why this is not a full framework migration:
|
||||
|
||||
> I intentionally did not migrate everything at once. Reads moved first because they are easier to compare using golden queries. Writes/indexing stayed on SDK to avoid mixing schema and retrieval behavior changes in one step. Advisors were not used as the main interface because hidden retrieval would weaken the Agent trace.
|
||||
|
||||
If asked what I would improve next:
|
||||
|
||||
> I would add query transformation for AIOps payloads, improve metadata taxonomy, use breadcrumb and section metadata for context expansion, and then evaluate whether hybrid retrieval or rerank is necessary based on measured recall and topK overlap.
|
||||
|
||||
## Interview Follow-Up Questions
|
||||
|
||||
### Why introduce Spring AI VectorStore if the SDK path already worked?
|
||||
|
||||
Because SDK-only retrieval made the project own too much low-level RAG infrastructure. `VectorStore` gives a standard abstraction for retrieval and makes future Spring AI features easier to adopt, while the facade keeps the Agent layer stable.
|
||||
|
||||
### Why keep custom code at all?
|
||||
|
||||
The custom code is where the Agent engineering value lives: AIOps payload mapping, L0 hints, evidence packing, score compatibility, and tool invocation tracing. Those are domain-specific and should remain visible.
|
||||
|
||||
### How do you know quality did not regress?
|
||||
|
||||
I compared SDK and VectorStore modes on representative live queries. Core troubleshooting and AIOps cases returned the same top3 documents in the same order. The differences were isolated to abstract design queries and metadata taxonomy, which are documented follow-up work.
|
||||
|
||||
### What is the most important design decision?
|
||||
|
||||
Keeping a stable boundary: `lookup_knowledge` calls `VectorSearchService`, and `VectorSearchService` decides whether to use Spring AI or SDK. That boundary made the migration small enough to validate and explain.
|
||||
@@ -0,0 +1,211 @@
|
||||
# RAG Retrieval Quality Report
|
||||
|
||||
## Purpose
|
||||
|
||||
This report compares the live retrieval behavior of the original Milvus SDK path and the new Spring AI VectorStore path.
|
||||
|
||||
The goal is to answer an interview-critical question:
|
||||
|
||||
> After moving retrieval to Spring AI VectorStore, how do we know retrieval quality did not regress?
|
||||
|
||||
This is not a full benchmark yet. It is a focused live smoke comparison using representative RAG queries against the current Milvus/Zilliz collection.
|
||||
|
||||
## Setup
|
||||
|
||||
Service endpoint:
|
||||
|
||||
```text
|
||||
GET http://127.0.0.1:9900/api/search/similar
|
||||
```
|
||||
|
||||
Collection:
|
||||
|
||||
```text
|
||||
biz
|
||||
```
|
||||
|
||||
Compared modes:
|
||||
|
||||
```text
|
||||
retrieval.vector-store.mode=sdk
|
||||
retrieval.vector-store.mode=spring-ai
|
||||
```
|
||||
|
||||
Each case used:
|
||||
|
||||
```text
|
||||
topK=3
|
||||
```
|
||||
|
||||
The application was restarted once per mode using command-line configuration so no repository config file had to be changed.
|
||||
|
||||
## Cases
|
||||
|
||||
| Case | Query | Purpose |
|
||||
| --- | --- | --- |
|
||||
| `err-timeout` | `ERR_TIMEOUT` | Exact error-code retrieval |
|
||||
| `payment-service-timeout` | `payment-service timeout` | Service timeout troubleshooting |
|
||||
| `mysql-connection-pool` | `MySQL connection pool is exhausted. How should I diagnose it?` | Database troubleshooting |
|
||||
| `high-cpu-payment` | `HighCPUUsage payment-service` | AIOps alert-style retrieval |
|
||||
| `rag-l0-l1` | `Should L0 keyword matching decide the final retrieval result?` | Abstract RAG design query |
|
||||
| `database-filter` | `mysql timeout`, category=`database` | Metadata filter behavior |
|
||||
|
||||
## Summary
|
||||
|
||||
| Case | SDK Count | VectorStore Count | Top1 Same | TopK Overlap | Notes |
|
||||
| --- | ---: | ---: | --- | ---: | --- |
|
||||
| `err-timeout` | 3 | 3 | Yes | 3/3 | Same ordering and same documents |
|
||||
| `payment-service-timeout` | 3 | 3 | Yes | 3/3 | Same ordering and same documents |
|
||||
| `mysql-connection-pool` | 3 | 3 | Yes | 3/3 | Same ordering and same documents |
|
||||
| `high-cpu-payment` | 3 | 3 | Yes | 3/3 | Same ordering and same documents |
|
||||
| `rag-l0-l1` | 3 | 1 | Yes | 1/3 | VectorStore returned only the strongest candidate |
|
||||
| `database-filter` | 0 | 0 | N/A | N/A | Both paths applied the filter consistently; no live docs matched `category=database` |
|
||||
|
||||
## Representative Results
|
||||
|
||||
### `ERR_TIMEOUT`
|
||||
|
||||
SDK:
|
||||
|
||||
```text
|
||||
1. ERR_TIMEOUT score=0.5659486 label=l2_distance
|
||||
2. ERR_GATEWAY_TIMEOUT score=0.6048740 label=l2_distance
|
||||
3. Error handling score=0.7735061 label=l2_distance
|
||||
```
|
||||
|
||||
VectorStore:
|
||||
|
||||
```text
|
||||
1. ERR_TIMEOUT score=0.5659486 rawScore=0.4340513 label=similarity
|
||||
2. ERR_GATEWAY_TIMEOUT score=0.6048740 rawScore=0.3951259 label=similarity
|
||||
3. Error handling score=0.7735061 rawScore=0.2264938 label=similarity
|
||||
```
|
||||
|
||||
Interpretation:
|
||||
|
||||
- Document ordering is identical.
|
||||
- Compatibility `score` is identical to SDK L2 distance.
|
||||
- VectorStore `rawScore` exposes Spring AI similarity separately.
|
||||
|
||||
### `MySQL connection pool`
|
||||
|
||||
Both paths returned:
|
||||
|
||||
```text
|
||||
1. MySQL connection pool config
|
||||
2. wait_timeout timeout
|
||||
3. idle-timeout
|
||||
```
|
||||
|
||||
Interpretation:
|
||||
|
||||
- The migration preserves a precise infrastructure troubleshooting retrieval case.
|
||||
- Metadata fields such as title, category, and source remain available.
|
||||
|
||||
### `HighCPUUsage payment-service`
|
||||
|
||||
Both paths returned:
|
||||
|
||||
```text
|
||||
1. 3. HighCPUUsage / payment-service troubleshooting steps
|
||||
2. evidence mapping table row for HighCPUUsage/payment-service
|
||||
3. 3.1 Symptom confirmation
|
||||
```
|
||||
|
||||
Interpretation:
|
||||
|
||||
- AIOps-style alert terms still retrieve the expected troubleshooting document.
|
||||
- This is important because AIOps diagnosis depends on knowledge retrieval plus metrics/log evidence.
|
||||
|
||||
### `rag-l0-l1`
|
||||
|
||||
SDK returned three results, while VectorStore returned one:
|
||||
|
||||
```text
|
||||
Top1: Return error information
|
||||
```
|
||||
|
||||
Interpretation:
|
||||
|
||||
- Top1 did not regress.
|
||||
- VectorStore appears stricter for low-similarity tail results because the Spring AI path uses `similarityThresholdAll()`.
|
||||
- This is acceptable for current read-path migration, but it is worth tracking because abstract design questions may need query rewriting, better indexed docs, or adjusted threshold behavior.
|
||||
|
||||
### `database-filter`
|
||||
|
||||
Both paths returned zero results for:
|
||||
|
||||
```text
|
||||
query=mysql timeout
|
||||
category=database
|
||||
```
|
||||
|
||||
Interpretation:
|
||||
|
||||
- The filter path is consistent.
|
||||
- The live indexed MySQL docs are categorized as `infrastructure`, not `database`.
|
||||
- This highlights a metadata taxonomy issue rather than a VectorStore migration regression.
|
||||
|
||||
## Score Compatibility
|
||||
|
||||
The comparison validates the score design:
|
||||
|
||||
```text
|
||||
SDK:
|
||||
score = L2 distance
|
||||
rawScore = L2 distance
|
||||
scoreLabel = l2_distance
|
||||
|
||||
VectorStore:
|
||||
score = Milvus metadata.distance
|
||||
rawScore = Spring AI similarity
|
||||
scoreLabel = similarity
|
||||
```
|
||||
|
||||
This keeps `lookup_knowledge` relevance normalization stable while still exposing the VectorStore score semantics for trace/debugging.
|
||||
|
||||
## Findings
|
||||
|
||||
### Finding 1: Main live cases are equivalent
|
||||
|
||||
For exact error code, service timeout, MySQL troubleshooting, and AIOps alert-style retrieval, SDK and VectorStore returned identical top3 documents in identical order.
|
||||
|
||||
This is strong evidence that the read-path migration did not regress the most important demo and troubleshooting cases.
|
||||
|
||||
### Finding 2: Abstract RAG design queries need better retrieval support
|
||||
|
||||
The `rag-l0-l1` query only returned one VectorStore candidate. The top result matched SDK top1, but the tail differed.
|
||||
|
||||
This suggests the next quality work should focus on:
|
||||
|
||||
- Query transformation for abstract design questions.
|
||||
- Better indexing of interview/devflow RAG design docs.
|
||||
- Context expansion around same-section chunks.
|
||||
- Possibly tuning VectorStore threshold behavior.
|
||||
|
||||
### Finding 3: Metadata taxonomy matters
|
||||
|
||||
The category filter case returned zero results in both modes because the relevant MySQL docs are categorized as `infrastructure`, not `database`.
|
||||
|
||||
This supports a previous RAG issue: category/domain metadata should be normalized before it is used as a hard filter.
|
||||
|
||||
## Acceptance Decision
|
||||
|
||||
The Spring AI VectorStore read path is accepted for current MVP/interview use:
|
||||
|
||||
- Core troubleshooting cases match SDK behavior.
|
||||
- Score compatibility is preserved.
|
||||
- The VectorStore path exposes better score semantics without changing the `lookup_knowledge` API.
|
||||
- SDK fallback remains available for runtime safety.
|
||||
|
||||
The next retrieval-quality improvements should not block this migration. They should be handled as separate RAG quality work.
|
||||
|
||||
## Next Work
|
||||
|
||||
Recommended next steps:
|
||||
|
||||
- Add a small automated live comparison script if repeated validation becomes common.
|
||||
- Add topK overlap and top1 hit metrics to the offline evaluator.
|
||||
- Normalize metadata categories such as `database` vs `infrastructure`.
|
||||
- Add query rewriting for abstract RAG questions.
|
||||
- Decide later whether to migrate indexing writes to Spring AI `VectorStore.add(...)`.
|
||||
@@ -0,0 +1,167 @@
|
||||
# RAG VectorStore Interview Notes
|
||||
|
||||
## 60-Second Explanation
|
||||
|
||||
I refactored the RAG retrieval path from a direct Milvus SDK-only implementation to a Spring AI `VectorStore` main path, while keeping the SDK path as a fallback.
|
||||
|
||||
The important part is not just the dependency change. I kept `VectorSearchService` as the boundary, so `lookup_knowledge` and the Agent workflow did not need to change. The system now supports three modes:
|
||||
|
||||
```text
|
||||
auto -> try Spring AI VectorStore, fallback to SDK
|
||||
spring-ai -> force VectorStore
|
||||
sdk -> force SDK
|
||||
```
|
||||
|
||||
During live verification, the first run found a real config mismatch: VectorStore was pointed at `business_knowledge`, but the real Zilliz collection was `biz`. The fallback worked, so the system still returned results through SDK. After aligning the collection name, the same query went through Spring AI VectorStore successfully.
|
||||
|
||||
I also fixed score compatibility. Spring AI Milvus exposes similarity as the document score, but the old `lookup_knowledge` logic expects L2 distance. So I preserve `rawScore` and `scoreLabel`, and use Milvus `metadata.distance` as the compatibility `score` when available.
|
||||
|
||||
## Architecture Answer
|
||||
|
||||
```text
|
||||
Agent / API
|
||||
-> lookup_knowledge or /api/search/similar
|
||||
-> VectorSearchService
|
||||
-> Spring AI VectorStore
|
||||
-> Milvus SDK fallback
|
||||
-> Milvus/Zilliz collection: biz
|
||||
```
|
||||
|
||||
The key design choice is that `VectorSearchService` remains the retrieval facade. This avoids spreading framework-specific code into the Agent tool layer.
|
||||
|
||||
## Why Keep The SDK Path?
|
||||
|
||||
I kept SDK fallback for three reasons:
|
||||
|
||||
- Migration safety: the existing SDK path was already proven against the live collection.
|
||||
- Runtime resilience: if VectorStore schema mapping or filtering fails, retrieval still works.
|
||||
- Interview/demo stability: a retrieval abstraction change should not break the main Agent diagnosis demo.
|
||||
|
||||
This was validated in practice. When VectorStore pointed at the wrong collection, `auto` mode fell back to SDK and still returned results.
|
||||
|
||||
## Why Use Spring AI VectorStore At All?
|
||||
|
||||
Using Spring AI `VectorStore` moves the project closer to a standard RAG abstraction:
|
||||
|
||||
- Retrieval code no longer needs to own all Milvus-specific search details.
|
||||
- Later features such as query transformers, document postprocessors, advisors, or retrievers can be introduced more naturally.
|
||||
- The code becomes easier to compare with common Spring AI RAG patterns in an interview.
|
||||
|
||||
But I did not blindly replace everything. Writes/indexing still use SDK because changing read and write paths at the same time would make failures harder to isolate.
|
||||
|
||||
## Why Keep L0?
|
||||
|
||||
L0 is no longer treated as the final source of truth. It is a deterministic hint layer:
|
||||
|
||||
- It extracts domain/entity hints from indexed metadata.
|
||||
- It helps constrain L1 retrieval by category when possible.
|
||||
- It gives the Agent a stable clue even when semantic retrieval is weak.
|
||||
|
||||
The current design is:
|
||||
|
||||
```text
|
||||
L0 = domain/entity hint
|
||||
L1 = semantic retrieval through VectorStore/SDK
|
||||
postprocess = evidence trace and relevance normalization
|
||||
```
|
||||
|
||||
This is easier to defend than saying "we only use vector search." Real incident diagnosis often has exact identifiers, error codes, service names, and alert names. L0 is useful for those.
|
||||
|
||||
## Why Not Use Hidden Spring AI Advisors Directly?
|
||||
|
||||
For this project, `lookup_knowledge` remains an explicit tool.
|
||||
|
||||
Reason:
|
||||
|
||||
- The Agent trace needs to show when knowledge was retrieved.
|
||||
- `tool_invocation` records input, output preview, relevance level, and evidence metadata.
|
||||
- The interview story is about auditable Agent execution, not only answer quality.
|
||||
|
||||
Spring AI Advisors may be useful later, but hiding retrieval inside an advisor would make the evidence chain less visible unless we rebuild trace hooks around it.
|
||||
|
||||
## Score Design
|
||||
|
||||
The result object intentionally separates these fields:
|
||||
|
||||
```text
|
||||
score -> compatibility score used by old relevance normalization
|
||||
rawScore -> raw score from the retrieval implementation
|
||||
scoreLabel -> semantic label for rawScore
|
||||
```
|
||||
|
||||
For SDK:
|
||||
|
||||
```text
|
||||
score = L2 distance
|
||||
rawScore = L2 distance
|
||||
scoreLabel = l2_distance
|
||||
```
|
||||
|
||||
For VectorStore:
|
||||
|
||||
```text
|
||||
score = metadata.distance if present
|
||||
rawScore = Spring AI document score
|
||||
scoreLabel = similarity
|
||||
```
|
||||
|
||||
This prevents a subtle bug: if we treat Spring AI similarity as L2 distance, relevance becomes wrong. If we only expose distance, we lose the ability to compare Spring AI behavior. Keeping both makes the migration inspectable.
|
||||
|
||||
## How I Verified It
|
||||
|
||||
I verified at three levels:
|
||||
|
||||
- Unit tests: SDK mode, auto VectorStore mode, fallback mode, category filter, distance metadata mapping.
|
||||
- Live API: `/api/search/similar?query=ERR_TIMEOUT&topK=3`.
|
||||
- Logs: confirmed whether the path was VectorStore success or SDK fallback.
|
||||
|
||||
The live API returned:
|
||||
|
||||
```text
|
||||
scoreLabel = similarity
|
||||
rawScore = Spring AI similarity
|
||||
score = Milvus distance metadata
|
||||
```
|
||||
|
||||
That means the main path was Spring AI VectorStore and compatibility scoring remained stable.
|
||||
|
||||
## What I Would Do Next
|
||||
|
||||
I would not immediately migrate indexing writes. The next responsible steps are:
|
||||
|
||||
- Add a small live acceptance report for several golden queries.
|
||||
- Compare `sdk` and `spring-ai` mode side by side for topK overlap.
|
||||
- Decide whether `VectorIndexService` should move to `VectorStore.add(...)`.
|
||||
- Add query transformation or hybrid retrieval only after we have baseline metrics.
|
||||
|
||||
This staged approach is intentional: first stabilize the read path, then evaluate retrieval quality, then migrate writes if the abstraction proves reliable.
|
||||
|
||||
## Interview Questions And Short Answers
|
||||
|
||||
### Why did you not remove the SDK?
|
||||
|
||||
Because this is a migration, not a rewrite. SDK fallback gives rollback safety and proved useful when VectorStore config was initially wrong.
|
||||
|
||||
### What changed for `lookup_knowledge`?
|
||||
|
||||
The public contract did not change. It still calls `VectorSearchService.searchSimilarDocuments(...)`. The implementation behind that facade changed.
|
||||
|
||||
### How do you know VectorStore is actually used?
|
||||
|
||||
The logs show `Starting Spring AI VectorStore search` followed by `Spring AI VectorStore search complete`. The API response also has `scoreLabel=similarity`, which only comes from the VectorStore path.
|
||||
|
||||
### What was the main bug found during live validation?
|
||||
|
||||
The configured collection name was wrong. Spring AI looked for `business_knowledge`, but the actual Milvus collection was `biz`.
|
||||
|
||||
### What did fallback prove?
|
||||
|
||||
It proved that `auto` mode is resilient: VectorStore failed, SDK search still returned valid results, and the API did not fail.
|
||||
|
||||
### Why is `metadata.distance` important?
|
||||
|
||||
Because `lookup_knowledge` uses L2 distance normalization. Spring AI returns similarity as the main document score, but the Milvus distance is available in metadata. Using it preserves old relevance behavior.
|
||||
|
||||
### Is this full Spring AI RAG now?
|
||||
|
||||
Not yet. It uses Spring AI VectorStore for the main read path, but keeps explicit tools, custom evidence trace, L0 hints, and SDK indexing. That is deliberate because the project values auditability and staged migration.
|
||||
@@ -0,0 +1,199 @@
|
||||
# RAG VectorStore Live Acceptance
|
||||
|
||||
## Purpose
|
||||
|
||||
This note records the live acceptance result for the RAG retrieval refactor.
|
||||
|
||||
The goal of this refactor was not only to add a Spring AI abstraction, but to prove that the production retrieval path can:
|
||||
|
||||
- Prefer Spring AI `VectorStore` for Milvus retrieval.
|
||||
- Preserve the existing Milvus SDK path as fallback.
|
||||
- Keep the `lookup_knowledge` tool contract stable.
|
||||
- Keep L2-distance based relevance normalization compatible.
|
||||
|
||||
## Current Retrieval Shape
|
||||
|
||||
```text
|
||||
lookup_knowledge / /api/search/similar
|
||||
-> VectorSearchService.searchSimilarDocuments(...)
|
||||
-> retrieval.vector-store.mode
|
||||
-> auto
|
||||
-> Spring AI VectorStore
|
||||
-> fallback to Milvus SDK if VectorStore fails
|
||||
-> spring-ai
|
||||
-> Spring AI VectorStore only
|
||||
-> sdk
|
||||
-> Milvus SDK only
|
||||
```
|
||||
|
||||
## Configuration Verified
|
||||
|
||||
The live Milvus/Zilliz database contains the collection:
|
||||
|
||||
```text
|
||||
biz
|
||||
```
|
||||
|
||||
The Spring AI VectorStore configuration was aligned with the existing SDK collection:
|
||||
|
||||
```yaml
|
||||
spring:
|
||||
ai:
|
||||
vectorstore:
|
||||
type: milvus
|
||||
milvus:
|
||||
initialize-schema: false
|
||||
database-name: ${milvus.database}
|
||||
collection-name: biz
|
||||
embedding-dimension: ${milvus.vector-dim}
|
||||
metric-type: L2
|
||||
id-field-name: id
|
||||
content-field-name: content
|
||||
metadata-field-name: metadata
|
||||
embedding-field-name: vector
|
||||
```
|
||||
|
||||
Why this matters: the earlier config used `business_knowledge`, but the SDK path and real collection use `biz`. That mismatch proved the fallback worked, but it also meant VectorStore was not the successful main path until the config was corrected.
|
||||
|
||||
## Commands Used
|
||||
|
||||
Health check:
|
||||
|
||||
```powershell
|
||||
Invoke-RestMethod `
|
||||
-Uri "http://127.0.0.1:9900/milvus/health" `
|
||||
-Method Get
|
||||
```
|
||||
|
||||
Observed result:
|
||||
|
||||
```json
|
||||
{
|
||||
"collections": ["biz"],
|
||||
"message": "ok"
|
||||
}
|
||||
```
|
||||
|
||||
Direct retrieval check:
|
||||
|
||||
```powershell
|
||||
Invoke-RestMethod `
|
||||
-Uri "http://127.0.0.1:9900/api/search/similar?query=ERR_TIMEOUT&topK=3" `
|
||||
-Method Get
|
||||
```
|
||||
|
||||
Observed result shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"code": 200,
|
||||
"message": "success",
|
||||
"data": [
|
||||
{
|
||||
"id": "f7dff7c8-5665-3145-9f75-ef741528b914",
|
||||
"content": "### ERR_TIMEOUT ...",
|
||||
"score": 0.5662,
|
||||
"rawScore": 0.4337,
|
||||
"scoreLabel": "similarity",
|
||||
"metadata": {
|
||||
"distance": 0.5662,
|
||||
"title": "ERR_TIMEOUT",
|
||||
"category": "api"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## What The Logs Proved
|
||||
|
||||
Before collection alignment:
|
||||
|
||||
```text
|
||||
Starting Spring AI VectorStore search
|
||||
SearchRequest collectionName:business_knowledge failed
|
||||
Spring AI VectorStore retrieval failed, falling back to Milvus SDK
|
||||
Starting Milvus SDK search
|
||||
```
|
||||
|
||||
After collection alignment:
|
||||
|
||||
```text
|
||||
Starting Spring AI VectorStore search: query=ERR_TIMEOUT
|
||||
Spring AI VectorStore search complete, candidates=3
|
||||
```
|
||||
|
||||
This proves:
|
||||
|
||||
- `auto` mode really attempts VectorStore first.
|
||||
- The fallback is functional when VectorStore fails.
|
||||
- After config alignment, the main path is Spring AI VectorStore rather than SDK fallback.
|
||||
|
||||
## Score Semantics
|
||||
|
||||
The project keeps three score fields intentionally:
|
||||
|
||||
```text
|
||||
rawScore -> the raw score from the active retrieval implementation
|
||||
scoreLabel -> the semantic meaning of rawScore
|
||||
score -> compatibility score used by existing lookup relevance normalization
|
||||
```
|
||||
|
||||
For SDK retrieval:
|
||||
|
||||
```text
|
||||
rawScore = L2 distance
|
||||
scoreLabel = l2_distance
|
||||
score = L2 distance
|
||||
```
|
||||
|
||||
For Spring AI VectorStore retrieval:
|
||||
|
||||
```text
|
||||
rawScore = Spring AI similarity score
|
||||
scoreLabel = similarity
|
||||
score = Milvus distance metadata when available
|
||||
```
|
||||
|
||||
Why use `metadata.distance` for `score`: `LookupKnowledgeTool` already normalizes relevance from L2 distance. Spring AI Milvus returns similarity as the document score, but also includes the Milvus distance in metadata. Using distance preserves the old relevance behavior while still exposing the new VectorStore score semantics through `rawScore` and `scoreLabel`.
|
||||
|
||||
## Regression Checks
|
||||
|
||||
Targeted tests:
|
||||
|
||||
```powershell
|
||||
mvn -q "-Dtest=VectorSearchServiceTest,LookupKnowledgeToolTest" test
|
||||
```
|
||||
|
||||
Spec validation:
|
||||
|
||||
```powershell
|
||||
openspec.cmd validate rag-knowledge-retrieval --specs
|
||||
openspec.cmd validate rag-retrieval-evaluation --specs
|
||||
```
|
||||
|
||||
Whitespace check:
|
||||
|
||||
```powershell
|
||||
git diff --check
|
||||
```
|
||||
|
||||
Observed result:
|
||||
|
||||
```text
|
||||
All targeted tests passed.
|
||||
All related specs passed.
|
||||
No diff-check errors.
|
||||
```
|
||||
|
||||
## Acceptance Conclusion
|
||||
|
||||
The VectorStore refactor is accepted for the read path:
|
||||
|
||||
- Spring AI VectorStore is integrated and selected in `auto` mode.
|
||||
- The SDK path remains available and was proven by fallback behavior.
|
||||
- The live collection configuration is aligned with the existing Milvus collection.
|
||||
- The `lookup_knowledge` public contract remains stable.
|
||||
- Existing L2-based relevance normalization remains compatible.
|
||||
|
||||
The write/indexing path still uses the Milvus SDK. That is an intentional staged migration decision, not a failed acceptance item.
|
||||
@@ -202,6 +202,7 @@ relevanceLevel
|
||||
- 覆盖 Chat 和 AIOps 场景。
|
||||
- 每条 query 标注 expected doc、breadcrumb、关键 chunk 或 evidence。
|
||||
- 用当前链路跑一遍,记录 baseline。
|
||||
- 初始离线基线落在 `eval/rag-retrieval/`,用于后续 change 对比。
|
||||
|
||||
验收:
|
||||
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-05
|
||||
@@ -0,0 +1,59 @@
|
||||
## Context
|
||||
|
||||
Chat diagnosis has a Verifier Agent that writes structured evaluation into `diagnosis_session.self_evaluation`. AIOps currently focuses on payload scoping, evidence tools, and trace persistence, but it has no quality gate that checks whether the final report stayed on target or used evidence.
|
||||
|
||||
The next stage should add a low-risk quality gate before considering a full AIOps LLM verifier.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Evaluate AIOps final reports with deterministic rules.
|
||||
- Persist the evaluation under a dedicated `aiops_rule_evaluation` self-evaluation key.
|
||||
- Keep trace replay able to show whether AIOps output passed, warned, or failed basic quality checks.
|
||||
- Add focused unit tests without requiring live LLMs or external tools.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not add an AIOps Verifier Agent yet.
|
||||
- Do not route/retry AIOps execution based on the evaluation result.
|
||||
- Do not change `tool_invocation` schema.
|
||||
- Do not require new database migrations.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision 1: Rule-Based Before LLM Verifier
|
||||
|
||||
The first AIOps verifier is a deterministic evaluator, not an LLM agent.
|
||||
|
||||
Rationale:
|
||||
|
||||
- AIOps quality risks are concrete at this stage: payload focus, evidence coverage, and report presence.
|
||||
- Rule evaluation is cheap, stable, and easy to explain in an interview.
|
||||
- A full verifier agent can be added later once AIOps trace expectations are stable.
|
||||
|
||||
### Decision 2: Dedicated Self-Evaluation Channel
|
||||
|
||||
Persist under `aiops_rule_evaluation` instead of reusing `rule_evaluation` or `verifier_evaluation`.
|
||||
|
||||
Rationale:
|
||||
|
||||
- `verifier_evaluation` is already associated with Chat's LLM verifier.
|
||||
- `rule_evaluation` may be used by generic diagnosis evaluation.
|
||||
- A dedicated key avoids conflating AIOps-specific checks with other evaluation channels.
|
||||
|
||||
### Decision 3: Evaluate After Final Report Persistence
|
||||
|
||||
Run the evaluator when `persistFinalReport(...)` is called.
|
||||
|
||||
Rationale:
|
||||
|
||||
- It has access to the final report and session id.
|
||||
- It can read persisted tool invocations for the same session.
|
||||
- It does not disturb the Agent execution path.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- [Risk] Rule evaluation can miss semantic hallucinations. -> Mitigation: position it as lightweight AIOps quality gate, not full groundedness verification.
|
||||
- [Risk] Strict keyword checks may warn on valid reports with different wording. -> Mitigation: use WARN for missing soft signals and FAIL only for critical absence.
|
||||
- [Risk] Evaluation after report persistence does not trigger retries. -> Mitigation: keep routing unchanged in this phase; later changes can consume the verdict.
|
||||
@@ -0,0 +1,26 @@
|
||||
## Why
|
||||
|
||||
AIOps now has traceable payload scope control and improved RAG retrieval, but it still lacks a quality gate comparable to Chat's verifier. A lightweight rule-based verifier can check the most important AIOps risks without introducing another LLM agent.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add a rule-based AIOps evaluation service that checks final report quality after the AIOps flow completes.
|
||||
- Persist the evaluation under `diagnosis_session.self_evaluation.aiops_rule_evaluation`.
|
||||
- Evaluate payload focus, evidence-tool coverage, and basic report completeness.
|
||||
- Expose the evaluation through the existing trace API self-evaluation payload.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `aiops-traceable-diagnosis-entry`: AIOps sessions include a lightweight rule evaluation for trace replay.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects AIOps session finalization and trace self-evaluation.
|
||||
- Does not change AIOps API input, Agent flow topology, tool signatures, or database schema.
|
||||
- Does not add an LLM verifier agent.
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: AIOps sessions SHALL persist lightweight rule evaluation
|
||||
When an AIOps final report is persisted, the system SHALL evaluate it with deterministic AIOps-specific quality rules and store the result in session self-evaluation.
|
||||
|
||||
#### Scenario: Payload-focused report is evaluated
|
||||
- **WHEN** an AIOps session has alert payload fields and a final report is persisted
|
||||
- **THEN** the system SHALL evaluate whether the report mentions the supplied alert and service
|
||||
- **AND** it SHALL store the result under `self_evaluation.aiops_rule_evaluation`
|
||||
|
||||
#### Scenario: Evidence coverage is evaluated
|
||||
- **WHEN** an AIOps final report is evaluated
|
||||
- **THEN** the system SHALL check whether evidence tool invocations such as `lookup_knowledge`, `query_metrics`, or `query_logs` were persisted for the session
|
||||
|
||||
#### Scenario: Evaluation is traceable
|
||||
- **WHEN** the diagnosis trace API returns an AIOps session
|
||||
- **THEN** the session self-evaluation payload SHALL include `aiops_rule_evaluation` when it has been generated
|
||||
|
||||
#### Scenario: Evaluation uses stable verdicts
|
||||
- **WHEN** AIOps rule evaluation completes
|
||||
- **THEN** it SHALL produce a verdict from `PASS`, `WARN`, or `FAIL`
|
||||
- **AND** it SHALL include check details and a human-readable rationale
|
||||
@@ -0,0 +1,19 @@
|
||||
## 1. Rule Evaluation
|
||||
|
||||
- [x] 1.1 Add an AIOps rule evaluation service with PASS/WARN/FAIL verdicts.
|
||||
- [x] 1.2 Check payload focus, evidence-tool coverage, and report completeness.
|
||||
|
||||
## 2. AIOps Integration
|
||||
|
||||
- [x] 2.1 Persist AIOps rule evaluation when the final AIOps report is saved.
|
||||
- [x] 2.2 Make trace summary indicate that AIOps rule evaluation exists.
|
||||
|
||||
## 3. Tests And Docs
|
||||
|
||||
- [x] 3.1 Add focused unit tests for the evaluator and AIOps integration.
|
||||
- [x] 3.2 Add interview notes for the lightweight AIOps verifier.
|
||||
|
||||
## 4. Verification
|
||||
|
||||
- [x] 4.1 Run focused service tests.
|
||||
- [x] 4.2 Validate the OpenSpec change and review git scope.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-04
|
||||
@@ -0,0 +1,53 @@
|
||||
## Context
|
||||
|
||||
The previous change converted L0 into a hint provider and made L0/L1 cooperate. The next step is to stop treating the tool output as an unstructured primary/supplement pair. The Agent can keep receiving compatible fields, but retrieval internals and traces should have structured evidence blocks.
|
||||
|
||||
This change is a bridge toward later DocumentPostProcessor-style behavior. It should be small enough to archive independently and should not introduce Spring AI dependencies.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Add `EvidenceBlock` DTOs to `LookupResult`.
|
||||
- Create evidence blocks from L0 matches and L1 results.
|
||||
- Deduplicate evidence by stable source key.
|
||||
- Capture source, title, breadcrumb, score, retrieval layer, hit reasons, and content preview.
|
||||
- Persist evidence blocks and postprocess counts in `tool_invocation.retrieval_details`.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not implement neighbor chunk expansion yet.
|
||||
- Do not replace primary/supplement output.
|
||||
- Do not add cross-encoder or LLM rerank.
|
||||
- Do not migrate to Spring AI DocumentPostProcessor yet.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision: Add evidence blocks while preserving existing result fields
|
||||
|
||||
`LookupResult` will gain `List<EvidenceBlock> evidenceBlocks`. Existing `primary`, `supplement`, `found`, `relevanceLevel`, and `completenessHint` remain compatible.
|
||||
|
||||
Rationale: this lets the Agent continue using the current shape while tests and traces begin validating the new evidence model.
|
||||
|
||||
### Decision: Keep postprocess rule-based
|
||||
|
||||
The evidence builder will use deterministic rules:
|
||||
|
||||
- L0 entries become `L0` evidence.
|
||||
- L1 candidates become `L1` evidence.
|
||||
- Same source key is deduplicated.
|
||||
- Hit reasons are collected from L0 hints, L1 rank, category filters, and fallback state.
|
||||
|
||||
Rationale: this is explainable, cheap, and suitable before introducing framework postprocessors.
|
||||
|
||||
### Decision: Persist compact evidence summaries
|
||||
|
||||
`ToolInvocationRecorder` will store compact evidence block metadata, not full content, inside `retrieval_details`.
|
||||
|
||||
Rationale: `tool_invocation` should remain useful for trace review without duplicating large chunks.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- Evidence source keys may be imperfect before full metadata normalization -> fall back to file path, metadata, title, then rank.
|
||||
- Agent prompts may ignore `evidenceBlocks` initially -> keep primary/supplement compatibility.
|
||||
- Adding content previews increases tool output size -> cap evidence content length.
|
||||
@@ -0,0 +1,28 @@
|
||||
## Why
|
||||
|
||||
The current `lookup_knowledge` result is still shaped as one L0 primary result plus one L1 supplement. That makes retrieval evidence hard to inspect, hard to deduplicate, and hard to reuse later by verifier/evaluator code. The RAG refactor needs a structured evidence layer before Spring AI retriever migration.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add structured evidence blocks to `LookupResult`.
|
||||
- Build evidence blocks from L0 hint matches and L1 candidates.
|
||||
- Deduplicate evidence by source identity where possible.
|
||||
- Add hit reasons such as L0 matched keywords, L1 semantic rank, category filter, and fallback.
|
||||
- Persist evidence block summaries and postprocess counts in `tool_invocation.retrieval_details`.
|
||||
- Keep existing `primary` and `supplement` fields for compatibility.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
- None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `rag-knowledge-retrieval`: Add evidence block post-processing requirements for explicit RAG knowledge retrieval.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects `LookupResult`, `LookupKnowledgeTool`, and `ToolInvocationRecorder`.
|
||||
- Updates tool tests and trace recorder tests.
|
||||
- Does not change Milvus schema, document upload, chunking, or Spring AI integration.
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL return structured evidence blocks
|
||||
The `lookup_knowledge` retrieval flow SHALL expose retrieved evidence as structured evidence blocks in addition to the existing compatibility fields.
|
||||
|
||||
#### Scenario: Evidence block contains source metadata
|
||||
- **WHEN** a `lookup_knowledge` call returns evidence
|
||||
- **THEN** each evidence block SHALL include source, title when available, breadcrumb when available, retrieval layer, content, and hit reasons
|
||||
|
||||
#### Scenario: Compatibility fields remain available
|
||||
- **WHEN** evidence blocks are returned
|
||||
- **THEN** the existing `primary` and `supplement` result fields SHALL remain available when their source evidence exists
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL deduplicate evidence blocks
|
||||
The retrieval flow SHALL remove duplicate evidence blocks before returning them to the Agent.
|
||||
|
||||
#### Scenario: Duplicate source deduplication
|
||||
- **WHEN** L0 and L1 produce evidence with the same source identity
|
||||
- **THEN** the retrieval flow SHALL keep a single evidence block for that source
|
||||
- **AND** the evidence block SHALL preserve hit reasons from both retrieval paths when available
|
||||
|
||||
#### Scenario: Postprocess count tracking
|
||||
- **WHEN** evidence post-processing completes
|
||||
- **THEN** the tool invocation details SHALL record candidate count and final evidence block count
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL persist evidence block summaries
|
||||
The system SHALL persist compact evidence block summaries in `tool_invocation.retrieval_details`.
|
||||
|
||||
#### Scenario: Evidence summaries are persisted
|
||||
- **WHEN** a `lookup_knowledge` call records a tool invocation
|
||||
- **THEN** `retrieval_details` SHALL include evidence block summaries containing source, title, retrieval layer, score when available, and hit reasons
|
||||
|
||||
#### Scenario: Full content is not duplicated into retrieval details
|
||||
- **WHEN** evidence block summaries are persisted
|
||||
- **THEN** full evidence content SHALL be omitted or truncated so the trace record remains compact
|
||||
@@ -0,0 +1,26 @@
|
||||
## 1. Evidence Model
|
||||
|
||||
- [x] 1.1 Add an `EvidenceBlock` DTO.
|
||||
- [x] 1.2 Add evidence block list and postprocess count fields to `LookupResult`.
|
||||
|
||||
## 2. Evidence Postprocess
|
||||
|
||||
- [x] 2.1 Build evidence blocks from L0 matches and L1 candidates in `LookupKnowledgeTool`.
|
||||
- [x] 2.2 Deduplicate evidence by stable source key.
|
||||
- [x] 2.3 Preserve existing primary/supplement compatibility behavior.
|
||||
|
||||
## 3. Trace Recording
|
||||
|
||||
- [x] 3.1 Extend `ToolInvocationRecorder.LookupKnowledgeRecord` with evidence block summaries and postprocess counts.
|
||||
- [x] 3.2 Persist evidence block summaries in `retrieval_details`.
|
||||
|
||||
## 4. Tests
|
||||
|
||||
- [x] 4.1 Add or update tests for evidence block creation and deduplication.
|
||||
- [x] 4.2 Add or update tests for persisted evidence block summaries.
|
||||
- [x] 4.3 Run targeted tests and the RAG retrieval baseline evaluator.
|
||||
|
||||
## 5. Validation
|
||||
|
||||
- [x] 5.1 Run OpenSpec validation for the change.
|
||||
- [x] 5.2 Review git diff to confirm only expected code/spec/test files changed.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-04
|
||||
@@ -0,0 +1,62 @@
|
||||
## Context
|
||||
|
||||
`LookupKnowledgeTool` currently performs L0 keyword matching first. If L0 returns exactly one document, the tool treats it as high confidence, skips L1 semantic retrieval, and returns the L0-derived primary result. This was useful for the MVP but conflicts with the RAG refactor direction: L0 should constrain and explain retrieval, not decide final evidence by itself.
|
||||
|
||||
The refactor plan keeps L0 and metadata as valuable business signals. This change narrows L0 to a domain/entity hint provider while keeping `lookup_knowledge` as the explicit Agent tool entry point and preserving `tool_invocation` observability.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Produce structured L0 hints from current keyword/frontmatter matches.
|
||||
- Include matched keywords, domains, entities, and titles in the trace.
|
||||
- Run L1 retrieval by default even for unique L0 hits.
|
||||
- Use a single clear L0 domain as a category filter for L1.
|
||||
- Preserve existing result shape as much as possible.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not migrate to Spring AI VectorStore.
|
||||
- Do not implement BM25, RRF, rerank, or evidence packing.
|
||||
- Do not change document upload, chunking, or Milvus schema.
|
||||
- Do not remove L0.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision: Add a structured L0 hint result beside existing exact matches
|
||||
|
||||
`KnowledgeIndexService` will expose an `analyzeQuery` style method that returns:
|
||||
|
||||
- matched entries
|
||||
- matched keywords
|
||||
- domains/categories
|
||||
- entity terms
|
||||
|
||||
The existing `exactMatch` method can remain for compatibility.
|
||||
|
||||
Rationale: this avoids rewriting all callers while giving `LookupKnowledgeTool` richer data for tracing and filtering.
|
||||
|
||||
### Decision: Treat L0 unique hit as a hint, not a short circuit
|
||||
|
||||
`LookupKnowledgeTool` will no longer skip L1 solely because L0 matched one document. L1 will be called using the query and an optional category filter when L0 provides exactly one clear domain.
|
||||
|
||||
Rationale: the upcoming Spring AI retriever and evidence post-processing pipeline needs L0 and L1 to cooperate rather than use early return semantics.
|
||||
|
||||
### Decision: Keep `PRECISE` only when L0 and L1 both support the result
|
||||
|
||||
The relevance assessment should not mark `PRECISE` just because L0 matched once. It may mark `PRECISE` when L0 has one match and L1 returns evidence above the configured high relevance threshold, or when L0 has one match and L1 cannot run but the L0 result is still available.
|
||||
|
||||
Rationale: this preserves a graceful fallback while reducing overconfidence when semantic evidence disagrees.
|
||||
|
||||
### Decision: Persist L0 hints in retrieval details
|
||||
|
||||
`ToolInvocationRecorder.LookupKnowledgeRecord` will include fields for L0 matched keywords, domains, and entities. These will be serialized into `retrieval_details`.
|
||||
|
||||
Rationale: evidence trace and later evaluation need to explain why metadata filters or query augmentation happened.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- Increased latency because L1 is called more often -> keep topK small and allow category filter to reduce search scope.
|
||||
- L0 domain filter may be too narrow -> only apply it when there is exactly one nonblank domain; otherwise search without filter.
|
||||
- Existing tests may assume `L0` retrieval layer for unique hits -> update expectations to `L0+L1` when L1 participates.
|
||||
- If L1 fails, the tool should still return L0 evidence rather than fail the entire knowledge lookup.
|
||||
@@ -0,0 +1,30 @@
|
||||
## Why
|
||||
|
||||
The current `lookup_knowledge` implementation treats a unique L0 keyword hit as high confidence and skips L1 semantic retrieval. That makes L0 too authoritative for the RAG refactor target: L0 should provide domain/entity hints, metadata-filter intent, and explainability while final evidence still comes from the retrieval pipeline.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Change L0 from final retrieval decision maker to domain/entity hint provider.
|
||||
- Add structured L0 hint output that includes matched keywords, domains, entities, and matched titles.
|
||||
- Make `lookup_knowledge` run L1 semantic retrieval by default even when L0 has a unique hit.
|
||||
- Use L0 domain hints to pass category metadata filters into L1 when a single clear domain is detected.
|
||||
- Persist L0 hint details in `tool_invocation.retrieval_details`.
|
||||
- Keep `lookup_knowledge` as the explicit Agent tool entry point.
|
||||
- No Spring AI VectorStore migration in this change.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
- `rag-knowledge-retrieval`: Defines runtime behavior for the explicit RAG knowledge retrieval tool, including L0 hinting and L1 retrieval cooperation.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- None.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects `KnowledgeIndexService`, `LookupKnowledgeTool`, and `ToolInvocationRecorder`.
|
||||
- May affect retrieval latency because L1 is no longer skipped for unique L0 hits.
|
||||
- Improves traceability by recording L0 matched keywords/entities/domains in retrieval details.
|
||||
- Does not change document upload, chunking, Milvus schema, or Agent flow.
|
||||
+47
@@ -0,0 +1,47 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL keep L0 as a hint provider
|
||||
The `lookup_knowledge` retrieval flow SHALL retain L0 keyword/frontmatter matching but use it as domain, entity, and explainability hint data rather than as the sole final retrieval decision.
|
||||
|
||||
#### Scenario: L0 produces traceable hint data
|
||||
- **WHEN** L0 matches one or more indexed knowledge entries
|
||||
- **THEN** the retrieval flow SHALL expose matched titles, matched keywords, domains or categories, and entity terms as structured hint data
|
||||
|
||||
#### Scenario: L0 does not bypass semantic retrieval by default
|
||||
- **WHEN** L0 returns exactly one match
|
||||
- **THEN** the retrieval flow SHALL still attempt semantic L1 retrieval unless L1 is unavailable or explicitly disabled by configuration
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL use L0 domain as optional L1 filter
|
||||
The retrieval flow SHALL use L0 domain/category information as an optional metadata filter for L1 retrieval when the domain is unambiguous.
|
||||
|
||||
#### Scenario: Single domain filter
|
||||
- **WHEN** L0 hint data contains exactly one nonblank domain or category
|
||||
- **THEN** the L1 retrieval request SHALL include that category as a metadata filter
|
||||
|
||||
#### Scenario: Ambiguous domain fallback
|
||||
- **WHEN** L0 hint data contains zero domains or multiple domains
|
||||
- **THEN** the L1 retrieval request SHALL run without an L0-derived category filter
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL preserve fallback evidence
|
||||
The retrieval flow SHALL still return useful L0 evidence when L1 produces no usable result.
|
||||
|
||||
#### Scenario: L1 has no results
|
||||
- **WHEN** L0 has at least one match and L1 returns no candidates
|
||||
- **THEN** the tool SHALL return an L0-based primary result
|
||||
- **AND** the relevance assessment SHALL not claim semantic support from L1
|
||||
|
||||
#### Scenario: L1 fails
|
||||
- **WHEN** L0 has at least one match and L1 retrieval throws or fails
|
||||
- **THEN** the tool SHALL return an L0-based primary result
|
||||
- **AND** the tool invocation record SHALL preserve the L0 hint details
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL persist L0 hints
|
||||
The system SHALL persist L0 hint details in `tool_invocation.retrieval_details` for `lookup_knowledge` calls.
|
||||
|
||||
#### Scenario: Retrieval details include L0 hints
|
||||
- **WHEN** a `lookup_knowledge` call records a tool invocation
|
||||
- **THEN** `retrieval_details` SHALL include L0 matched keywords, domains, entities, and titles when available
|
||||
|
||||
#### Scenario: Retrieval layer reflects cooperating retrieval
|
||||
- **WHEN** both L0 hint data and L1 candidates participate in a lookup
|
||||
- **THEN** the recorded retrieval layer SHALL be `L0+L1`
|
||||
@@ -0,0 +1,27 @@
|
||||
## 1. L0 Hint Model
|
||||
|
||||
- [x] 1.1 Add structured L0 hint analysis in `KnowledgeIndexService`.
|
||||
- [x] 1.2 Preserve `exactMatch` compatibility for existing callers.
|
||||
|
||||
## 2. Retrieval Flow
|
||||
|
||||
- [x] 2.1 Update `LookupKnowledgeTool` so unique L0 hits no longer skip L1 by default.
|
||||
- [x] 2.2 Apply a category filter to L1 only when L0 hint data has one clear domain.
|
||||
- [x] 2.3 Preserve L0 fallback evidence when L1 is empty or fails.
|
||||
- [x] 2.4 Adjust relevance assessment so `PRECISE` no longer depends only on unique L0.
|
||||
|
||||
## 3. Trace Recording
|
||||
|
||||
- [x] 3.1 Extend `ToolInvocationRecorder.LookupKnowledgeRecord` with L0 matched keywords, domains, and entities.
|
||||
- [x] 3.2 Persist L0 hint fields in `retrieval_details`.
|
||||
|
||||
## 4. Tests
|
||||
|
||||
- [x] 4.1 Add or update unit tests for L0 hint extraction.
|
||||
- [x] 4.2 Add or update tests for `lookup_knowledge` unique-L0 plus L1 participation.
|
||||
- [x] 4.3 Run targeted tests and the RAG retrieval baseline evaluator.
|
||||
|
||||
## 5. Validation
|
||||
|
||||
- [x] 5.1 Run OpenSpec validation for the change.
|
||||
- [x] 5.2 Review git diff to confirm only expected code/spec/test files changed.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-04
|
||||
@@ -0,0 +1,62 @@
|
||||
## Context
|
||||
|
||||
The repository already has diagnosis-level evaluation specs and trace observability, but the RAG refactor plan needs a narrower retrieval baseline. The upcoming changes will alter L0 responsibilities, query augmentation, evidence post-processing, and eventually the vector store implementation. Those changes need a fixed set of retrieval cases and deterministic scoring before production retrieval behavior changes.
|
||||
|
||||
The first baseline must be offline. It should not require MySQL, Milvus, Redis, LLM calls, or a running Spring Boot application. It can evaluate saved retrieval result fixtures that represent the current behavior and produce reports that future changes can compare against.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Add a small fixed golden query set for RAG retrieval.
|
||||
- Evaluate retrieval fixtures against expected documents, breadcrumbs, evidence keywords, and hit levels.
|
||||
- Produce JSON and Markdown baseline reports.
|
||||
- Document how to regenerate the reports.
|
||||
- Keep the evaluator simple enough to run from the repository with Python.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not change `lookup_knowledge`, `VectorSearchService`, Milvus schema, L0 matching, or Agent prompts.
|
||||
- Do not require live services.
|
||||
- Do not implement Spring AI VectorStore migration in this change.
|
||||
- Do not implement RRF, BM25, rerank, or evidence packing in this change.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision: Use offline retrieval fixtures first
|
||||
|
||||
The evaluator will read saved retrieval fixtures rather than calling the live application.
|
||||
|
||||
Rationale: the first change should establish a stable measurement surface before the RAG internals change. Live retrieval depends on embeddings, Milvus state, and service configuration, which makes it a poor first baseline.
|
||||
|
||||
Alternative considered: call `SearchController` or `lookup_knowledge` directly. That is useful later, but it would require a running app and seeded knowledge base.
|
||||
|
||||
### Decision: Score by hit level, not exact chunk id only
|
||||
|
||||
The evaluator will classify each case as:
|
||||
|
||||
- `strong`: expected document plus expected breadcrumb or key evidence coverage.
|
||||
- `medium`: expected document found, but breadcrumb or evidence coverage is incomplete.
|
||||
- `weak`: related evidence is present but the expected document is missing.
|
||||
- `miss`: no expected document or expected evidence is found.
|
||||
|
||||
Rationale: chunk indexes can change after splitter changes, so exact chunk-only scoring would make later refactors look worse even when evidence quality is preserved.
|
||||
|
||||
### Decision: Keep case format explicit and reviewable
|
||||
|
||||
Golden cases will be stored as JSON with fields such as `caseId`, `query`, `expectedDocIds`, `expectedBreadcrumbs`, `expectedKeywords`, and optional `notes`.
|
||||
|
||||
Rationale: the case file should be easy to inspect in code review and easy to extend during interviews or later refactors.
|
||||
|
||||
### Decision: Preserve both machine and human reports
|
||||
|
||||
The evaluator will write JSON for automation and Markdown for review.
|
||||
|
||||
Rationale: future changes can compare JSON, while the Markdown report is easier to use during design review and interview preparation.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- Offline fixtures can drift from real runtime behavior -> add documentation that this is a baseline harness, not a live retrieval accuracy claim.
|
||||
- Keyword-based evidence checks are approximate -> use them only as deterministic guardrails, not as a replacement for human review.
|
||||
- Small golden set may underrepresent production queries -> start with 10-20 cases and expand as new RAG issues are found.
|
||||
- Fixture schema may not match future retrieval outputs -> normalize fixtures into a simple candidate shape and keep raw fields optional.
|
||||
@@ -0,0 +1,27 @@
|
||||
## Why
|
||||
|
||||
The RAG refactor needs a repeatable baseline before changing L0, metadata filtering, post-processing, or Spring AI retriever integration. Without fixed retrieval cases and measurable output, later changes can look cleaner architecturally while silently degrading recall or evidence quality.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add a retrieval evaluation baseline for RAG queries, separate from full diagnosis evaluation.
|
||||
- Define golden retrieval cases covering Chat-style knowledge lookup and AIOps-style alert diagnosis retrieval.
|
||||
- Add a lightweight offline evaluator that compares retrieved candidates against expected documents, breadcrumbs, and evidence keywords.
|
||||
- Preserve baseline JSON and Markdown reports so future changes can compare retrieval behavior.
|
||||
- No production retrieval behavior changes in this change.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
- `rag-retrieval-evaluation`: Defines fixed retrieval golden cases, deterministic retrieval evaluation, and baseline report preservation.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- None.
|
||||
|
||||
## Impact
|
||||
|
||||
- Adds retrieval evaluation fixtures, documentation, and scripts.
|
||||
- May read existing retrieval/tool trace output or saved fixtures, but does not require live LLM calls.
|
||||
- Does not change the `lookup_knowledge` runtime behavior, Milvus schema, document upload API, or Agent flow.
|
||||
+61
@@ -0,0 +1,61 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL define fixed golden cases
|
||||
The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
|
||||
|
||||
#### Scenario: Golden case includes expected retrieval evidence
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
|
||||
|
||||
#### Scenario: Golden case distinguishes scenario type
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL run offline against fixtures
|
||||
The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
|
||||
|
||||
#### Scenario: Fixture evaluation
|
||||
- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
|
||||
- **THEN** it SHALL evaluate each case against its matching fixture file
|
||||
- **AND** it SHALL not call external services
|
||||
|
||||
#### Scenario: Missing fixture is reported
|
||||
- **WHEN** a golden case has no matching retrieval fixture
|
||||
- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL classify hit quality
|
||||
The evaluator SHALL classify each case into a deterministic hit level.
|
||||
|
||||
#### Scenario: Strong hit classification
|
||||
- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `strong`
|
||||
|
||||
#### Scenario: Medium hit classification
|
||||
- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `medium`
|
||||
|
||||
#### Scenario: Miss classification
|
||||
- **WHEN** retrieved candidates do not include expected documents or expected evidence
|
||||
- **THEN** the evaluator SHALL classify the case as `miss`
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL report ranking signals
|
||||
The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
|
||||
|
||||
#### Scenario: Per-case ranking output
|
||||
- **WHEN** a case is evaluated
|
||||
- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
|
||||
|
||||
#### Scenario: Aggregate metrics output
|
||||
- **WHEN** multiple cases are evaluated
|
||||
- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL preserve baseline reports
|
||||
The system SHALL preserve generated baseline reports in JSON and Markdown formats.
|
||||
|
||||
#### Scenario: Baseline report generation
|
||||
- **WHEN** the baseline evaluator is run for the fixed golden case set
|
||||
- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
|
||||
|
||||
#### Scenario: Baseline regeneration is documented
|
||||
- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
|
||||
- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
|
||||
@@ -0,0 +1,26 @@
|
||||
## 1. Golden Cases
|
||||
|
||||
- [x] 1.1 Create retrieval evaluation directory structure.
|
||||
- [x] 1.2 Add fixed golden retrieval cases covering Chat and AIOps retrieval scenarios.
|
||||
- [x] 1.3 Add matching offline retrieval fixtures for every golden case.
|
||||
|
||||
## 2. Evaluator
|
||||
|
||||
- [x] 2.1 Implement an offline retrieval evaluator script.
|
||||
- [x] 2.2 Support hit-level classification and first expected document rank.
|
||||
- [x] 2.3 Support JSON and Markdown report output.
|
||||
|
||||
## 3. Baseline Report
|
||||
|
||||
- [x] 3.1 Generate the baseline JSON report from the fixed cases and fixtures.
|
||||
- [x] 3.2 Generate the baseline Markdown report from the fixed cases and fixtures.
|
||||
|
||||
## 4. Documentation
|
||||
|
||||
- [x] 4.1 Document the retrieval baseline purpose, file layout, and regeneration command.
|
||||
- [x] 4.2 Link the retrieval baseline from the RAG refactor issue or related MVP documentation.
|
||||
|
||||
## 5. Verification
|
||||
|
||||
- [x] 5.1 Run the evaluator successfully against the fixed baseline cases.
|
||||
- [x] 5.2 Run OpenSpec status/validation for the change and confirm tasks are complete.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-04
|
||||
@@ -0,0 +1,71 @@
|
||||
## Context
|
||||
|
||||
The project already uses Spring AI/Spring AI Alibaba for model and agent capabilities, but RAG vector retrieval still uses the Milvus Java SDK directly. The current main path is now observable and covered by golden retrieval cases, so the next migration step should compare framework retrieval behavior without changing Chat or AIOps runtime behavior.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Introduce a Spring AI VectorStore sidecar behind configuration.
|
||||
- Keep `lookup_knowledge` and `VectorSearchService` as the default production path.
|
||||
- Normalize sidecar results into the same comparable shape as current `VectorSearchService.SearchResult`.
|
||||
- Add an offline or developer-triggered comparison report that runs golden cases through both retrieval paths.
|
||||
- Capture schema and scoring differences before deciding whether to replace the current implementation.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not replace `VectorSearchService` in this change.
|
||||
- Do not change document upload, chunking, or Milvus collection schema.
|
||||
- Do not introduce query transformer, multi-query, RRF, or rerank behavior.
|
||||
- Do not make Spring AI Advisor the RAG entry point.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision: Sidecar over replacement
|
||||
|
||||
Add a separate sidecar service/adapter instead of changing the existing retrieval service.
|
||||
|
||||
Rationale: the current path is already used by Chat and AIOps, and framework behavior may differ in score semantics, metadata filtering, or expected schema. A sidecar lets us compare before cutting over.
|
||||
|
||||
Alternative considered: replace `VectorSearchService` immediately. Rejected because it would conflate dependency integration with retrieval behavior migration.
|
||||
|
||||
### Decision: Preserve explicit tool boundary
|
||||
|
||||
The sidecar will be called by evaluation or diagnostic code, not by implicit Chat Advisor behavior.
|
||||
|
||||
Rationale: the interview value of the project is Agent engineering observability: explicit tool calls, evidence blocks, and `tool_invocation` traces.
|
||||
|
||||
Alternative considered: use Spring AI Advisor directly. Rejected for now because it hides the decision point where the Agent chooses retrieval.
|
||||
|
||||
### Decision: Compare normalized results
|
||||
|
||||
Both retrieval paths should be mapped into a small comparable result shape containing source/doc id, title, breadcrumb, category, score/distance, rank, and content preview.
|
||||
|
||||
Rationale: direct score equality is unlikely because the current path uses Milvus L2 distance while Spring AI abstractions may expose similarity scores or provider-specific values. The first useful comparison is source/rank/metadata coverage.
|
||||
|
||||
### Decision: Keep dependency risk isolated
|
||||
|
||||
If the current dependency set does not expose a compatible Milvus VectorStore, the first implementation should add a narrow optional dependency/config class and keep it disabled by default.
|
||||
|
||||
Rationale: Spring AI version compatibility is a migration risk. The project should still build and run with the current main path if sidecar configuration is absent.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- Spring AI Milvus schema may not match the existing collection -> keep sidecar disabled by default and report incompatibility rather than failing the app.
|
||||
- Score semantics may differ from current L2 distance -> compare rank/source metadata first and label score fields by retrieval path.
|
||||
- Adding framework dependencies may affect startup auto-configuration -> guard sidecar beans behind properties or conditions.
|
||||
- Sidecar evaluation may require live Milvus unlike the baseline fixture evaluator -> make live comparison opt-in and keep offline baseline unchanged.
|
||||
|
||||
## Migration Plan
|
||||
|
||||
1. Add sidecar configuration and adapter behind `rag.sidecar.spring-ai.enabled=false`.
|
||||
2. Add comparison command/script/service that runs golden cases through current retrieval plus sidecar when enabled.
|
||||
3. Store comparison reports separately from the offline baseline reports.
|
||||
4. Use report differences to decide whether a later change should replace `VectorSearchService` internals.
|
||||
5. Rollback is disabling the sidecar property or reverting the sidecar dependency/config only; the main path remains unchanged.
|
||||
|
||||
## Open Questions
|
||||
|
||||
- Which exact Spring AI Milvus VectorStore artifact is compatible with the existing Spring AI/Spring AI Alibaba BOM versions?
|
||||
- Can the current Milvus collection be queried by Spring AI VectorStore without schema migration, or do we need a second collection for sidecar experiments?
|
||||
- Should sidecar comparison run from Java tests, a script, or a developer-only endpoint/runner?
|
||||
@@ -0,0 +1,27 @@
|
||||
## Why
|
||||
|
||||
The current RAG retrieval path talks to Milvus through the raw Java SDK, so framework-level retrieval behavior cannot be compared safely. Before replacing the main path, we need a Spring AI VectorStore sidecar that can run the same golden cases and expose differences without affecting `lookup_knowledge`.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add a disabled-by-default Spring AI VectorStore sidecar retrieval path.
|
||||
- Keep the current `VectorSearchService` as the production path for Chat and AIOps.
|
||||
- Add an adapter/reporting surface that can run golden retrieval cases against both current and sidecar paths.
|
||||
- Record comparable fields: result id/source, title, breadcrumb, score/distance, category, and metadata.
|
||||
- Document incompatibilities between the current Milvus schema and Spring AI VectorStore behavior.
|
||||
- No breaking changes.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
- None.
|
||||
|
||||
### Modified Capabilities
|
||||
- `rag-knowledge-retrieval`: Add requirements for sidecar Spring AI retrieval comparison while preserving the explicit `lookup_knowledge` tool boundary.
|
||||
- `rag-retrieval-evaluation`: Add requirements for comparing baseline retrieval with the sidecar retriever on the golden case set.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects retrieval service wiring, configuration, and evaluation scripts.
|
||||
- May add Spring AI VectorStore dependency/configuration if the current dependency set does not already expose it.
|
||||
- Does not change document upload, chunking, Milvus collection schema, Agent prompts, AIOps diagnosis flow, or the default `lookup_knowledge` runtime path.
|
||||
+25
@@ -0,0 +1,25 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL support a disabled-by-default Spring AI sidecar
|
||||
The retrieval system SHALL allow a Spring AI VectorStore retrieval path to be wired as a sidecar without changing the default `lookup_knowledge` runtime path.
|
||||
|
||||
#### Scenario: Sidecar disabled by default
|
||||
- **WHEN** the application starts without explicit sidecar enablement
|
||||
- **THEN** `lookup_knowledge` SHALL continue using the existing retrieval path
|
||||
- **AND** Chat and AIOps runtime behavior SHALL not depend on the sidecar
|
||||
|
||||
#### Scenario: Sidecar failure does not break main retrieval
|
||||
- **WHEN** the Spring AI sidecar is enabled but cannot initialize or query successfully
|
||||
- **THEN** the existing retrieval path SHALL remain usable
|
||||
- **AND** the failure SHALL be reported as sidecar status rather than as a main retrieval failure
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL normalize sidecar results for comparison
|
||||
The sidecar retrieval path SHALL expose results in a comparable structure aligned with the current retrieval result shape.
|
||||
|
||||
#### Scenario: Comparable result metadata
|
||||
- **WHEN** sidecar retrieval returns candidates
|
||||
- **THEN** each comparable result SHALL include source or doc id, title when available, breadcrumb when available, category when available, rank, content preview, and the sidecar score label/value
|
||||
|
||||
#### Scenario: Score semantics are explicit
|
||||
- **WHEN** current retrieval and sidecar retrieval scores are compared
|
||||
- **THEN** the report SHALL label score semantics by path instead of assuming direct numeric equivalence
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL compare current and sidecar retrieval paths
|
||||
The retrieval evaluation system SHALL provide an opt-in comparison between the existing retrieval path and the Spring AI sidecar retrieval path.
|
||||
|
||||
#### Scenario: Sidecar comparison report
|
||||
- **WHEN** sidecar comparison is run for the golden case set
|
||||
- **THEN** the report SHALL include per-case current-path top candidates and sidecar top candidates
|
||||
- **AND** it SHALL highlight source, breadcrumb, category, rank, and score-label differences
|
||||
|
||||
#### Scenario: Offline baseline remains unchanged
|
||||
- **WHEN** the fixture-based offline baseline evaluator is run
|
||||
- **THEN** it SHALL not require live Milvus, Spring Boot, or Spring AI sidecar configuration
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL make sidecar readiness visible
|
||||
The sidecar comparison report SHALL show whether the Spring AI sidecar was runnable for the current environment.
|
||||
|
||||
#### Scenario: Sidecar unavailable
|
||||
- **WHEN** sidecar comparison is requested but the sidecar is disabled or unavailable
|
||||
- **THEN** the report SHALL mark sidecar status as unavailable
|
||||
- **AND** it SHALL keep current-path baseline results available for review
|
||||
@@ -0,0 +1,24 @@
|
||||
## 1. Dependency And Configuration
|
||||
|
||||
- [x] 1.1 Inspect available Spring AI VectorStore/Milvus classes for the current dependency set.
|
||||
- [x] 1.2 Add the narrow dependency or optional configuration needed for the sidecar path.
|
||||
- [x] 1.3 Add disabled-by-default sidecar properties under RAG configuration.
|
||||
|
||||
## 2. Sidecar Retrieval Adapter
|
||||
|
||||
- [x] 2.1 Define a comparable retrieval result DTO for current and sidecar paths.
|
||||
- [x] 2.2 Implement a Spring AI sidecar retrieval service that reports readiness and failures without breaking the main path.
|
||||
- [x] 2.3 Normalize sidecar metadata into source/doc id, title, breadcrumb, category, rank, content preview, and score label/value.
|
||||
|
||||
## 3. Comparison Evaluation
|
||||
|
||||
- [x] 3.1 Add a comparison service or script that runs golden cases against current retrieval and the sidecar path when enabled.
|
||||
- [x] 3.2 Write sidecar comparison JSON/Markdown reports separate from the offline baseline reports.
|
||||
- [x] 3.3 Preserve the existing offline evaluator behavior without requiring live Spring AI/Milvus services.
|
||||
|
||||
## 4. Tests And Validation
|
||||
|
||||
- [x] 4.1 Add tests for disabled sidecar fallback/readiness behavior.
|
||||
- [x] 4.2 Add tests for comparable result normalization and report generation.
|
||||
- [x] 4.3 Run targeted tests, the offline RAG retrieval baseline evaluator, and OpenSpec validation.
|
||||
- [x] 4.4 Review git diff to confirm the default `lookup_knowledge` runtime path is unchanged.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-05
|
||||
@@ -0,0 +1,53 @@
|
||||
## Context
|
||||
|
||||
`AiOpsService.buildTaskPrompt(...)` already distinguishes two modes:
|
||||
|
||||
- `PAYLOAD_TARGETED`: diagnose the supplied alert payload.
|
||||
- `AUTO_DISCOVERY`: discover active alerts first.
|
||||
|
||||
In payload-targeted mode, the prompt includes alert fields, but it does not provide a normalized retrieval query for `lookup_knowledge`. The Agent may still call the tool, but the exact query is left to model behavior.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Build a deterministic retrieval query from AIOps payload fields.
|
||||
- Preserve the original payload fields in the prompt.
|
||||
- Make the recommended knowledge query visible in prompt text for trace/debugging.
|
||||
- Keep the Agent responsible for deciding when to call `lookup_knowledge`.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not add automatic pre-Agent retrieval.
|
||||
- Do not add verifier logic.
|
||||
- Do not change tool invocation schema.
|
||||
- Do not change L0/L1 retrieval internals.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision 1: Prompt-Level Query Augmentation
|
||||
|
||||
Add a recommended knowledge query to the payload-targeted prompt instead of calling `lookup_knowledge` directly.
|
||||
|
||||
Rationale:
|
||||
|
||||
- The current AIOps flow is Agent-driven; tools remain explicit.
|
||||
- Prompt-level augmentation is low risk and easy to inspect.
|
||||
- It avoids introducing another hidden retrieval path that would complicate trace semantics.
|
||||
|
||||
Alternative considered: automatically call `lookup_knowledge` before invoking the Supervisor. This was rejected because it changes execution behavior and may create evidence that the Agent did not request.
|
||||
|
||||
### Decision 2: Preserve Original Query Terms
|
||||
|
||||
The generated query includes raw alert/service/symptom terms rather than replacing them with broad domains.
|
||||
|
||||
Rationale:
|
||||
|
||||
- Alert name, service name, severity, and symptom are high-value retrieval terms.
|
||||
- Broad categories such as `infrastructure` are useful hints but should not replace concrete terms.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- [Risk] Prompt grows slightly longer. -> Mitigation: keep the query compact and skip blank fields.
|
||||
- [Risk] The model may ignore the recommendation. -> Mitigation: make the instruction explicit and test prompt inclusion.
|
||||
- [Risk] Query construction duplicates some summary fields. -> Mitigation: treat the retrieval query as a compact, tool-oriented view of the payload.
|
||||
@@ -0,0 +1,25 @@
|
||||
## Why
|
||||
|
||||
AIOps payload-targeted diagnosis already scopes the Agent to the supplied alert, but the prompt does not provide a deterministic knowledge-retrieval query. This leaves the Agent to invent lookup terms from the full prompt, which can omit high-value alert fields such as alert name, service, severity, and symptom.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Build a stable knowledge retrieval query from AIOps payload fields.
|
||||
- Include the generated retrieval query in payload-targeted prompts as the recommended `lookup_knowledge` query.
|
||||
- Keep retrieval explicit through the Agent tool; do not automatically call `lookup_knowledge` before the Agent runs.
|
||||
- Add focused tests for query construction and prompt inclusion.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `aiops-traceable-diagnosis-entry`: Payload-targeted AIOps prompts include a deterministic knowledge retrieval query derived from alert payload fields.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects `AiOpsService` prompt construction only.
|
||||
- Does not change the `lookup_knowledge` tool signature, VectorStore retrieval, AIOps API contract, or trace schema.
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: AIOps payload prompts SHALL include a recommended knowledge query
|
||||
When an AIOps request includes alert payload fields, the system SHALL include a deterministic recommended knowledge retrieval query in the prompt sent to the Agent flow.
|
||||
|
||||
#### Scenario: Payload-targeted prompt includes knowledge query
|
||||
- **WHEN** an AIOps request contains alert name, service, severity, or description
|
||||
- **THEN** the generated task prompt SHALL include a recommended `lookup_knowledge` query derived from the supplied payload fields
|
||||
|
||||
#### Scenario: Query skips blank fields
|
||||
- **WHEN** some AIOps payload fields are blank
|
||||
- **THEN** the recommended knowledge query SHALL omit those blank fields
|
||||
- **AND** it SHALL preserve the non-blank alert-specific terms
|
||||
|
||||
#### Scenario: Auto-discovery prompt does not invent payload query
|
||||
- **WHEN** an AIOps request does not include alert payload fields
|
||||
- **THEN** the generated task prompt SHALL remain in auto-discovery mode
|
||||
- **AND** it SHALL not include a payload-derived recommended knowledge query
|
||||
@@ -0,0 +1,14 @@
|
||||
## 1. Prompt Query Construction
|
||||
|
||||
- [x] 1.1 Add a deterministic AIOps knowledge query builder from payload fields.
|
||||
- [x] 1.2 Include the recommended query in payload-targeted task prompts.
|
||||
|
||||
## 2. Tests And Docs
|
||||
|
||||
- [x] 2.1 Add unit tests for query construction and prompt inclusion.
|
||||
- [x] 2.2 Update interview/RAG notes to reflect AIOps payload query augmentation.
|
||||
|
||||
## 3. Verification
|
||||
|
||||
- [x] 3.1 Run focused AIOps service tests.
|
||||
- [x] 3.2 Validate the OpenSpec change and review git scope.
|
||||
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-05
|
||||
@@ -0,0 +1,72 @@
|
||||
## Context
|
||||
|
||||
The indexing path now builds embeddings from structured text:
|
||||
|
||||
```text
|
||||
Title: {title}
|
||||
Path: {breadcrumb}
|
||||
Content:
|
||||
{content}
|
||||
```
|
||||
|
||||
The persisted Milvus `content` field remains the raw chunk content. This improves semantic recall for section-aware questions, but only after documents are reindexed. Existing vectors were generated from the previous content-only input and cannot reflect the new breadcrumb signal.
|
||||
|
||||
The repository already has an offline fixture-based retrieval baseline. That baseline is useful for deterministic regression checks, but it does not prove that the live Milvus/Zilliz collection has been reindexed or that the running service returns breadcrumb-aware results.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Provide an explicit post-reindex live acceptance flow.
|
||||
- Make the reindex prerequisite visible in documentation.
|
||||
- Add a small script that calls the live retrieval endpoint with representative queries and writes reviewable reports.
|
||||
- Keep the live flow optional so unit tests and offline evaluation remain service-free.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not add a new reindex API in this change.
|
||||
- Do not automatically mutate live Milvus/Zilliz data from the acceptance script.
|
||||
- Do not change `lookup_knowledge`, VectorStore retrieval, or Milvus schema.
|
||||
- Do not commit environment-specific live results unless they were intentionally captured for interview evidence.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision 1: Keep Reindex Manual And Explicit
|
||||
|
||||
The acceptance flow documents that reindexing must happen before live validation, but it does not perform the reindex itself.
|
||||
|
||||
Rationale:
|
||||
|
||||
- Reindexing is a data mutation and can be slow or environment-specific.
|
||||
- The existing project already has indexing paths through upload, document management, and knowledge-base initialization.
|
||||
- Keeping mutation separate from validation makes failures easier to diagnose.
|
||||
|
||||
Alternative considered: add a script that triggers reindex and then validates. This was rejected for now because it would need environment-specific credentials, source selection, and safety controls.
|
||||
|
||||
### Decision 2: Use HTTP Endpoint Validation
|
||||
|
||||
The script calls `/api/search/similar` instead of invoking Java services directly.
|
||||
|
||||
Rationale:
|
||||
|
||||
- It validates the same runtime path used in demos.
|
||||
- It works across SDK, Spring AI, and auto retrieval modes.
|
||||
- It produces a simple artifact that can be shown in interview material.
|
||||
|
||||
Alternative considered: add a Java integration test. This was rejected because live Milvus and Spring Boot availability should remain optional.
|
||||
|
||||
### Decision 3: Preserve Offline Baseline Separately
|
||||
|
||||
The existing fixture-based evaluator remains the deterministic baseline. The new live acceptance flow is a smoke/regression companion, not a replacement.
|
||||
|
||||
Rationale:
|
||||
|
||||
- Offline reports are stable and CI-friendly.
|
||||
- Live reports prove environment readiness and post-reindex behavior.
|
||||
- Keeping both avoids mixing deterministic fixture checks with external-service validation.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- [Risk] Live results vary by environment, indexed documents, and retrieval mode. -> Mitigation: report the base URL, query set, result count, top candidates, score labels, and timestamp.
|
||||
- [Risk] A developer may run live validation before reindexing. -> Mitigation: document the prerequisite clearly and include a report note.
|
||||
- [Risk] The script could be mistaken for a benchmark. -> Mitigation: position it as acceptance smoke coverage; keep offline baseline for deterministic metrics.
|
||||
@@ -0,0 +1,26 @@
|
||||
## Why
|
||||
|
||||
`title` and `breadcrumb` now participate in embedding text, but that improvement only affects newly indexed vectors. We need a repeatable acceptance path that tells us how to reindex the knowledge base and verify live retrieval after the embedding input changes.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add a live RAG retrieval acceptance flow for breadcrumb-aware embedding changes.
|
||||
- Document the reindex prerequisite so reviewers understand old vectors do not change automatically.
|
||||
- Provide a small repeatable script for calling live retrieval cases and writing JSON/Markdown reports.
|
||||
- Add interview-facing acceptance notes that explain what was verified and what remains manual or environment-dependent.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
|
||||
None.
|
||||
|
||||
### Modified Capabilities
|
||||
|
||||
- `rag-retrieval-evaluation`: Extend retrieval evaluation with an opt-in live acceptance flow for post-reindex verification.
|
||||
|
||||
## Impact
|
||||
|
||||
- Adds scripts and documentation under the retrieval evaluation/interview areas.
|
||||
- Does not change the Agent runtime path, `lookup_knowledge`, VectorStore search logic, or Milvus schema.
|
||||
- Live verification depends on a running Spring Boot service and a reindexed Milvus/Zilliz collection.
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL provide live post-reindex acceptance
|
||||
The retrieval evaluation system SHALL provide an opt-in live acceptance flow for validating retrieval behavior after embedding input changes require a knowledge-base reindex.
|
||||
|
||||
#### Scenario: Live acceptance requires a running service
|
||||
- **WHEN** live retrieval acceptance is run
|
||||
- **THEN** it SHALL call the configured Spring Boot retrieval endpoint
|
||||
- **AND** it SHALL not be required by the offline fixture baseline
|
||||
|
||||
#### Scenario: Live acceptance records retrieval evidence
|
||||
- **WHEN** a live retrieval case is executed
|
||||
- **THEN** the report SHALL include the query, requested topK, result count, top candidate titles or sources, score labels, and raw response fields needed for review
|
||||
|
||||
#### Scenario: Reindex prerequisite is documented
|
||||
- **WHEN** a developer prepares to validate breadcrumb-aware embedding behavior
|
||||
- **THEN** the repository SHALL explain that existing vectors must be reindexed before live validation can reflect the new embedding text
|
||||
|
||||
#### Scenario: Live report is reviewable
|
||||
- **WHEN** the live acceptance script completes
|
||||
- **THEN** it SHALL write JSON and Markdown outputs that can be inspected or attached to interview evidence
|
||||
@@ -0,0 +1,14 @@
|
||||
## 1. Live Acceptance Tooling
|
||||
|
||||
- [x] 1.1 Add a script that runs representative live `/api/search/similar` queries and writes JSON/Markdown reports.
|
||||
- [x] 1.2 Include breadcrumb-sensitive and core troubleshooting cases in the default live query set.
|
||||
|
||||
## 2. Documentation
|
||||
|
||||
- [x] 2.1 Document the post-reindex validation flow under `eval/rag-retrieval`.
|
||||
- [x] 2.2 Add interview-facing acceptance notes for breadcrumb-aware embedding validation.
|
||||
|
||||
## 3. Verification
|
||||
|
||||
- [x] 3.1 Run targeted tests or syntax checks for the new script.
|
||||
- [x] 3.2 Validate the OpenSpec change and confirm the working tree only contains expected files.
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
schema: spec-driven
|
||||
created: 2026-07-05
|
||||
+72
@@ -0,0 +1,72 @@
|
||||
## Context
|
||||
|
||||
The project currently uses `VectorSearchService` to call Milvus directly through the Java SDK. A previous change added a Spring AI `VectorStore` sidecar and normalized its results, but the production path still uses SDK-only retrieval. Spring AI provides an official Milvus VectorStore starter, so the main path can now move to the framework abstraction without deleting the proven SDK implementation.
|
||||
|
||||
## Goals / Non-Goals
|
||||
|
||||
**Goals:**
|
||||
|
||||
- Use official Spring AI Milvus VectorStore integration.
|
||||
- Keep existing Milvus collection compatibility by configuring field names and embedding dimension.
|
||||
- Preserve current `VectorSearchService` public API.
|
||||
- Support retrieval mode selection:
|
||||
- `spring-ai`: use VectorStore and fail if unavailable.
|
||||
- `sdk`: use existing SDK path.
|
||||
- `auto`: try VectorStore, then fall back to SDK.
|
||||
- Preserve existing score semantics in normalized results by labeling Spring AI scores as `similarity` and SDK scores as `l2_distance`.
|
||||
|
||||
**Non-Goals:**
|
||||
|
||||
- Do not remove Milvus SDK code.
|
||||
- Do not migrate document writes/indexing to Spring AI in this change.
|
||||
- Do not change chunking, metadata shape, or evidence post-processing behavior.
|
||||
- Do not introduce QueryTransformer, MultiQuery, rerank, or neighbor chunk expansion.
|
||||
|
||||
## Decisions
|
||||
|
||||
### Decision: VectorSearchService remains the boundary
|
||||
|
||||
`LookupKnowledgeTool` will keep calling `VectorSearchService.searchSimilarDocuments(...)`.
|
||||
|
||||
Rationale: this protects Agent and AIOps behavior from retrieval implementation churn and keeps the refactor testable.
|
||||
|
||||
### Decision: Auto fallback is the default
|
||||
|
||||
Configure `retrieval.vector-store.mode=auto` so the system prefers Spring AI VectorStore when available but falls back to the existing SDK path on missing beans or runtime errors.
|
||||
|
||||
Rationale: official VectorStore integration may expose schema or scoring differences; fallback keeps the MVP runnable.
|
||||
|
||||
### Decision: Existing collection is reused
|
||||
|
||||
Spring AI Milvus configuration will map to the current collection:
|
||||
|
||||
- id field: `id`
|
||||
- content field: `content`
|
||||
- embedding field: `vector`
|
||||
- metadata field: `metadata`
|
||||
- embedding dimension: `1024`
|
||||
- metric type: `L2`
|
||||
|
||||
Rationale: this avoids reindexing as part of this change and lets golden cases reveal behavior differences first.
|
||||
|
||||
### Decision: SDK indexing remains for now
|
||||
|
||||
`VectorIndexService` continues writing to Milvus using SDK.
|
||||
|
||||
Rationale: replacing both read and write paths at once would make failures harder to isolate. The current change is read-path migration.
|
||||
|
||||
## Risks / Trade-offs
|
||||
|
||||
- Spring AI filter syntax may not map perfectly to Milvus JSON metadata filters -> keep SDK fallback and add tests for filter expression creation.
|
||||
- Spring AI score may be similarity while SDK score is L2 distance -> keep score label explicit.
|
||||
- Auto-configuration could create a VectorStore bean against an incompatible collection -> make retrieval mode configurable and validate with golden cases.
|
||||
- Keeping two paths adds temporary complexity -> isolate SDK and VectorStore code paths inside `VectorSearchService`.
|
||||
|
||||
## Migration Plan
|
||||
|
||||
1. Add Spring AI Milvus starter dependency and configuration.
|
||||
2. Add retrieval mode properties.
|
||||
3. Refactor `VectorSearchService` to prefer VectorStore based on mode.
|
||||
4. Preserve and test SDK fallback.
|
||||
5. Run targeted tests and the offline RAG baseline.
|
||||
6. In a later change, decide whether to migrate indexing/writes after read-path behavior is stable.
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
## Why
|
||||
|
||||
The previous sidecar change proved the project can normalize Spring AI `VectorStore` results without changing the Agent tool boundary. The next step is to integrate the official Spring AI Milvus VectorStore into the main retrieval service while preserving the existing Milvus SDK path as a fallback.
|
||||
|
||||
## What Changes
|
||||
|
||||
- Add the official Spring AI Milvus VectorStore starter dependency.
|
||||
- Configure Spring AI Milvus to reuse the existing collection, field names, embedding dimension, metric type, and connection settings.
|
||||
- Update `VectorSearchService` to support selectable retrieval modes: Spring AI VectorStore, SDK, or automatic fallback.
|
||||
- Preserve the existing `searchSimilarDocuments(query, topK, category)` API used by `lookup_knowledge`.
|
||||
- Keep the current Milvus SDK implementation available and covered by tests.
|
||||
- Add tests proving SDK fallback is used when VectorStore is unavailable or fails.
|
||||
- No breaking API changes.
|
||||
|
||||
## Capabilities
|
||||
|
||||
### New Capabilities
|
||||
- None.
|
||||
|
||||
### Modified Capabilities
|
||||
- `rag-knowledge-retrieval`: Add requirements for using Spring AI VectorStore as the preferred retrieval abstraction while preserving SDK fallback and traceable score semantics.
|
||||
- `rag-retrieval-evaluation`: Add requirements that the offline baseline remains stable after the retrieval implementation changes.
|
||||
|
||||
## Impact
|
||||
|
||||
- Affects `pom.xml`, RAG/Milvus configuration, `VectorSearchService`, and related tests.
|
||||
- Does not change `lookup_knowledge` tool signature, evidence block format, document chunking, upload API, or `tool_invocation` schema.
|
||||
- Uses Spring AI Milvus integration but keeps the existing Milvus SDK code path for rollback and compatibility.
|
||||
+46
@@ -0,0 +1,46 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL prefer Spring AI VectorStore when configured
|
||||
The retrieval service SHALL support Spring AI VectorStore as the preferred vector retrieval abstraction without changing the `lookup_knowledge` tool contract.
|
||||
|
||||
#### Scenario: VectorStore mode uses Spring AI
|
||||
- **WHEN** retrieval vector store mode is configured as `spring-ai`
|
||||
- **THEN** semantic retrieval SHALL query through Spring AI `VectorStore`
|
||||
- **AND** the returned candidates SHALL be normalized into the existing vector search result shape
|
||||
|
||||
#### Scenario: Auto mode prefers VectorStore
|
||||
- **WHEN** retrieval vector store mode is configured as `auto`
|
||||
- **AND** a Spring AI `VectorStore` bean is available
|
||||
- **THEN** semantic retrieval SHALL attempt Spring AI `VectorStore` before the SDK path
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL preserve SDK fallback
|
||||
The retrieval service SHALL keep the existing Milvus SDK retrieval implementation available.
|
||||
|
||||
#### Scenario: SDK mode bypasses VectorStore
|
||||
- **WHEN** retrieval vector store mode is configured as `sdk`
|
||||
- **THEN** semantic retrieval SHALL use the existing Milvus SDK path
|
||||
|
||||
#### Scenario: Auto fallback uses SDK
|
||||
- **WHEN** retrieval vector store mode is `auto`
|
||||
- **AND** Spring AI `VectorStore` is unavailable or fails
|
||||
- **THEN** semantic retrieval SHALL fall back to the existing Milvus SDK path
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL keep score semantics explicit
|
||||
The retrieval service SHALL preserve score semantics when results come from different retrieval implementations.
|
||||
|
||||
#### Scenario: SDK score remains L2 distance
|
||||
- **WHEN** a candidate is returned by the SDK path
|
||||
- **THEN** its score semantics SHALL remain compatible with existing L2 distance normalization
|
||||
|
||||
#### Scenario: VectorStore score is mapped without changing tool contract
|
||||
- **WHEN** a candidate is returned by Spring AI `VectorStore`
|
||||
- **THEN** it SHALL be mapped into the existing result shape
|
||||
- **AND** trace or comparison code SHALL be able to distinguish it as a VectorStore similarity score when needed
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL reuse the existing Milvus collection
|
||||
Spring AI Milvus integration SHALL be configured to use the existing collection schema unless explicitly changed.
|
||||
|
||||
#### Scenario: Existing field mapping
|
||||
- **WHEN** Spring AI Milvus VectorStore is configured
|
||||
- **THEN** it SHALL use the existing id, content, vector, and metadata field names
|
||||
- **AND** it SHALL use the configured embedding dimension and metric type compatible with existing vectors
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
## ADDED Requirements
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL remain stable after VectorStore migration
|
||||
The offline RAG retrieval baseline SHALL remain runnable after the main retrieval service gains Spring AI VectorStore support.
|
||||
|
||||
#### Scenario: Offline evaluator remains service-free
|
||||
- **WHEN** the offline baseline evaluator is run
|
||||
- **THEN** it SHALL not require Spring Boot, live Milvus, Spring AI VectorStore, or the SDK path
|
||||
|
||||
#### Scenario: Baseline is checked during migration
|
||||
- **WHEN** the VectorStore integration change is implemented
|
||||
- **THEN** the existing offline baseline evaluator SHALL be run and its generated report noise SHALL not be committed unless the baseline intentionally changes
|
||||
+26
@@ -0,0 +1,26 @@
|
||||
## 1. Spring AI Milvus Setup
|
||||
|
||||
- [x] 1.1 Add the official Spring AI Milvus VectorStore starter dependency.
|
||||
- [x] 1.2 Configure Spring AI Milvus to reuse the existing collection field names, dimension, metric type, and connection settings.
|
||||
- [x] 1.3 Add retrieval mode configuration for `auto`, `spring-ai`, and `sdk`.
|
||||
|
||||
## 2. Retrieval Service Refactor
|
||||
|
||||
- [x] 2.1 Refactor `VectorSearchService` to inject optional Spring AI `VectorStore`.
|
||||
- [x] 2.2 Implement VectorStore search and map Spring AI `Document` results to existing `SearchResult`.
|
||||
- [x] 2.3 Preserve the existing SDK search path as a dedicated fallback method.
|
||||
- [x] 2.4 Route retrieval by configured mode and fallback rules.
|
||||
|
||||
## 3. Tests
|
||||
|
||||
- [x] 3.1 Add tests for SDK mode bypassing VectorStore.
|
||||
- [x] 3.2 Add tests for auto mode using VectorStore when available.
|
||||
- [x] 3.3 Add tests for auto mode falling back to SDK when VectorStore fails or is unavailable.
|
||||
- [x] 3.4 Add tests for category metadata filter behavior.
|
||||
|
||||
## 4. Validation And Archive
|
||||
|
||||
- [x] 4.1 Run targeted retrieval tests.
|
||||
- [x] 4.2 Run the offline RAG retrieval baseline evaluator.
|
||||
- [x] 4.3 Run OpenSpec validation.
|
||||
- [x] 4.4 Review git diff to confirm `lookup_knowledge` API and evidence output remain compatible.
|
||||
@@ -42,3 +42,20 @@ The system SHALL continue using existing `agent_step` and `tool_invocation` pers
|
||||
#### Scenario: AIOps uses evidence tools
|
||||
- **WHEN** the AIOps flow calls available evidence tools
|
||||
- **THEN** existing hooks and recorders persist agent steps and tool invocations under the resolved AIOps session id
|
||||
|
||||
### Requirement: AIOps payload prompts SHALL include a recommended knowledge query
|
||||
When an AIOps request includes alert payload fields, the system SHALL include a deterministic recommended knowledge retrieval query in the prompt sent to the Agent flow.
|
||||
|
||||
#### Scenario: Payload-targeted prompt includes knowledge query
|
||||
- **WHEN** an AIOps request contains alert name, service, severity, or description
|
||||
- **THEN** the generated task prompt SHALL include a recommended `lookup_knowledge` query derived from the supplied payload fields
|
||||
|
||||
#### Scenario: Query skips blank fields
|
||||
- **WHEN** some AIOps payload fields are blank
|
||||
- **THEN** the recommended knowledge query SHALL omit those blank fields
|
||||
- **AND** it SHALL preserve the non-blank alert-specific terms
|
||||
|
||||
#### Scenario: Auto-discovery prompt does not invent payload query
|
||||
- **WHEN** an AIOps request does not include alert payload fields
|
||||
- **THEN** the generated task prompt SHALL remain in auto-discovery mode
|
||||
- **AND** it SHALL not include a payload-derived recommended knowledge query
|
||||
|
||||
@@ -0,0 +1,153 @@
|
||||
# rag-knowledge-retrieval Specification
|
||||
|
||||
## Purpose
|
||||
Define the runtime contract for the explicit `lookup_knowledge` Agent tool, including how L0 keyword/frontmatter hints cooperate with L1 semantic retrieval while preserving metadata filters, fallback evidence, and traceable retrieval details.
|
||||
## Requirements
|
||||
### Requirement: Knowledge retrieval SHALL keep L0 as a hint provider
|
||||
The `lookup_knowledge` retrieval flow SHALL retain L0 keyword/frontmatter matching but use it as domain, entity, and explainability hint data rather than as the sole final retrieval decision.
|
||||
|
||||
#### Scenario: L0 produces traceable hint data
|
||||
- **WHEN** L0 matches one or more indexed knowledge entries
|
||||
- **THEN** the retrieval flow SHALL expose matched titles, matched keywords, domains or categories, and entity terms as structured hint data
|
||||
|
||||
#### Scenario: L0 does not bypass semantic retrieval by default
|
||||
- **WHEN** L0 returns exactly one match
|
||||
- **THEN** the retrieval flow SHALL still attempt semantic L1 retrieval unless L1 is unavailable or explicitly disabled by configuration
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL use L0 domain as optional L1 filter
|
||||
The retrieval flow SHALL use L0 domain/category information as an optional metadata filter for L1 retrieval when the domain is unambiguous.
|
||||
|
||||
#### Scenario: Single domain filter
|
||||
- **WHEN** L0 hint data contains exactly one nonblank domain or category
|
||||
- **THEN** the L1 retrieval request SHALL include that category as a metadata filter
|
||||
|
||||
#### Scenario: Ambiguous domain fallback
|
||||
- **WHEN** L0 hint data contains zero domains or multiple domains
|
||||
- **THEN** the L1 retrieval request SHALL run without an L0-derived category filter
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL preserve fallback evidence
|
||||
The retrieval flow SHALL still return useful L0 evidence when L1 produces no usable result.
|
||||
|
||||
#### Scenario: L1 has no results
|
||||
- **WHEN** L0 has at least one match and L1 returns no candidates
|
||||
- **THEN** the tool SHALL return an L0-based primary result
|
||||
- **AND** the relevance assessment SHALL not claim semantic support from L1
|
||||
|
||||
#### Scenario: L1 fails
|
||||
- **WHEN** L0 has at least one match and L1 retrieval throws or fails
|
||||
- **THEN** the tool SHALL return an L0-based primary result
|
||||
- **AND** the tool invocation record SHALL preserve the L0 hint details
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL persist L0 hints
|
||||
The system SHALL persist L0 hint details in `tool_invocation.retrieval_details` for `lookup_knowledge` calls.
|
||||
|
||||
#### Scenario: Retrieval details include L0 hints
|
||||
- **WHEN** a `lookup_knowledge` call records a tool invocation
|
||||
- **THEN** `retrieval_details` SHALL include L0 matched keywords, domains, entities, and titles when available
|
||||
|
||||
#### Scenario: Retrieval layer reflects cooperating retrieval
|
||||
- **WHEN** both L0 hint data and L1 candidates participate in a lookup
|
||||
- **THEN** the recorded retrieval layer SHALL be `L0+L1`
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL return structured evidence blocks
|
||||
The `lookup_knowledge` retrieval flow SHALL expose retrieved evidence as structured evidence blocks in addition to the existing compatibility fields.
|
||||
|
||||
#### Scenario: Evidence block contains source metadata
|
||||
- **WHEN** a `lookup_knowledge` call returns evidence
|
||||
- **THEN** each evidence block SHALL include source, title when available, breadcrumb when available, retrieval layer, content, and hit reasons
|
||||
|
||||
#### Scenario: Compatibility fields remain available
|
||||
- **WHEN** evidence blocks are returned
|
||||
- **THEN** the existing `primary` and `supplement` result fields SHALL remain available when their source evidence exists
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL deduplicate evidence blocks
|
||||
The retrieval flow SHALL remove duplicate evidence blocks before returning them to the Agent.
|
||||
|
||||
#### Scenario: Duplicate source deduplication
|
||||
- **WHEN** L0 and L1 produce evidence with the same source identity
|
||||
- **THEN** the retrieval flow SHALL keep a single evidence block for that source
|
||||
- **AND** the evidence block SHALL preserve hit reasons from both retrieval paths when available
|
||||
|
||||
#### Scenario: Postprocess count tracking
|
||||
- **WHEN** evidence post-processing completes
|
||||
- **THEN** the tool invocation details SHALL record candidate count and final evidence block count
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL persist evidence block summaries
|
||||
The system SHALL persist compact evidence block summaries in `tool_invocation.retrieval_details`.
|
||||
|
||||
#### Scenario: Evidence summaries are persisted
|
||||
- **WHEN** a `lookup_knowledge` call records a tool invocation
|
||||
- **THEN** `retrieval_details` SHALL include evidence block summaries containing source, title, retrieval layer, score when available, and hit reasons
|
||||
|
||||
#### Scenario: Full content is not duplicated into retrieval details
|
||||
- **WHEN** evidence block summaries are persisted
|
||||
- **THEN** full evidence content SHALL be omitted or truncated so the trace record remains compact
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL support a disabled-by-default Spring AI sidecar
|
||||
The retrieval system SHALL allow a Spring AI VectorStore retrieval path to be wired as a sidecar without changing the default `lookup_knowledge` runtime path.
|
||||
|
||||
#### Scenario: Sidecar disabled by default
|
||||
- **WHEN** the application starts without explicit sidecar enablement
|
||||
- **THEN** `lookup_knowledge` SHALL continue using the existing retrieval path
|
||||
- **AND** Chat and AIOps runtime behavior SHALL not depend on the sidecar
|
||||
|
||||
#### Scenario: Sidecar failure does not break main retrieval
|
||||
- **WHEN** the Spring AI sidecar is enabled but cannot initialize or query successfully
|
||||
- **THEN** the existing retrieval path SHALL remain usable
|
||||
- **AND** the failure SHALL be reported as sidecar status rather than as a main retrieval failure
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL normalize sidecar results for comparison
|
||||
The sidecar retrieval path SHALL expose results in a comparable structure aligned with the current retrieval result shape.
|
||||
|
||||
#### Scenario: Comparable result metadata
|
||||
- **WHEN** sidecar retrieval returns candidates
|
||||
- **THEN** each comparable result SHALL include source or doc id, title when available, breadcrumb when available, category when available, rank, content preview, and the sidecar score label/value
|
||||
|
||||
#### Scenario: Score semantics are explicit
|
||||
- **WHEN** current retrieval and sidecar retrieval scores are compared
|
||||
- **THEN** the report SHALL label score semantics by path instead of assuming direct numeric equivalence
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL prefer Spring AI VectorStore when configured
|
||||
The retrieval service SHALL support Spring AI VectorStore as the preferred vector retrieval abstraction without changing the `lookup_knowledge` tool contract.
|
||||
|
||||
#### Scenario: VectorStore mode uses Spring AI
|
||||
- **WHEN** retrieval vector store mode is configured as `spring-ai`
|
||||
- **THEN** semantic retrieval SHALL query through Spring AI `VectorStore`
|
||||
- **AND** the returned candidates SHALL be normalized into the existing vector search result shape
|
||||
|
||||
#### Scenario: Auto mode prefers VectorStore
|
||||
- **WHEN** retrieval vector store mode is configured as `auto`
|
||||
- **AND** a Spring AI `VectorStore` bean is available
|
||||
- **THEN** semantic retrieval SHALL attempt Spring AI `VectorStore` before the SDK path
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL preserve SDK fallback
|
||||
The retrieval service SHALL keep the existing Milvus SDK retrieval implementation available.
|
||||
|
||||
#### Scenario: SDK mode bypasses VectorStore
|
||||
- **WHEN** retrieval vector store mode is configured as `sdk`
|
||||
- **THEN** semantic retrieval SHALL use the existing Milvus SDK path
|
||||
|
||||
#### Scenario: Auto fallback uses SDK
|
||||
- **WHEN** retrieval vector store mode is `auto`
|
||||
- **AND** Spring AI `VectorStore` is unavailable or fails
|
||||
- **THEN** semantic retrieval SHALL fall back to the existing Milvus SDK path
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL keep score semantics explicit
|
||||
The retrieval service SHALL preserve score semantics when results come from different retrieval implementations.
|
||||
|
||||
#### Scenario: SDK score remains L2 distance
|
||||
- **WHEN** a candidate is returned by the SDK path
|
||||
- **THEN** its score semantics SHALL remain compatible with existing L2 distance normalization
|
||||
|
||||
#### Scenario: VectorStore score is mapped without changing tool contract
|
||||
- **WHEN** a candidate is returned by Spring AI `VectorStore`
|
||||
- **THEN** it SHALL be mapped into the existing result shape
|
||||
- **AND** trace or comparison code SHALL be able to distinguish it as a VectorStore similarity score when needed
|
||||
|
||||
### Requirement: Knowledge retrieval SHALL reuse the existing Milvus collection
|
||||
Spring AI Milvus integration SHALL be configured to use the existing collection schema unless explicitly changed.
|
||||
|
||||
#### Scenario: Existing field mapping
|
||||
- **WHEN** Spring AI Milvus VectorStore is configured
|
||||
- **THEN** it SHALL use the existing id, content, vector, and metadata field names
|
||||
- **AND** it SHALL use the configured embedding dimension and metric type compatible with existing vectors
|
||||
@@ -0,0 +1,115 @@
|
||||
# rag-retrieval-evaluation Specification
|
||||
|
||||
## Purpose
|
||||
Provide a repeatable offline evaluation baseline for RAG retrieval behavior, so L0, query augmentation, evidence post-processing, and vector store changes can be checked against fixed golden retrieval cases before they affect Agent diagnosis quality.
|
||||
## Requirements
|
||||
### Requirement: Retrieval evaluation SHALL define fixed golden cases
|
||||
The system SHALL provide a fixed set of RAG retrieval golden cases that can be evaluated deterministically.
|
||||
|
||||
#### Scenario: Golden case includes expected retrieval evidence
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL include a case id, query, expected document identifiers or labels, and expected evidence keywords or breadcrumbs
|
||||
|
||||
#### Scenario: Golden case distinguishes scenario type
|
||||
- **WHEN** a retrieval golden case is defined
|
||||
- **THEN** it SHALL indicate whether it covers Chat-style knowledge lookup, AIOps alert diagnosis retrieval, or another explicit retrieval scenario type
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL run offline against fixtures
|
||||
The evaluator SHALL run without requiring live MySQL, Redis, Milvus, LLM, or Spring Boot services.
|
||||
|
||||
#### Scenario: Fixture evaluation
|
||||
- **WHEN** the evaluator is run with a golden case file and retrieval fixture directory
|
||||
- **THEN** it SHALL evaluate each case against its matching fixture file
|
||||
- **AND** it SHALL not call external services
|
||||
|
||||
#### Scenario: Missing fixture is reported
|
||||
- **WHEN** a golden case has no matching retrieval fixture
|
||||
- **THEN** the evaluator SHALL report the case as failed or not run with a clear reason
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL classify hit quality
|
||||
The evaluator SHALL classify each case into a deterministic hit level.
|
||||
|
||||
#### Scenario: Strong hit classification
|
||||
- **WHEN** retrieved candidates include an expected document and satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `strong`
|
||||
|
||||
#### Scenario: Medium hit classification
|
||||
- **WHEN** retrieved candidates include an expected document but do not satisfy expected breadcrumb or evidence keyword coverage
|
||||
- **THEN** the evaluator SHALL classify the case as `medium`
|
||||
|
||||
#### Scenario: Miss classification
|
||||
- **WHEN** retrieved candidates do not include expected documents or expected evidence
|
||||
- **THEN** the evaluator SHALL classify the case as `miss`
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL report ranking signals
|
||||
The evaluator SHALL report ranking and aggregate retrieval signals suitable for future regression checks.
|
||||
|
||||
#### Scenario: Per-case ranking output
|
||||
- **WHEN** a case is evaluated
|
||||
- **THEN** the report SHALL include hit level, first expected document rank when available, top candidate labels, and failed checks
|
||||
|
||||
#### Scenario: Aggregate metrics output
|
||||
- **WHEN** multiple cases are evaluated
|
||||
- **THEN** the report SHALL include case count, strong hit count, medium hit count, miss count, recall at configured K, and average first hit rank when available
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL preserve baseline reports
|
||||
The system SHALL preserve generated baseline reports in JSON and Markdown formats.
|
||||
|
||||
#### Scenario: Baseline report generation
|
||||
- **WHEN** the baseline evaluator is run for the fixed golden case set
|
||||
- **THEN** it SHALL write a JSON report and a Markdown report under the retrieval evaluation documentation area
|
||||
|
||||
#### Scenario: Baseline regeneration is documented
|
||||
- **WHEN** a developer changes golden cases, fixtures, or evaluator logic
|
||||
- **THEN** the repository SHALL explain how to regenerate the retrieval baseline reports
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL compare current and sidecar retrieval paths
|
||||
The retrieval evaluation system SHALL provide an opt-in comparison between the existing retrieval path and the Spring AI sidecar retrieval path.
|
||||
|
||||
#### Scenario: Sidecar comparison report
|
||||
- **WHEN** sidecar comparison is run for the golden case set
|
||||
- **THEN** the report SHALL include per-case current-path top candidates and sidecar top candidates
|
||||
- **AND** it SHALL highlight source, breadcrumb, category, rank, and score-label differences
|
||||
|
||||
#### Scenario: Offline baseline remains unchanged
|
||||
- **WHEN** the fixture-based offline baseline evaluator is run
|
||||
- **THEN** it SHALL not require live Milvus, Spring Boot, or Spring AI sidecar configuration
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL make sidecar readiness visible
|
||||
The sidecar comparison report SHALL show whether the Spring AI sidecar was runnable for the current environment.
|
||||
|
||||
#### Scenario: Sidecar unavailable
|
||||
- **WHEN** sidecar comparison is requested but the sidecar is disabled or unavailable
|
||||
- **THEN** the report SHALL mark sidecar status as unavailable
|
||||
- **AND** it SHALL keep current-path baseline results available for review
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL remain stable after VectorStore migration
|
||||
The offline RAG retrieval baseline SHALL remain runnable after the main retrieval service gains Spring AI VectorStore support.
|
||||
|
||||
#### Scenario: Offline evaluator remains service-free
|
||||
- **WHEN** the offline baseline evaluator is run
|
||||
- **THEN** it SHALL not require Spring Boot, live Milvus, Spring AI VectorStore, or the SDK path
|
||||
|
||||
#### Scenario: Baseline is checked during migration
|
||||
- **WHEN** the VectorStore integration change is implemented
|
||||
- **THEN** the existing offline baseline evaluator SHALL be run and its generated report noise SHALL not be committed unless the baseline intentionally changes
|
||||
|
||||
### Requirement: Retrieval evaluation SHALL provide live post-reindex acceptance
|
||||
The retrieval evaluation system SHALL provide an opt-in live acceptance flow for validating retrieval behavior after embedding input changes require a knowledge-base reindex.
|
||||
|
||||
#### Scenario: Live acceptance requires a running service
|
||||
- **WHEN** live retrieval acceptance is run
|
||||
- **THEN** it SHALL call the configured Spring Boot retrieval endpoint
|
||||
- **AND** it SHALL not be required by the offline fixture baseline
|
||||
|
||||
#### Scenario: Live acceptance records retrieval evidence
|
||||
- **WHEN** a live retrieval case is executed
|
||||
- **THEN** the report SHALL include the query, requested topK, result count, top candidate titles or sources, score labels, and raw response fields needed for review
|
||||
|
||||
#### Scenario: Reindex prerequisite is documented
|
||||
- **WHEN** a developer prepares to validate breadcrumb-aware embedding behavior
|
||||
- **THEN** the repository SHALL explain that existing vectors must be reindexed before live validation can reflect the new embedding text
|
||||
|
||||
#### Scenario: Live report is reviewable
|
||||
- **WHEN** the live acceptance script completes
|
||||
- **THEN** it SHALL write JSON and Markdown outputs that can be inspected or attached to interview evidence
|
||||
@@ -101,6 +101,10 @@
|
||||
<artifactId>milvus-sdk-java</artifactId>
|
||||
<version>2.6.10</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.springframework.ai</groupId>
|
||||
<artifactId>spring-ai-starter-vector-store-milvus</artifactId>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.springframework.boot</groupId>
|
||||
<artifactId>spring-boot-configuration-processor</artifactId>
|
||||
@@ -223,4 +227,4 @@
|
||||
</plugins>
|
||||
</build>
|
||||
|
||||
</project>
|
||||
</project>
|
||||
|
||||
@@ -0,0 +1,292 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Live acceptance runner for post-reindex RAG retrieval checks.
|
||||
|
||||
This script calls the running Spring Boot retrieval endpoint. It is intentionally
|
||||
separate from the offline fixture baseline because it depends on live service and
|
||||
Milvus/Zilliz state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
DEFAULT_BASE_URL = "http://127.0.0.1:9900"
|
||||
DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/live-post-reindex.json")
|
||||
DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/live-post-reindex.md")
|
||||
|
||||
|
||||
DEFAULT_CASES: list[dict[str, Any]] = [
|
||||
{
|
||||
"caseId": "breadcrumb-rag-chunk-context",
|
||||
"query": "If a long RAG section is split into multiple chunks, how do we keep retrieval context?",
|
||||
"topK": 5,
|
||||
"purpose": "Breadcrumb-sensitive RAG chunk context retrieval.",
|
||||
},
|
||||
{
|
||||
"caseId": "breadcrumb-diagnosis-flow",
|
||||
"query": "What is the standard troubleshooting flow for an application incident?",
|
||||
"topK": 5,
|
||||
"purpose": "Process-style retrieval where section path matters.",
|
||||
},
|
||||
{
|
||||
"caseId": "core-err-timeout",
|
||||
"query": "ERR_TIMEOUT",
|
||||
"topK": 3,
|
||||
"purpose": "Exact error-code retrieval should remain stable.",
|
||||
},
|
||||
{
|
||||
"caseId": "core-mysql-connection-pool",
|
||||
"query": "MySQL connection pool is exhausted. How should I diagnose it?",
|
||||
"topK": 3,
|
||||
"purpose": "Core infrastructure troubleshooting retrieval.",
|
||||
},
|
||||
{
|
||||
"caseId": "aiops-payment-latency",
|
||||
"query": "Alert HighLatency on payment-service with p95 latency above threshold",
|
||||
"topK": 3,
|
||||
"purpose": "AIOps alert-style retrieval.",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@dataclass
|
||||
class LiveCase:
|
||||
case_id: str
|
||||
query: str
|
||||
top_k: int
|
||||
purpose: str
|
||||
category: str | None = None
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, raw: dict[str, Any]) -> "LiveCase":
|
||||
return cls(
|
||||
case_id=str(raw["caseId"]),
|
||||
query=str(raw["query"]),
|
||||
top_k=int(raw.get("topK") or 3),
|
||||
purpose=str(raw.get("purpose") or raw.get("notes") or ""),
|
||||
category=(
|
||||
str(raw.get("category"))
|
||||
if raw.get("category") not in (None, "")
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def load_cases(path: Path | None) -> list[LiveCase]:
|
||||
if path is None:
|
||||
return [LiveCase.from_json(item) for item in DEFAULT_CASES]
|
||||
with path.open("r", encoding="utf-8") as handle:
|
||||
payload = json.load(handle)
|
||||
raw_cases = payload.get("cases", payload)
|
||||
return [LiveCase.from_json(item) for item in raw_cases]
|
||||
|
||||
|
||||
def write_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
json.dump(payload, handle, ensure_ascii=False, indent=2)
|
||||
handle.write("\n")
|
||||
|
||||
|
||||
def write_text(path: Path, content: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
handle.write(content)
|
||||
|
||||
|
||||
def request_case(base_url: str, case: LiveCase, timeout_seconds: float) -> dict[str, Any]:
|
||||
endpoint = base_url.rstrip("/") + "/api/search/similar"
|
||||
params: dict[str, str] = {
|
||||
"query": case.query,
|
||||
"topK": str(case.top_k),
|
||||
}
|
||||
if case.category:
|
||||
params["category"] = case.category
|
||||
url = endpoint + "?" + urllib.parse.urlencode(params)
|
||||
|
||||
started_at = datetime.now(timezone.utc)
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=timeout_seconds) as response:
|
||||
body = response.read().decode("utf-8")
|
||||
payload = json.loads(body)
|
||||
status = int(getattr(response, "status", 200))
|
||||
except (urllib.error.URLError, TimeoutError, json.JSONDecodeError) as exc:
|
||||
return {
|
||||
"caseId": case.case_id,
|
||||
"query": case.query,
|
||||
"topK": case.top_k,
|
||||
"category": case.category,
|
||||
"purpose": case.purpose,
|
||||
"url": url,
|
||||
"ok": False,
|
||||
"error": str(exc),
|
||||
"resultCount": 0,
|
||||
"topCandidates": [],
|
||||
"rawResponse": None,
|
||||
"startedAt": started_at.isoformat(),
|
||||
}
|
||||
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if not isinstance(data, list):
|
||||
data = []
|
||||
|
||||
ok = status == 200 and payload.get("code") == 200
|
||||
return {
|
||||
"caseId": case.case_id,
|
||||
"query": case.query,
|
||||
"topK": case.top_k,
|
||||
"category": case.category,
|
||||
"purpose": case.purpose,
|
||||
"url": url,
|
||||
"ok": ok,
|
||||
"httpStatus": status,
|
||||
"responseCode": payload.get("code"),
|
||||
"responseMessage": payload.get("message"),
|
||||
"resultCount": len(data),
|
||||
"topCandidates": [summarize_candidate(item, index + 1) for index, item in enumerate(data)],
|
||||
"rawResponse": payload,
|
||||
"startedAt": started_at.isoformat(),
|
||||
}
|
||||
|
||||
|
||||
def summarize_candidate(raw: dict[str, Any], rank: int) -> dict[str, Any]:
|
||||
metadata = parse_metadata(raw.get("metadata"))
|
||||
return {
|
||||
"rank": rank,
|
||||
"id": raw.get("id"),
|
||||
"title": metadata.get("title"),
|
||||
"breadcrumb": metadata.get("breadcrumb"),
|
||||
"category": metadata.get("category"),
|
||||
"source": metadata.get("_source") or metadata.get("source"),
|
||||
"score": raw.get("score"),
|
||||
"rawScore": raw.get("rawScore"),
|
||||
"scoreLabel": raw.get("scoreLabel"),
|
||||
"contentPreview": preview(raw.get("content")),
|
||||
}
|
||||
|
||||
|
||||
def parse_metadata(value: Any) -> dict[str, Any]:
|
||||
if isinstance(value, dict):
|
||||
return value
|
||||
if isinstance(value, str) and value.strip():
|
||||
try:
|
||||
parsed = json.loads(value)
|
||||
return parsed if isinstance(parsed, dict) else {}
|
||||
except json.JSONDecodeError:
|
||||
return {}
|
||||
return {}
|
||||
|
||||
|
||||
def preview(value: Any, limit: int = 180) -> str:
|
||||
text = " ".join(str(value or "").split())
|
||||
if len(text) <= limit:
|
||||
return text
|
||||
return text[: limit - 3] + "..."
|
||||
|
||||
|
||||
def render_markdown(report: dict[str, Any]) -> str:
|
||||
lines = [
|
||||
"# RAG Live Post-Reindex Acceptance",
|
||||
"",
|
||||
f"Generated at: `{report['generatedAt']}`",
|
||||
f"Base URL: `{report['baseUrl']}`",
|
||||
"",
|
||||
"> Reindex prerequisite: this report only reflects breadcrumb-aware embedding if the knowledge base was reindexed after the embedding-text change.",
|
||||
"",
|
||||
"## Summary",
|
||||
"",
|
||||
"| Metric | Value |",
|
||||
"|---|---:|",
|
||||
f"| Cases | {report['caseCount']} |",
|
||||
f"| Successful calls | {report['successfulCalls']} |",
|
||||
f"| Empty result cases | {report['emptyResultCases']} |",
|
||||
"",
|
||||
"## Cases",
|
||||
"",
|
||||
"| Case | Purpose | Results | Top Candidates |",
|
||||
"|---|---|---:|---|",
|
||||
]
|
||||
for item in report["results"]:
|
||||
top = "<br>".join(format_candidate(candidate) for candidate in item["topCandidates"])
|
||||
if not top and item.get("error"):
|
||||
top = "ERROR: " + str(item["error"])
|
||||
lines.append(
|
||||
"| {case} | {purpose} | {count} | {top} |".format(
|
||||
case=item["caseId"],
|
||||
purpose=item.get("purpose") or "",
|
||||
count=item["resultCount"],
|
||||
top=top,
|
||||
)
|
||||
)
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def format_candidate(candidate: dict[str, Any]) -> str:
|
||||
label = candidate.get("title") or candidate.get("source") or candidate.get("id") or ""
|
||||
breadcrumb = candidate.get("breadcrumb") or ""
|
||||
score_label = candidate.get("scoreLabel") or ""
|
||||
score = candidate.get("score")
|
||||
raw_score = candidate.get("rawScore")
|
||||
details = f"score={score}"
|
||||
if raw_score is not None:
|
||||
details += f", raw={raw_score}"
|
||||
if score_label:
|
||||
details += f", label={score_label}"
|
||||
if breadcrumb:
|
||||
return f"{candidate['rank']}. {label} ({breadcrumb}; {details})"
|
||||
return f"{candidate['rank']}. {label} ({details})"
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--base-url", default=DEFAULT_BASE_URL)
|
||||
parser.add_argument("--cases", type=Path, default=None)
|
||||
parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT)
|
||||
parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT)
|
||||
parser.add_argument("--timeout-seconds", type=float, default=10.0)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
cases = load_cases(args.cases)
|
||||
results = [
|
||||
request_case(args.base_url, case, args.timeout_seconds)
|
||||
for case in cases
|
||||
]
|
||||
successful = [item for item in results if item["ok"]]
|
||||
empty = [item for item in results if item["ok"] and item["resultCount"] == 0]
|
||||
report = {
|
||||
"generatedAt": datetime.now(timezone.utc).isoformat(),
|
||||
"baseUrl": args.base_url,
|
||||
"caseCount": len(results),
|
||||
"successfulCalls": len(successful),
|
||||
"emptyResultCases": len(empty),
|
||||
"reindexPrerequisite": "Run or trigger knowledge-base reindex before treating this as breadcrumb-aware embedding evidence.",
|
||||
"results": results,
|
||||
}
|
||||
write_json(args.json_report, report)
|
||||
write_text(args.markdown_report, render_markdown(report))
|
||||
print(
|
||||
"Ran {total} live cases: successful={successful}, empty={empty}".format(
|
||||
total=len(results),
|
||||
successful=len(successful),
|
||||
empty=len(empty),
|
||||
)
|
||||
)
|
||||
return 1 if len(successful) != len(results) else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,290 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Offline evaluator for RAG retrieval golden cases.
|
||||
|
||||
The evaluator reads fixed golden cases and saved retrieval fixtures. It does not
|
||||
call the running application or any external service.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
DEFAULT_CASES = Path("eval/rag-retrieval/cases/golden-cases.json")
|
||||
DEFAULT_FIXTURES = Path("eval/rag-retrieval/fixtures")
|
||||
DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/baseline.json")
|
||||
DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/baseline.md")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Candidate:
|
||||
rank: int
|
||||
doc_id: str
|
||||
title: str
|
||||
breadcrumb: str
|
||||
content: str
|
||||
score: float | None
|
||||
retrieval_layer: str | None
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, raw: dict[str, Any], fallback_rank: int) -> "Candidate":
|
||||
return cls(
|
||||
rank=int(raw.get("rank") or fallback_rank),
|
||||
doc_id=str(raw.get("docId") or raw.get("id") or ""),
|
||||
title=str(raw.get("title") or ""),
|
||||
breadcrumb=str(raw.get("breadcrumb") or ""),
|
||||
content=str(raw.get("content") or ""),
|
||||
score=_optional_float(raw.get("score")),
|
||||
retrieval_layer=(
|
||||
str(raw.get("retrievalLayer"))
|
||||
if raw.get("retrievalLayer") is not None
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
def searchable_text(self) -> str:
|
||||
return " ".join(
|
||||
[self.doc_id, self.title, self.breadcrumb, self.content]
|
||||
).lower()
|
||||
|
||||
def label(self) -> str:
|
||||
label = self.doc_id or self.title or f"rank-{self.rank}"
|
||||
return f"{self.rank}:{label}"
|
||||
|
||||
|
||||
def _optional_float(value: Any) -> float | None:
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def load_json(path: Path) -> Any:
|
||||
with path.open("r", encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def write_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
json.dump(payload, handle, ensure_ascii=False, indent=2)
|
||||
handle.write("\n")
|
||||
|
||||
|
||||
def write_text(path: Path, content: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
||||
handle.write(content)
|
||||
|
||||
|
||||
def normalize_terms(values: list[Any]) -> list[str]:
|
||||
return [str(value).lower() for value in values if str(value).strip()]
|
||||
|
||||
|
||||
def evaluate_case(case: dict[str, Any], fixture_dir: Path, top_k: int) -> dict[str, Any]:
|
||||
case_id = str(case["caseId"])
|
||||
fixture_path = fixture_dir / f"{case_id}.json"
|
||||
expected_doc_ids = normalize_terms(case.get("expectedDocIds", []))
|
||||
expected_breadcrumbs = normalize_terms(case.get("expectedBreadcrumbs", []))
|
||||
expected_keywords = normalize_terms(case.get("expectedKeywords", []))
|
||||
|
||||
if not fixture_path.exists():
|
||||
return {
|
||||
"caseId": case_id,
|
||||
"scenario": case.get("scenario"),
|
||||
"query": case.get("query"),
|
||||
"hitLevel": "miss",
|
||||
"passed": False,
|
||||
"firstExpectedRank": None,
|
||||
"topCandidates": [],
|
||||
"failedChecks": [f"missing fixture: {fixture_path.as_posix()}"],
|
||||
}
|
||||
|
||||
fixture = load_json(fixture_path)
|
||||
raw_candidates = fixture.get("candidates", [])
|
||||
candidates = [
|
||||
Candidate.from_json(raw, index + 1)
|
||||
for index, raw in enumerate(raw_candidates[:top_k])
|
||||
]
|
||||
|
||||
first_expected = None
|
||||
expected_doc_candidate = None
|
||||
for candidate in candidates:
|
||||
candidate_doc = candidate.doc_id.lower()
|
||||
if any(expected == candidate_doc for expected in expected_doc_ids):
|
||||
first_expected = candidate.rank
|
||||
expected_doc_candidate = candidate
|
||||
break
|
||||
|
||||
breadcrumb_match = False
|
||||
keyword_matches: list[str] = []
|
||||
|
||||
if expected_doc_candidate is not None:
|
||||
breadcrumb_text = expected_doc_candidate.breadcrumb.lower()
|
||||
breadcrumb_match = any(
|
||||
expected in breadcrumb_text or breadcrumb_text in expected
|
||||
for expected in expected_breadcrumbs
|
||||
)
|
||||
|
||||
searchable = expected_doc_candidate.searchable_text()
|
||||
keyword_matches = [
|
||||
keyword for keyword in expected_keywords if keyword in searchable
|
||||
]
|
||||
else:
|
||||
all_text = " ".join(candidate.searchable_text() for candidate in candidates)
|
||||
keyword_matches = [keyword for keyword in expected_keywords if keyword in all_text]
|
||||
|
||||
failed_checks: list[str] = []
|
||||
if expected_doc_candidate is None:
|
||||
failed_checks.append("expected document not found")
|
||||
if expected_doc_candidate is not None and expected_breadcrumbs and not breadcrumb_match:
|
||||
failed_checks.append("expected breadcrumb not found on expected document")
|
||||
if expected_keywords and not keyword_matches:
|
||||
failed_checks.append("expected evidence keywords not found")
|
||||
|
||||
if expected_doc_candidate is not None and (
|
||||
breadcrumb_match or bool(keyword_matches)
|
||||
):
|
||||
hit_level = "strong"
|
||||
elif expected_doc_candidate is not None:
|
||||
hit_level = "medium"
|
||||
elif keyword_matches:
|
||||
hit_level = "weak"
|
||||
else:
|
||||
hit_level = "miss"
|
||||
|
||||
return {
|
||||
"caseId": case_id,
|
||||
"scenario": case.get("scenario"),
|
||||
"query": case.get("query"),
|
||||
"hitLevel": hit_level,
|
||||
"passed": hit_level in {"strong", "medium"},
|
||||
"firstExpectedRank": first_expected,
|
||||
"topCandidates": [candidate.label() for candidate in candidates],
|
||||
"matchedKeywords": keyword_matches,
|
||||
"breadcrumbMatched": breadcrumb_match,
|
||||
"failedChecks": failed_checks,
|
||||
}
|
||||
|
||||
|
||||
def aggregate(results: list[dict[str, Any]], top_k: int) -> dict[str, Any]:
|
||||
total = len(results)
|
||||
counts = {
|
||||
"strong": sum(1 for item in results if item["hitLevel"] == "strong"),
|
||||
"medium": sum(1 for item in results if item["hitLevel"] == "medium"),
|
||||
"weak": sum(1 for item in results if item["hitLevel"] == "weak"),
|
||||
"miss": sum(1 for item in results if item["hitLevel"] == "miss"),
|
||||
}
|
||||
expected_ranks = [
|
||||
item["firstExpectedRank"]
|
||||
for item in results
|
||||
if item.get("firstExpectedRank") is not None
|
||||
]
|
||||
passed = counts["strong"] + counts["medium"]
|
||||
return {
|
||||
"caseCount": total,
|
||||
"topK": top_k,
|
||||
"strongHitCount": counts["strong"],
|
||||
"mediumHitCount": counts["medium"],
|
||||
"weakHitCount": counts["weak"],
|
||||
"missCount": counts["miss"],
|
||||
"recallAtK": round(passed / total, 4) if total else 0,
|
||||
"strongHitRate": round(counts["strong"] / total, 4) if total else 0,
|
||||
"averageFirstHitRank": (
|
||||
round(sum(expected_ranks) / len(expected_ranks), 4)
|
||||
if expected_ranks
|
||||
else None
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def render_markdown(report: dict[str, Any]) -> str:
|
||||
metrics = report["aggregate"]
|
||||
lines = [
|
||||
"# RAG Retrieval Baseline",
|
||||
"",
|
||||
f"Generated at: `{report['generatedAt']}`",
|
||||
"",
|
||||
"## Aggregate",
|
||||
"",
|
||||
"| Metric | Value |",
|
||||
"|---|---:|",
|
||||
f"| Cases | {metrics['caseCount']} |",
|
||||
f"| Top K | {metrics['topK']} |",
|
||||
f"| Recall@K | {metrics['recallAtK']} |",
|
||||
f"| Strong hit rate | {metrics['strongHitRate']} |",
|
||||
f"| Strong hits | {metrics['strongHitCount']} |",
|
||||
f"| Medium hits | {metrics['mediumHitCount']} |",
|
||||
f"| Weak hits | {metrics['weakHitCount']} |",
|
||||
f"| Misses | {metrics['missCount']} |",
|
||||
f"| Average first hit rank | {metrics['averageFirstHitRank']} |",
|
||||
"",
|
||||
"## Cases",
|
||||
"",
|
||||
"| Case | Scenario | Hit | First Expected Rank | Top Candidates | Failed Checks |",
|
||||
"|---|---|---|---:|---|---|",
|
||||
]
|
||||
for item in report["results"]:
|
||||
failed = "<br>".join(item["failedChecks"]) if item["failedChecks"] else ""
|
||||
top = "<br>".join(item["topCandidates"])
|
||||
first_rank = item["firstExpectedRank"]
|
||||
lines.append(
|
||||
"| {case} | {scenario} | {hit} | {rank} | {top} | {failed} |".format(
|
||||
case=item["caseId"],
|
||||
scenario=item.get("scenario") or "",
|
||||
hit=item["hitLevel"],
|
||||
rank=first_rank if first_rank is not None else "",
|
||||
top=top,
|
||||
failed=failed,
|
||||
)
|
||||
)
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--cases", type=Path, default=DEFAULT_CASES)
|
||||
parser.add_argument("--fixtures", type=Path, default=DEFAULT_FIXTURES)
|
||||
parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT)
|
||||
parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
case_file = load_json(args.cases)
|
||||
cases = case_file.get("cases", [])
|
||||
top_k = int(case_file.get("topK") or 5)
|
||||
results = [evaluate_case(case, args.fixtures, top_k) for case in cases]
|
||||
report = {
|
||||
"generatedAt": datetime.now(timezone.utc).isoformat(),
|
||||
"caseFile": args.cases.as_posix(),
|
||||
"fixtureDir": args.fixtures.as_posix(),
|
||||
"aggregate": aggregate(results, top_k),
|
||||
"results": results,
|
||||
}
|
||||
write_json(args.json_report, report)
|
||||
write_text(args.markdown_report, render_markdown(report))
|
||||
|
||||
failed = [item for item in results if item["hitLevel"] == "miss"]
|
||||
print(
|
||||
"Evaluated {total} cases: recall@{top_k}={recall}, misses={misses}".format(
|
||||
total=len(results),
|
||||
top_k=top_k,
|
||||
recall=report["aggregate"]["recallAtK"],
|
||||
misses=len(failed),
|
||||
)
|
||||
)
|
||||
return 1 if failed else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,23 @@
|
||||
package com.superbiz.agent.config;
|
||||
|
||||
import lombok.Getter;
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
|
||||
@Getter
|
||||
@Configuration
|
||||
@ConfigurationProperties(prefix = "rag.sidecar.spring-ai")
|
||||
public class RagSidecarProperties {
|
||||
|
||||
private boolean enabled = false;
|
||||
|
||||
private int contentPreviewLimit = 300;
|
||||
|
||||
public void setEnabled(boolean enabled) {
|
||||
this.enabled = enabled;
|
||||
}
|
||||
|
||||
public void setContentPreviewLimit(int contentPreviewLimit) {
|
||||
this.contentPreviewLimit = contentPreviewLimit;
|
||||
}
|
||||
}
|
||||
@@ -248,7 +248,7 @@ public class ChatController {
|
||||
if (finalReportOptional.isPresent()) {
|
||||
String finalReportText = finalReportOptional.get();
|
||||
logger.info("提取到 Planner 最终报告,长度: {}", finalReportText.length());
|
||||
aiOpsService.persistFinalReport(sessionId, finalReportText);
|
||||
aiOpsService.persistFinalReport(sessionId, finalReportText, request);
|
||||
|
||||
// 发送分隔线
|
||||
emitter.send(SseEmitter.event().name("message")
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
public class ComparableRetrievalResult {
|
||||
|
||||
private String path;
|
||||
|
||||
private Integer rank;
|
||||
|
||||
private String id;
|
||||
|
||||
private String source;
|
||||
|
||||
private String docId;
|
||||
|
||||
private String title;
|
||||
|
||||
private String breadcrumb;
|
||||
|
||||
private String category;
|
||||
|
||||
private String contentPreview;
|
||||
|
||||
private String scoreLabel;
|
||||
|
||||
private Double scoreValue;
|
||||
}
|
||||
@@ -97,6 +97,7 @@ public class DiagnosisTraceResponse {
|
||||
private int persistedToolCallCount;
|
||||
private int returnedToolCallCount;
|
||||
private boolean hasVerifierEvaluation;
|
||||
private boolean hasAiOpsRuleEvaluation;
|
||||
private boolean hasFeedback;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Structured evidence returned by knowledge retrieval.
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
public class EvidenceBlock {
|
||||
|
||||
private String source;
|
||||
|
||||
private String title;
|
||||
|
||||
private String breadcrumb;
|
||||
|
||||
private String retrievalLayer;
|
||||
|
||||
private String content;
|
||||
|
||||
private Double score;
|
||||
|
||||
private List<String> hitReasons;
|
||||
}
|
||||
@@ -27,6 +27,21 @@ public class LookupResult {
|
||||
*/
|
||||
private SupplementResult supplement;
|
||||
|
||||
/**
|
||||
* Structured evidence blocks after retrieval post-processing.
|
||||
*/
|
||||
private List<EvidenceBlock> evidenceBlocks;
|
||||
|
||||
/**
|
||||
* Candidate count before evidence deduplication.
|
||||
*/
|
||||
private Integer evidenceCandidateCount;
|
||||
|
||||
/**
|
||||
* Evidence block count after post-processing.
|
||||
*/
|
||||
private Integer evidenceBlockCount;
|
||||
|
||||
/**
|
||||
* 归一化质量等级:PRECISE / HIGHLY_RELEVANT / REFERENCE
|
||||
*/
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
public class RetrievalComparisonCase {
|
||||
|
||||
private String caseId;
|
||||
|
||||
private String scenario;
|
||||
|
||||
private String query;
|
||||
|
||||
private String category;
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
public class RetrievalComparisonReport {
|
||||
|
||||
private String generatedAt;
|
||||
|
||||
private int caseCount;
|
||||
|
||||
private int topK;
|
||||
|
||||
private String sidecarStatus;
|
||||
|
||||
private List<RetrievalComparisonResult> results;
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
public class RetrievalComparisonResult {
|
||||
|
||||
private String caseId;
|
||||
|
||||
private String scenario;
|
||||
|
||||
private String query;
|
||||
|
||||
private String category;
|
||||
|
||||
private List<ComparableRetrievalResult> currentResults;
|
||||
|
||||
private SidecarRetrievalResponse sidecar;
|
||||
|
||||
private List<String> differences;
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Data
|
||||
@Builder
|
||||
public class SidecarRetrievalResponse {
|
||||
|
||||
private boolean enabled;
|
||||
|
||||
private boolean available;
|
||||
|
||||
private String status;
|
||||
|
||||
private String errorMessage;
|
||||
|
||||
private List<ComparableRetrievalResult> results;
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||
import com.superbiz.agent.dto.AIOpsRequest;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
|
||||
@Service
|
||||
public class AiOpsRuleEvaluationService {
|
||||
|
||||
public static final String PASS = "PASS";
|
||||
public static final String WARN = "WARN";
|
||||
public static final String FAIL = "FAIL";
|
||||
|
||||
public Map<String, Object> evaluate(AIOpsRequest request,
|
||||
String finalReport,
|
||||
List<ToolInvocation> toolInvocations) {
|
||||
List<Map<String, Object>> checks = new ArrayList<>();
|
||||
checks.add(checkReportPresent(finalReport));
|
||||
checks.add(checkPayloadFocus(request, finalReport));
|
||||
checks.add(checkEvidenceCoverage(toolInvocations));
|
||||
|
||||
String verdict = aggregateVerdict(checks);
|
||||
Map<String, Object> evaluation = new LinkedHashMap<>();
|
||||
evaluation.put("verdict", verdict);
|
||||
evaluation.put("checks", checks);
|
||||
evaluation.put("rationale", buildRationale(verdict, checks));
|
||||
evaluation.put("traceability_version", "aiops-rule-v1");
|
||||
return evaluation;
|
||||
}
|
||||
|
||||
private Map<String, Object> checkReportPresent(String finalReport) {
|
||||
boolean passed = finalReport != null && finalReport.trim().length() >= 40;
|
||||
return check(
|
||||
"final_report_present",
|
||||
passed ? PASS : FAIL,
|
||||
passed ? "Final report is present." : "Final report is missing or too short."
|
||||
);
|
||||
}
|
||||
|
||||
private Map<String, Object> checkPayloadFocus(AIOpsRequest request, String finalReport) {
|
||||
if (!hasAlertPayload(request)) {
|
||||
return check("payload_focus", PASS, "No alert payload was supplied; payload focus is not required.");
|
||||
}
|
||||
|
||||
String report = lower(finalReport);
|
||||
List<String> missing = new ArrayList<>();
|
||||
if (!contains(report, request.getAlertName())) {
|
||||
missing.add("alertName");
|
||||
}
|
||||
if (!contains(report, request.getService())) {
|
||||
missing.add("service");
|
||||
}
|
||||
|
||||
if (missing.isEmpty()) {
|
||||
return check("payload_focus", PASS, "Final report mentions the supplied alert and service.");
|
||||
}
|
||||
return check(
|
||||
"payload_focus",
|
||||
WARN,
|
||||
"Final report is missing payload focus terms: " + String.join(", ", missing)
|
||||
);
|
||||
}
|
||||
|
||||
private Map<String, Object> checkEvidenceCoverage(List<ToolInvocation> toolInvocations) {
|
||||
List<String> evidenceTools = safeTools(toolInvocations).stream()
|
||||
.filter(tool -> tool.equals("lookup_knowledge")
|
||||
|| tool.equals("query_metrics")
|
||||
|| tool.equals("query_logs"))
|
||||
.distinct()
|
||||
.toList();
|
||||
|
||||
if (evidenceTools.isEmpty()) {
|
||||
return check("evidence_tool_coverage", WARN, "No persisted AIOps evidence tool calls were found.");
|
||||
}
|
||||
return check(
|
||||
"evidence_tool_coverage",
|
||||
PASS,
|
||||
"Persisted evidence tools: " + String.join(", ", evidenceTools)
|
||||
);
|
||||
}
|
||||
|
||||
private List<String> safeTools(List<ToolInvocation> toolInvocations) {
|
||||
if (toolInvocations == null) {
|
||||
return List.of();
|
||||
}
|
||||
return toolInvocations.stream()
|
||||
.map(ToolInvocation::getToolName)
|
||||
.filter(name -> name != null && !name.isBlank())
|
||||
.map(name -> name.trim().toLowerCase(Locale.ROOT))
|
||||
.toList();
|
||||
}
|
||||
|
||||
private String aggregateVerdict(List<Map<String, Object>> checks) {
|
||||
boolean hasFail = checks.stream().anyMatch(check -> FAIL.equals(check.get("verdict")));
|
||||
if (hasFail) {
|
||||
return FAIL;
|
||||
}
|
||||
boolean hasWarn = checks.stream().anyMatch(check -> WARN.equals(check.get("verdict")));
|
||||
return hasWarn ? WARN : PASS;
|
||||
}
|
||||
|
||||
private String buildRationale(String verdict, List<Map<String, Object>> checks) {
|
||||
long passCount = checks.stream().filter(check -> PASS.equals(check.get("verdict"))).count();
|
||||
long warnCount = checks.stream().filter(check -> WARN.equals(check.get("verdict"))).count();
|
||||
long failCount = checks.stream().filter(check -> FAIL.equals(check.get("verdict"))).count();
|
||||
return "AIOps rule evaluation %s: pass=%d, warn=%d, fail=%d"
|
||||
.formatted(verdict, passCount, warnCount, failCount);
|
||||
}
|
||||
|
||||
private Map<String, Object> check(String name, String verdict, String detail) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("name", name);
|
||||
result.put("verdict", verdict);
|
||||
result.put("detail", detail);
|
||||
return result;
|
||||
}
|
||||
|
||||
private boolean hasAlertPayload(AIOpsRequest request) {
|
||||
if (request == null) {
|
||||
return false;
|
||||
}
|
||||
return !isBlank(request.getAlertName())
|
||||
|| !isBlank(request.getService())
|
||||
|| !isBlank(request.getSeverity())
|
||||
|| !isBlank(request.getDescription())
|
||||
|| !isBlank(request.getTimeRange());
|
||||
}
|
||||
|
||||
private boolean contains(String lowerText, String value) {
|
||||
return isBlank(value) || lowerText.contains(value.trim().toLowerCase(Locale.ROOT));
|
||||
}
|
||||
|
||||
private String lower(String value) {
|
||||
return value == null ? "" : value.toLowerCase(Locale.ROOT);
|
||||
}
|
||||
|
||||
private boolean isBlank(String value) {
|
||||
return value == null || value.trim().isEmpty();
|
||||
}
|
||||
}
|
||||
@@ -27,6 +27,7 @@ import com.superbiz.agent.config.AiOpsPromptProperties;
|
||||
import com.superbiz.agent.tool.LookupKnowledgeTool;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.UUID;
|
||||
|
||||
@@ -66,6 +67,12 @@ public class AiOpsService {
|
||||
@Autowired
|
||||
private ToolInvocationRepository toolInvocationRepository;
|
||||
|
||||
@Autowired
|
||||
private AiOpsRuleEvaluationService aiOpsRuleEvaluationService;
|
||||
|
||||
@Autowired
|
||||
private SelfEvaluationMergeService selfEvaluationMergeService;
|
||||
|
||||
/**
|
||||
* 执行 AI Ops 告警分析流程
|
||||
*
|
||||
@@ -170,11 +177,20 @@ public class AiOpsService {
|
||||
}
|
||||
|
||||
public void persistFinalReport(String sessionId, String finalReport) {
|
||||
persistFinalReport(sessionId, finalReport, null);
|
||||
}
|
||||
|
||||
public void persistFinalReport(String sessionId, String finalReport, AIOpsRequest request) {
|
||||
if (isBlank(sessionId) || isBlank(finalReport)) {
|
||||
return;
|
||||
}
|
||||
diagnosisSessionRepository.findBySessionId(sessionId.trim()).ifPresent(session -> {
|
||||
session.setAnswer(finalReport);
|
||||
List<com.superbiz.agent.domain.entity.ToolInvocation> invocations =
|
||||
toolInvocationRepository.findBySessionIdOrderByIdAsc(session.getSessionId());
|
||||
Map<String, Object> evaluation = aiOpsRuleEvaluationService.evaluate(request, finalReport, invocations);
|
||||
session.setSelfEvaluation(selfEvaluationMergeService.mergeAiOpsRuleEvaluation(
|
||||
session.getSelfEvaluation(), evaluation));
|
||||
diagnosisSessionRepository.save(session);
|
||||
});
|
||||
}
|
||||
@@ -205,15 +221,33 @@ public class AiOpsService {
|
||||
|| !isBlank(request.getTimeRange());
|
||||
}
|
||||
|
||||
String buildKnowledgeRetrievalQuery(AIOpsRequest request) {
|
||||
if (request == null || !hasAlertPayload(request)) {
|
||||
return "";
|
||||
}
|
||||
|
||||
StringBuilder query = new StringBuilder();
|
||||
appendQueryTerm(query, request.getAlertName());
|
||||
appendQueryTerm(query, request.getService());
|
||||
appendQueryTerm(query, request.getSeverity());
|
||||
appendQueryTerm(query, request.getDescription());
|
||||
appendQueryTerm(query, request.getTimeRange());
|
||||
appendQueryTerm(query, request.getUserRequest());
|
||||
return query.toString();
|
||||
}
|
||||
|
||||
String buildTaskPrompt(AIOpsRequest request) {
|
||||
StringBuilder prompt = new StringBuilder();
|
||||
prompt.append("你是企业级 SRE,接到了自动化告警排查任务。请结合工具调用,执行**规划→执行→再规划**的闭环,并最终按照固定模板输出《告警分析报告》。禁止编造虚假数据,如连续多次查询失败需诚实反馈无法完成的原因。");
|
||||
prompt.append("\n\n本次告警输入:\n");
|
||||
prompt.append(buildQuerySummary(request));
|
||||
if (hasAlertPayload(request)) {
|
||||
String knowledgeQuery = buildKnowledgeRetrievalQuery(request);
|
||||
prompt.append("\n\nAIOps scope mode: PAYLOAD_TARGETED\n");
|
||||
prompt.append("- The request includes an alert payload. Treat the supplied alert payload as the primary and only main diagnosis target.\n");
|
||||
prompt.append("- The final report must focus on the supplied alert fields such as alertName, service, severity, description, and timeRange.\n");
|
||||
prompt.append("- Recommended lookup_knowledge query: ").append(knowledgeQuery).append("\n");
|
||||
prompt.append("- If knowledge-base evidence is needed, call lookup_knowledge with the recommended query or a narrower query that preserves alertName and service.\n");
|
||||
prompt.append("- You may call queryPrometheusAlerts only to verify whether the supplied alert is still active or to identify related risk/context.\n");
|
||||
prompt.append("- If queryPrometheusAlerts returns unrelated active alerts, do not create full root-cause or remediation sections for them.\n");
|
||||
prompt.append("- Mention unrelated active alerts only briefly in a Related Risk section when they help explain the supplied alert.\n");
|
||||
@@ -316,6 +350,15 @@ public class AiOpsService {
|
||||
}
|
||||
}
|
||||
|
||||
private void appendQueryTerm(StringBuilder builder, String value) {
|
||||
if (!isBlank(value)) {
|
||||
if (!builder.isEmpty()) {
|
||||
builder.append(' ');
|
||||
}
|
||||
builder.append(value.trim());
|
||||
}
|
||||
}
|
||||
|
||||
private boolean isBlank(String value) {
|
||||
return value == null || value.trim().isEmpty();
|
||||
}
|
||||
|
||||
@@ -115,6 +115,7 @@ public class DiagnosisTraceService {
|
||||
.persistedToolCallCount(defaultInt(session.getToolCallCount()))
|
||||
.returnedToolCallCount(toolInvocations.size())
|
||||
.hasVerifierEvaluation(selfEvaluation != null && selfEvaluation.containsKey("verifier_evaluation"))
|
||||
.hasAiOpsRuleEvaluation(selfEvaluation != null && selfEvaluation.containsKey("aiops_rule_evaluation"))
|
||||
.hasFeedback(session.getFeedback() != null && !session.getFeedback().isBlank())
|
||||
.build();
|
||||
}
|
||||
|
||||
@@ -19,9 +19,11 @@ import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.Paths;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* 知识库索引服务
|
||||
@@ -123,39 +125,73 @@ public class KnowledgeIndexService {
|
||||
}
|
||||
|
||||
public List<KnowledgeEntry> exactMatch(String query) {
|
||||
return analyzeQuery(query).matches();
|
||||
}
|
||||
|
||||
public L0Hint analyzeQuery(String query) {
|
||||
long startTime = System.currentTimeMillis();
|
||||
|
||||
if (query == null || query.trim().isEmpty()) {
|
||||
log.debug("查询关键词为空,返回空结果");
|
||||
return List.of();
|
||||
return L0Hint.empty();
|
||||
}
|
||||
|
||||
String queryLower = query.toLowerCase();
|
||||
List<KnowledgeEntry> results = new ArrayList<>();
|
||||
Set<String> matchedKeywords = new LinkedHashSet<>();
|
||||
Set<String> domains = new LinkedHashSet<>();
|
||||
Set<String> entities = new LinkedHashSet<>();
|
||||
Set<String> titles = new LinkedHashSet<>();
|
||||
|
||||
List<KnowledgeEntry> results = knowledgeIndex.stream()
|
||||
.filter(entry -> matchesKeywords(entry, queryLower))
|
||||
.collect(Collectors.toList());
|
||||
for (KnowledgeEntry entry : knowledgeIndex) {
|
||||
List<String> entryMatchedKeywords = matchedKeywords(entry, queryLower);
|
||||
if (entryMatchedKeywords.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
long elapsedTime = System.currentTimeMillis() - startTime;
|
||||
log.debug("L0精确匹配: query={}, matches={}, indexSize={}, time={}ms",
|
||||
query, results.size(), knowledgeIndex.size(), elapsedTime);
|
||||
results.add(entry);
|
||||
matchedKeywords.addAll(entryMatchedKeywords);
|
||||
entities.addAll(entryMatchedKeywords);
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
private boolean matchesKeywords(KnowledgeEntry entry, String query) {
|
||||
if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
for (String keyword : entry.getKeywords()) {
|
||||
String keywordLower = keyword.toLowerCase();
|
||||
if (query.contains(keywordLower) || keywordLower.contains(query)) {
|
||||
return true;
|
||||
if (entry.getCategory() != null && !entry.getCategory().isBlank()) {
|
||||
domains.add(entry.getCategory());
|
||||
}
|
||||
if (entry.getTitle() != null && !entry.getTitle().isBlank()) {
|
||||
titles.add(entry.getTitle());
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
long elapsedTime = System.currentTimeMillis() - startTime;
|
||||
log.debug("L0 Hint分析: query={}, matches={}, domains={}, keywords={}, indexSize={}, time={}ms",
|
||||
query, results.size(), domains, matchedKeywords, knowledgeIndex.size(), elapsedTime);
|
||||
|
||||
return new L0Hint(
|
||||
List.copyOf(results),
|
||||
List.copyOf(matchedKeywords),
|
||||
List.copyOf(domains),
|
||||
List.copyOf(entities),
|
||||
List.copyOf(titles)
|
||||
);
|
||||
}
|
||||
|
||||
private boolean matchesKeywords(KnowledgeEntry entry, String query) {
|
||||
return !matchedKeywords(entry, query).isEmpty();
|
||||
}
|
||||
|
||||
private List<String> matchedKeywords(KnowledgeEntry entry, String query) {
|
||||
if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
List<String> matches = new ArrayList<>();
|
||||
for (String keyword : entry.getKeywords()) {
|
||||
String keywordLower = keyword.toLowerCase();
|
||||
if (query.contains(keywordLower) || keywordLower.contains(query)) {
|
||||
matches.add(keyword);
|
||||
}
|
||||
}
|
||||
|
||||
return matches;
|
||||
}
|
||||
|
||||
public String readDocument(String filePath, int maxChars) {
|
||||
@@ -224,4 +260,20 @@ public class KnowledgeIndexService {
|
||||
public List<KnowledgeEntry> getAllEntries() {
|
||||
return List.copyOf(knowledgeIndex);
|
||||
}
|
||||
|
||||
public record L0Hint(
|
||||
List<KnowledgeEntry> matches,
|
||||
List<String> matchedKeywords,
|
||||
List<String> domains,
|
||||
List<String> entities,
|
||||
List<String> titles
|
||||
) {
|
||||
public static L0Hint empty() {
|
||||
return new L0Hint(List.of(), List.of(), List.of(), List.of(), List.of());
|
||||
}
|
||||
|
||||
public String singleDomainOrNull() {
|
||||
return domains.size() == 1 ? domains.get(0) : null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,179 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.config.RagSidecarProperties;
|
||||
import com.superbiz.agent.dto.ComparableRetrievalResult;
|
||||
import com.superbiz.agent.dto.RetrievalComparisonCase;
|
||||
import com.superbiz.agent.dto.RetrievalComparisonReport;
|
||||
import com.superbiz.agent.dto.RetrievalComparisonResult;
|
||||
import com.superbiz.agent.dto.SidecarRetrievalResponse;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.time.OffsetDateTime;
|
||||
import java.time.ZoneOffset;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
|
||||
@Service
|
||||
public class RagRetrievalSidecarComparisonService {
|
||||
|
||||
private final VectorSearchService vectorSearchService;
|
||||
private final SpringAiVectorStoreSidecarService sidecarService;
|
||||
private final RetrievalResultNormalizer normalizer;
|
||||
private final RagSidecarProperties properties;
|
||||
private final ObjectMapper objectMapper;
|
||||
|
||||
public RagRetrievalSidecarComparisonService(VectorSearchService vectorSearchService,
|
||||
SpringAiVectorStoreSidecarService sidecarService,
|
||||
RetrievalResultNormalizer normalizer,
|
||||
RagSidecarProperties properties,
|
||||
ObjectMapper objectMapper) {
|
||||
this.vectorSearchService = vectorSearchService;
|
||||
this.sidecarService = sidecarService;
|
||||
this.normalizer = normalizer;
|
||||
this.properties = properties;
|
||||
this.objectMapper = objectMapper;
|
||||
}
|
||||
|
||||
public RetrievalComparisonReport compare(List<RetrievalComparisonCase> cases, int topK) {
|
||||
List<RetrievalComparisonResult> results = new ArrayList<>();
|
||||
String sidecarStatus = "not_run";
|
||||
for (RetrievalComparisonCase comparisonCase : cases) {
|
||||
List<ComparableRetrievalResult> currentResults = normalizeCurrentResults(
|
||||
vectorSearchService.searchSimilarDocuments(
|
||||
comparisonCase.getQuery(),
|
||||
topK,
|
||||
comparisonCase.getCategory()
|
||||
)
|
||||
);
|
||||
SidecarRetrievalResponse sidecar = sidecarService.search(
|
||||
comparisonCase.getQuery(),
|
||||
topK,
|
||||
comparisonCase.getCategory()
|
||||
);
|
||||
sidecarStatus = sidecar.getStatus();
|
||||
results.add(RetrievalComparisonResult.builder()
|
||||
.caseId(comparisonCase.getCaseId())
|
||||
.scenario(comparisonCase.getScenario())
|
||||
.query(comparisonCase.getQuery())
|
||||
.category(comparisonCase.getCategory())
|
||||
.currentResults(currentResults)
|
||||
.sidecar(sidecar)
|
||||
.differences(compareDifferences(currentResults, sidecar.getResults()))
|
||||
.build());
|
||||
}
|
||||
|
||||
return RetrievalComparisonReport.builder()
|
||||
.generatedAt(OffsetDateTime.now(ZoneOffset.UTC).toString())
|
||||
.caseCount(cases.size())
|
||||
.topK(topK)
|
||||
.sidecarStatus(sidecarStatus)
|
||||
.results(results)
|
||||
.build();
|
||||
}
|
||||
|
||||
public RetrievalComparisonReport compareGoldenCases(Path caseFile) throws IOException {
|
||||
var root = objectMapper.readTree(caseFile.toFile());
|
||||
int topK = root.path("topK").asInt(5);
|
||||
List<RetrievalComparisonCase> cases = new ArrayList<>();
|
||||
for (var node : root.path("cases")) {
|
||||
cases.add(RetrievalComparisonCase.builder()
|
||||
.caseId(node.path("caseId").asText())
|
||||
.scenario(node.path("scenario").asText())
|
||||
.query(node.path("query").asText())
|
||||
.build());
|
||||
}
|
||||
return compare(cases, topK);
|
||||
}
|
||||
|
||||
public void writeReports(RetrievalComparisonReport report, Path jsonPath, Path markdownPath) throws IOException {
|
||||
createParentDirectories(jsonPath);
|
||||
createParentDirectories(markdownPath);
|
||||
objectMapper.writerWithDefaultPrettyPrinter().writeValue(jsonPath.toFile(), report);
|
||||
Files.writeString(markdownPath, renderMarkdown(report));
|
||||
}
|
||||
|
||||
private void createParentDirectories(Path path) throws IOException {
|
||||
Path parent = path.getParent();
|
||||
if (parent != null) {
|
||||
Files.createDirectories(parent);
|
||||
}
|
||||
}
|
||||
|
||||
private List<ComparableRetrievalResult> normalizeCurrentResults(List<VectorSearchService.SearchResult> rawResults) {
|
||||
List<ComparableRetrievalResult> results = new ArrayList<>();
|
||||
for (int i = 0; i < rawResults.size(); i++) {
|
||||
results.add(normalizer.fromCurrent(rawResults.get(i), i + 1, properties.getContentPreviewLimit()));
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
private List<String> compareDifferences(List<ComparableRetrievalResult> currentResults,
|
||||
List<ComparableRetrievalResult> sidecarResults) {
|
||||
if (sidecarResults == null || sidecarResults.isEmpty()) {
|
||||
return List.of("sidecar_unavailable_or_empty");
|
||||
}
|
||||
List<String> differences = new ArrayList<>();
|
||||
String currentTopSource = currentResults.isEmpty() ? null : currentResults.get(0).getSource();
|
||||
String sidecarTopSource = sidecarResults.get(0).getSource();
|
||||
if (!Objects.equals(currentTopSource, sidecarTopSource)) {
|
||||
differences.add("top_source_differs");
|
||||
}
|
||||
String currentTopBreadcrumb = currentResults.isEmpty() ? null : currentResults.get(0).getBreadcrumb();
|
||||
String sidecarTopBreadcrumb = sidecarResults.get(0).getBreadcrumb();
|
||||
if (!Objects.equals(currentTopBreadcrumb, sidecarTopBreadcrumb)) {
|
||||
differences.add("top_breadcrumb_differs");
|
||||
}
|
||||
String currentScoreLabel = currentResults.isEmpty() ? null : currentResults.get(0).getScoreLabel();
|
||||
String sidecarScoreLabel = sidecarResults.get(0).getScoreLabel();
|
||||
if (!Objects.equals(currentScoreLabel, sidecarScoreLabel)) {
|
||||
differences.add("score_label_differs");
|
||||
}
|
||||
return differences;
|
||||
}
|
||||
|
||||
private String renderMarkdown(RetrievalComparisonReport report) {
|
||||
StringBuilder builder = new StringBuilder();
|
||||
builder.append("# RAG Sidecar Retrieval Comparison\n\n");
|
||||
builder.append("Generated at: `").append(report.getGeneratedAt()).append("`\n\n");
|
||||
builder.append("- Cases: ").append(report.getCaseCount()).append("\n");
|
||||
builder.append("- Top K: ").append(report.getTopK()).append("\n");
|
||||
builder.append("- Sidecar status: `").append(report.getSidecarStatus()).append("`\n\n");
|
||||
builder.append("| Case | Query | Current Top | Sidecar Top | Differences |\n");
|
||||
builder.append("|---|---|---|---|---|\n");
|
||||
for (RetrievalComparisonResult result : report.getResults()) {
|
||||
builder.append("| ")
|
||||
.append(nullToBlank(result.getCaseId()))
|
||||
.append(" | ")
|
||||
.append(escapePipe(result.getQuery()))
|
||||
.append(" | ")
|
||||
.append(formatTop(result.getCurrentResults()))
|
||||
.append(" | ")
|
||||
.append(formatTop(result.getSidecar() != null ? result.getSidecar().getResults() : List.of()))
|
||||
.append(" | ")
|
||||
.append(String.join("<br>", result.getDifferences()))
|
||||
.append(" |\n");
|
||||
}
|
||||
return builder.toString();
|
||||
}
|
||||
|
||||
private String formatTop(List<ComparableRetrievalResult> results) {
|
||||
if (results == null || results.isEmpty()) {
|
||||
return "";
|
||||
}
|
||||
ComparableRetrievalResult top = results.get(0);
|
||||
return escapePipe(nullToBlank(top.getSource())) + " (" + nullToBlank(top.getScoreLabel()) + ")";
|
||||
}
|
||||
|
||||
private String escapePipe(String value) {
|
||||
return nullToBlank(value).replace("|", "\\|");
|
||||
}
|
||||
|
||||
private String nullToBlank(String value) {
|
||||
return value == null ? "" : value;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.dto.ComparableRetrievalResult;
|
||||
import org.springframework.ai.document.Document;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
|
||||
@Component
|
||||
public class RetrievalResultNormalizer {
|
||||
|
||||
private final ObjectMapper objectMapper;
|
||||
|
||||
public RetrievalResultNormalizer(ObjectMapper objectMapper) {
|
||||
this.objectMapper = objectMapper;
|
||||
}
|
||||
|
||||
public ComparableRetrievalResult fromCurrent(VectorSearchService.SearchResult result, int rank, int previewLimit) {
|
||||
Map<String, String> metadata = parseMetadata(result.getMetadata());
|
||||
String source = firstNonBlank(metadata.get("_source"), metadata.get("source"), result.getMetadata(), result.getId());
|
||||
return ComparableRetrievalResult.builder()
|
||||
.path("current")
|
||||
.rank(rank)
|
||||
.id(result.getId())
|
||||
.source(source)
|
||||
.docId(metadata.get("docId"))
|
||||
.title(metadata.get("title"))
|
||||
.breadcrumb(metadata.get("breadcrumb"))
|
||||
.category(metadata.get("category"))
|
||||
.contentPreview(truncate(result.getContent(), previewLimit))
|
||||
.scoreLabel("l2_distance")
|
||||
.scoreValue((double) result.getScore())
|
||||
.build();
|
||||
}
|
||||
|
||||
public ComparableRetrievalResult fromSidecar(Document document, int rank, int previewLimit) {
|
||||
Map<String, String> metadata = stringifyMetadata(document.getMetadata());
|
||||
String source = firstNonBlank(metadata.get("_source"), metadata.get("source"), metadata.get("docId"), document.getId());
|
||||
return ComparableRetrievalResult.builder()
|
||||
.path("sidecar")
|
||||
.rank(rank)
|
||||
.id(document.getId())
|
||||
.source(source)
|
||||
.docId(metadata.get("docId"))
|
||||
.title(metadata.get("title"))
|
||||
.breadcrumb(metadata.get("breadcrumb"))
|
||||
.category(metadata.get("category"))
|
||||
.contentPreview(truncate(document.getText(), previewLimit))
|
||||
.scoreLabel("similarity")
|
||||
.scoreValue(document.getScore())
|
||||
.build();
|
||||
}
|
||||
|
||||
private Map<String, String> parseMetadata(String metadata) {
|
||||
if (metadata == null || metadata.isBlank()) {
|
||||
return Map.of();
|
||||
}
|
||||
try {
|
||||
Map<?, ?> raw = objectMapper.readValue(metadata, Map.class);
|
||||
return stringifyMetadata(raw);
|
||||
} catch (Exception e) {
|
||||
return Map.of();
|
||||
}
|
||||
}
|
||||
|
||||
private Map<String, String> stringifyMetadata(Map<?, ?> raw) {
|
||||
if (raw == null || raw.isEmpty()) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> result = new LinkedHashMap<>();
|
||||
for (Map.Entry<?, ?> entry : raw.entrySet()) {
|
||||
if (entry.getKey() != null && entry.getValue() != null) {
|
||||
result.put(String.valueOf(entry.getKey()), String.valueOf(entry.getValue()));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
private String firstNonBlank(String... values) {
|
||||
for (String value : values) {
|
||||
if (value != null && !value.isBlank()) {
|
||||
return value;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private String truncate(String text, int maxLength) {
|
||||
if (text == null || text.length() <= maxLength) {
|
||||
return text;
|
||||
}
|
||||
return text.substring(0, maxLength) + "...";
|
||||
}
|
||||
}
|
||||
@@ -27,6 +27,10 @@ public class SelfEvaluationMergeService {
|
||||
return merge(existingJson, "verifier_evaluation", verifierEvaluation);
|
||||
}
|
||||
|
||||
public String mergeAiOpsRuleEvaluation(String existingJson, Map<String, Object> aiOpsRuleEvaluation) {
|
||||
return merge(existingJson, "aiops_rule_evaluation", aiOpsRuleEvaluation);
|
||||
}
|
||||
|
||||
private String merge(String existingJson, String key, Map<String, Object> value) {
|
||||
try {
|
||||
Map<String, Object> root = parseRoot(existingJson);
|
||||
@@ -44,7 +48,9 @@ public class SelfEvaluationMergeService {
|
||||
}
|
||||
|
||||
Map<String, Object> parsed = objectMapper.readValue(existingJson, MAP_TYPE);
|
||||
if (parsed.containsKey("rule_evaluation") || parsed.containsKey("verifier_evaluation")) {
|
||||
if (parsed.containsKey("rule_evaluation")
|
||||
|| parsed.containsKey("verifier_evaluation")
|
||||
|| parsed.containsKey("aiops_rule_evaluation")) {
|
||||
return new LinkedHashMap<>(parsed);
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.config.RagSidecarProperties;
|
||||
import com.superbiz.agent.dto.ComparableRetrievalResult;
|
||||
import com.superbiz.agent.dto.SidecarRetrievalResponse;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.ai.document.Document;
|
||||
import org.springframework.ai.vectorstore.SearchRequest;
|
||||
import org.springframework.ai.vectorstore.VectorStore;
|
||||
import org.springframework.beans.factory.ObjectProvider;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
@Slf4j
|
||||
@Service
|
||||
public class SpringAiVectorStoreSidecarService {
|
||||
|
||||
private final RagSidecarProperties properties;
|
||||
private final ObjectProvider<VectorStore> vectorStoreProvider;
|
||||
private final RetrievalResultNormalizer normalizer;
|
||||
|
||||
public SpringAiVectorStoreSidecarService(RagSidecarProperties properties,
|
||||
ObjectProvider<VectorStore> vectorStoreProvider,
|
||||
RetrievalResultNormalizer normalizer) {
|
||||
this.properties = properties;
|
||||
this.vectorStoreProvider = vectorStoreProvider;
|
||||
this.normalizer = normalizer;
|
||||
}
|
||||
|
||||
public SidecarRetrievalResponse search(String query, int topK, String category) {
|
||||
if (!properties.isEnabled()) {
|
||||
return unavailable("disabled", null);
|
||||
}
|
||||
|
||||
VectorStore vectorStore = vectorStoreProvider.getIfAvailable();
|
||||
if (vectorStore == null) {
|
||||
return unavailable("missing_vector_store", "No Spring AI VectorStore bean is available");
|
||||
}
|
||||
|
||||
try {
|
||||
SearchRequest.Builder builder = SearchRequest.builder()
|
||||
.query(query)
|
||||
.topK(topK)
|
||||
.similarityThresholdAll();
|
||||
if (category != null && !category.isBlank()) {
|
||||
builder.filterExpression("category == '" + escapeFilterValue(category) + "'");
|
||||
}
|
||||
|
||||
List<Document> documents = vectorStore.similaritySearch(builder.build());
|
||||
List<ComparableRetrievalResult> results = new ArrayList<>();
|
||||
for (int i = 0; i < documents.size(); i++) {
|
||||
results.add(normalizer.fromSidecar(documents.get(i), i + 1, properties.getContentPreviewLimit()));
|
||||
}
|
||||
return SidecarRetrievalResponse.builder()
|
||||
.enabled(true)
|
||||
.available(true)
|
||||
.status("available")
|
||||
.results(results)
|
||||
.build();
|
||||
} catch (Exception e) {
|
||||
log.warn("Spring AI sidecar retrieval failed: {}", e.getMessage());
|
||||
return unavailable("query_failed", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private SidecarRetrievalResponse unavailable(String status, String errorMessage) {
|
||||
return SidecarRetrievalResponse.builder()
|
||||
.enabled(properties.isEnabled())
|
||||
.available(false)
|
||||
.status(status)
|
||||
.errorMessage(errorMessage)
|
||||
.results(List.of())
|
||||
.build();
|
||||
}
|
||||
|
||||
private String escapeFilterValue(String value) {
|
||||
return value.replace("'", "\\'");
|
||||
}
|
||||
}
|
||||
@@ -3,10 +3,10 @@ package com.superbiz.agent.service;
|
||||
import com.fasterxml.jackson.core.JsonProcessingException;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||
import com.superbiz.agent.dto.EvidenceBlock;
|
||||
import com.superbiz.agent.dto.LookupResult;
|
||||
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||||
import com.superbiz.agent.dto.KnowledgeEntry;
|
||||
import com.superbiz.agent.service.VectorSearchService;
|
||||
import com.superbiz.agent.util.SessionContextHolder;
|
||||
import lombok.Builder;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
@@ -108,6 +108,15 @@ public class ToolInvocationRecorder {
|
||||
if (record.l0Titles() != null && !record.l0Titles().isEmpty()) {
|
||||
details.put("l0_titles", record.l0Titles());
|
||||
}
|
||||
if (record.l0MatchedKeywords() != null && !record.l0MatchedKeywords().isEmpty()) {
|
||||
details.put("l0_matched_keywords", record.l0MatchedKeywords());
|
||||
}
|
||||
if (record.l0Domains() != null && !record.l0Domains().isEmpty()) {
|
||||
details.put("l0_domains", record.l0Domains());
|
||||
}
|
||||
if (record.l0Entities() != null && !record.l0Entities().isEmpty()) {
|
||||
details.put("l0_entities", record.l0Entities());
|
||||
}
|
||||
if (record.l1TopScore() != null) {
|
||||
details.put("l1_top_score", record.l1TopScore());
|
||||
}
|
||||
@@ -120,6 +129,15 @@ public class ToolInvocationRecorder {
|
||||
if (record.l1Scores() != null && !record.l1Scores().isEmpty()) {
|
||||
details.put("l1_scores", record.l1Scores());
|
||||
}
|
||||
if (record.evidenceCandidateCount() != null) {
|
||||
details.put("evidence_candidate_count", record.evidenceCandidateCount());
|
||||
}
|
||||
if (record.evidenceBlockCount() != null) {
|
||||
details.put("evidence_block_count", record.evidenceBlockCount());
|
||||
}
|
||||
if (record.evidenceBlocks() != null && !record.evidenceBlocks().isEmpty()) {
|
||||
details.put("evidence_blocks", record.evidenceBlocks());
|
||||
}
|
||||
if (record.relevanceLevel() != null) {
|
||||
details.put("relevance_level", record.relevanceLevel());
|
||||
}
|
||||
@@ -196,12 +214,18 @@ public class ToolInvocationRecorder {
|
||||
String evidenceStatus,
|
||||
String errorMessage,
|
||||
List<String> l0Titles,
|
||||
List<String> l0MatchedKeywords,
|
||||
List<String> l0Domains,
|
||||
List<String> l0Entities,
|
||||
Double l1TopScore,
|
||||
Double l1TopSimilarity,
|
||||
List<Double> l1Scores
|
||||
List<Double> l1Scores,
|
||||
Integer evidenceCandidateCount,
|
||||
Integer evidenceBlockCount,
|
||||
List<Map<String, Object>> evidenceBlocks
|
||||
) {
|
||||
public static LookupKnowledgeRecord from(String query,
|
||||
List<KnowledgeEntry> l0Matches,
|
||||
KnowledgeIndexService.L0Hint l0Hint,
|
||||
List<VectorSearchService.SearchResult> l1Results,
|
||||
boolean highConfidence,
|
||||
LookupResult result,
|
||||
@@ -209,6 +233,7 @@ public class ToolInvocationRecorder {
|
||||
String dedupReason,
|
||||
int durationMs,
|
||||
double l1TopSimilarity) {
|
||||
List<KnowledgeEntry> l0Matches = l0Hint != null ? l0Hint.matches() : List.of();
|
||||
boolean hasL0 = l0Matches != null && !l0Matches.isEmpty();
|
||||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||||
String layer;
|
||||
@@ -272,10 +297,39 @@ public class ToolInvocationRecorder {
|
||||
.success(true)
|
||||
.evidenceStatus(evidenceStatus)
|
||||
.l0Titles(l0Titles)
|
||||
.l0MatchedKeywords(l0Hint != null ? l0Hint.matchedKeywords() : List.of())
|
||||
.l0Domains(l0Hint != null ? l0Hint.domains() : List.of())
|
||||
.l0Entities(l0Hint != null ? l0Hint.entities() : List.of())
|
||||
.l1TopScore(hasL1 ? (double) l1Results.get(0).getScore() : null)
|
||||
.l1TopSimilarity(hasL1 ? l1TopSimilarity : null)
|
||||
.l1Scores(l1Scores)
|
||||
.evidenceCandidateCount(result != null ? result.getEvidenceCandidateCount() : null)
|
||||
.evidenceBlockCount(result != null ? result.getEvidenceBlockCount() : null)
|
||||
.evidenceBlocks(result != null ? summarizeEvidenceBlocks(result.getEvidenceBlocks()) : List.of())
|
||||
.build();
|
||||
}
|
||||
|
||||
private static List<Map<String, Object>> summarizeEvidenceBlocks(List<EvidenceBlock> blocks) {
|
||||
if (blocks == null || blocks.isEmpty()) {
|
||||
return List.of();
|
||||
}
|
||||
List<Map<String, Object>> summaries = new ArrayList<>();
|
||||
for (int i = 0; i < Math.min(5, blocks.size()); i++) {
|
||||
EvidenceBlock block = blocks.get(i);
|
||||
Map<String, Object> summary = new LinkedHashMap<>();
|
||||
summary.put("source", block.getSource());
|
||||
summary.put("title", block.getTitle());
|
||||
summary.put("breadcrumb", block.getBreadcrumb());
|
||||
summary.put("retrieval_layer", block.getRetrievalLayer());
|
||||
summary.put("score", block.getScore());
|
||||
summary.put("hit_reasons", block.getHitReasons());
|
||||
String content = block.getContent();
|
||||
if (content != null) {
|
||||
summary.put("content_preview", content.length() <= 180 ? content : content.substring(0, 180) + "...");
|
||||
}
|
||||
summaries.add(summary);
|
||||
}
|
||||
return summaries;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -148,7 +148,7 @@ public class VectorIndexService {
|
||||
|
||||
try {
|
||||
// 生成向量
|
||||
List<Float> vector = embeddingService.generateEmbedding(chunk.getContent());
|
||||
List<Float> vector = embeddingService.generateEmbedding(buildEmbeddingText(chunk));
|
||||
|
||||
// 构建元数据(包含文件信息)
|
||||
Map<String, Object> metadata = buildMetadata(path.toString(), chunk, chunks.size());
|
||||
@@ -191,7 +191,7 @@ public class VectorIndexService {
|
||||
|
||||
try {
|
||||
// 生成向量
|
||||
List<Float> vector = embeddingService.generateEmbedding(chunk.getContent());
|
||||
List<Float> vector = embeddingService.generateEmbedding(buildEmbeddingText(chunk));
|
||||
|
||||
// 构建元数据(使用 docId 和 category)
|
||||
Map<String, Object> metadata = buildDocumentMetadata(docId, chunk, chunks.size(), category);
|
||||
@@ -280,6 +280,30 @@ public class VectorIndexService {
|
||||
return metadata;
|
||||
}
|
||||
|
||||
static String buildEmbeddingText(DocumentChunk chunk) {
|
||||
String content = trimToEmpty(chunk.getContent());
|
||||
String title = trimToEmpty(chunk.getTitle());
|
||||
String breadcrumb = trimToEmpty(chunk.getBreadcrumb());
|
||||
|
||||
if (title.isEmpty() && breadcrumb.isEmpty()) {
|
||||
return content;
|
||||
}
|
||||
|
||||
StringBuilder text = new StringBuilder();
|
||||
if (!title.isEmpty()) {
|
||||
text.append("Title: ").append(title).append("\n");
|
||||
}
|
||||
if (!breadcrumb.isEmpty()) {
|
||||
text.append("Path: ").append(breadcrumb).append("\n");
|
||||
}
|
||||
text.append("Content:\n").append(content);
|
||||
return text.toString();
|
||||
}
|
||||
|
||||
private static String trimToEmpty(String value) {
|
||||
return value == null ? "" : value.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* 删除文件的旧数据(根据 metadata._source)
|
||||
*/
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.core.JsonProcessingException;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.constant.MilvusConstants;
|
||||
import io.milvus.client.MilvusServiceClient;
|
||||
import io.milvus.grpc.SearchResults;
|
||||
import io.milvus.param.R;
|
||||
@@ -7,19 +10,26 @@ import io.milvus.param.dml.SearchParam;
|
||||
import io.milvus.response.SearchResultsWrapper;
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
import com.superbiz.agent.constant.MilvusConstants;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
import org.springframework.ai.document.Document;
|
||||
import org.springframework.ai.vectorstore.SearchRequest;
|
||||
import org.springframework.ai.vectorstore.VectorStore;
|
||||
import org.springframework.beans.factory.ObjectProvider;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* 向量搜索服务
|
||||
* 负责从 Milvus 中搜索相似向量
|
||||
* Vector retrieval facade used by lookup_knowledge.
|
||||
*
|
||||
* <p>The public API stays stable while the implementation can route to Spring AI
|
||||
* VectorStore, the original Milvus SDK path, or automatic fallback.</p>
|
||||
*/
|
||||
@Service
|
||||
public class VectorSearchService {
|
||||
@@ -32,34 +42,84 @@ public class VectorSearchService {
|
||||
@Autowired
|
||||
private VectorEmbeddingService embeddingService;
|
||||
|
||||
/**
|
||||
* 搜索相似文档
|
||||
*
|
||||
* @param query 查询文本
|
||||
* @param topK 返回最相似的K个结果
|
||||
* @return 搜索结果列表
|
||||
*/
|
||||
@Autowired
|
||||
private ObjectProvider<VectorStore> vectorStoreProvider;
|
||||
|
||||
@Autowired
|
||||
private ObjectMapper objectMapper;
|
||||
|
||||
@Value("${retrieval.vector-store.mode:auto}")
|
||||
private String vectorStoreMode = "auto";
|
||||
|
||||
@Value("${retrieval.normalization.max-l2-distance:2.0}")
|
||||
private double maxL2Distance = 2.0;
|
||||
|
||||
public List<SearchResult> searchSimilarDocuments(String query, int topK) {
|
||||
return searchSimilarDocuments(query, topK, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* 搜索相似文档(支持类别过滤)
|
||||
*
|
||||
* @param query 查询文本
|
||||
* @param topK 返回最相似的K个结果
|
||||
* @param category 类别过滤(可选,null 表示不过滤)
|
||||
* @return 搜索结果列表
|
||||
*/
|
||||
public List<SearchResult> searchSimilarDocuments(String query, int topK, String category) {
|
||||
String mode = vectorStoreMode == null ? "auto" : vectorStoreMode.trim().toLowerCase();
|
||||
return switch (mode) {
|
||||
case "sdk" -> searchSimilarDocumentsWithSdk(query, topK, category);
|
||||
case "spring-ai" -> searchSimilarDocumentsWithVectorStore(query, topK, category);
|
||||
case "auto" -> searchWithAutoFallback(query, topK, category);
|
||||
default -> {
|
||||
logger.warn("Unknown retrieval.vector-store.mode={}, using auto mode", vectorStoreMode);
|
||||
yield searchWithAutoFallback(query, topK, category);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
private List<SearchResult> searchWithAutoFallback(String query, int topK, String category) {
|
||||
try {
|
||||
logger.info("开始搜索相似文档, 查询: {}, topK: {}, 类别: {}", query, topK, category);
|
||||
return searchSimilarDocumentsWithVectorStore(query, topK, category);
|
||||
} catch (Exception e) {
|
||||
logger.warn("Spring AI VectorStore retrieval failed, falling back to Milvus SDK: {}", e.getMessage());
|
||||
return searchSimilarDocumentsWithSdk(query, topK, category);
|
||||
}
|
||||
}
|
||||
|
||||
List<SearchResult> searchSimilarDocumentsWithVectorStore(String query, int topK, String category) {
|
||||
VectorStore vectorStore = vectorStoreProvider != null ? vectorStoreProvider.getIfAvailable() : null;
|
||||
if (vectorStore == null) {
|
||||
throw new IllegalStateException("Spring AI VectorStore bean is unavailable");
|
||||
}
|
||||
|
||||
logger.info("Starting Spring AI VectorStore search: query={}, topK={}, category={}", query, topK, category);
|
||||
SearchRequest.Builder builder = SearchRequest.builder()
|
||||
.query(query)
|
||||
.topK(topK)
|
||||
.similarityThresholdAll();
|
||||
if (category != null && !category.trim().isEmpty()) {
|
||||
String filterExpression = "category == '" + escapeFilterValue(category.trim()) + "'";
|
||||
builder.filterExpression(filterExpression);
|
||||
logger.info("Spring AI VectorStore category filter: {}", filterExpression);
|
||||
}
|
||||
|
||||
List<Document> documents = vectorStore.similaritySearch(builder.build());
|
||||
List<SearchResult> results = new ArrayList<>();
|
||||
for (Document document : documents) {
|
||||
SearchResult result = new SearchResult();
|
||||
result.setId(document.getId());
|
||||
result.setContent(document.getText());
|
||||
result.setMetadata(toJson(document.getMetadata()));
|
||||
result.setRawScore(document.getScore());
|
||||
result.setScoreLabel("similarity");
|
||||
result.setScore(toCompatibleL2Distance(document));
|
||||
results.add(result);
|
||||
}
|
||||
logger.info("Spring AI VectorStore search complete, candidates={}", results.size());
|
||||
return results;
|
||||
}
|
||||
|
||||
List<SearchResult> searchSimilarDocumentsWithSdk(String query, int topK, String category) {
|
||||
try {
|
||||
logger.info("Starting Milvus SDK search: query={}, topK={}, category={}", query, topK, category);
|
||||
|
||||
// 1. 将查询文本向量化
|
||||
List<Float> queryVector = embeddingService.generateQueryVector(query);
|
||||
logger.debug("查询向量生成成功, 维度: {}", queryVector.size());
|
||||
logger.debug("Query vector generated, dimension={}", queryVector.size());
|
||||
|
||||
// 2. 构建搜索参数
|
||||
SearchParam.Builder searchParamBuilder = SearchParam.newBuilder()
|
||||
.withCollectionName(MilvusConstants.MILVUS_COLLECTION_NAME)
|
||||
.withVectorFieldName("vector")
|
||||
@@ -69,33 +129,27 @@ public class VectorSearchService {
|
||||
.withOutFields(List.of("id", "content", "metadata"))
|
||||
.withParams("{\"nprobe\":10}");
|
||||
|
||||
// 添加类别过滤
|
||||
if (category != null && !category.trim().isEmpty()) {
|
||||
String expr = String.format("metadata[\"category\"] == \"%s\"", category);
|
||||
searchParamBuilder.withExpr(expr);
|
||||
logger.info("添加类别过滤: {}", expr);
|
||||
logger.info("Milvus SDK category filter: {}", expr);
|
||||
}
|
||||
|
||||
SearchParam searchParam = searchParamBuilder.build();
|
||||
|
||||
// 3. 执行搜索
|
||||
R<SearchResults> searchResponse = milvusClient.search(searchParam);
|
||||
|
||||
R<SearchResults> searchResponse = milvusClient.search(searchParamBuilder.build());
|
||||
if (searchResponse.getStatus() != 0) {
|
||||
throw new RuntimeException("向量搜索失败: " + searchResponse.getMessage());
|
||||
throw new RuntimeException("Vector search failed: " + searchResponse.getMessage());
|
||||
}
|
||||
|
||||
// 4. 解析搜索结果
|
||||
SearchResultsWrapper wrapper = new SearchResultsWrapper(searchResponse.getData().getResults());
|
||||
List<SearchResult> results = new ArrayList<>();
|
||||
|
||||
for (int i = 0; i < wrapper.getRowRecords(0).size(); i++) {
|
||||
SearchResult result = new SearchResult();
|
||||
result.setId((String) wrapper.getIDScore(0).get(i).get("id"));
|
||||
result.setContent((String) wrapper.getFieldData("content", 0).get(i));
|
||||
result.setScore(wrapper.getIDScore(0).get(i).getScore());
|
||||
result.setRawScore((double) result.getScore());
|
||||
result.setScoreLabel("l2_distance");
|
||||
|
||||
// 解析 metadata
|
||||
Object metadataObj = wrapper.getFieldData("metadata", 0).get(i);
|
||||
if (metadataObj != null) {
|
||||
result.setMetadata(metadataObj.toString());
|
||||
@@ -104,25 +158,76 @@ public class VectorSearchService {
|
||||
results.add(result);
|
||||
}
|
||||
|
||||
logger.info("搜索完成, 找到 {} 个相似文档", results.size());
|
||||
logger.info("Milvus SDK search complete, candidates={}", results.size());
|
||||
return results;
|
||||
|
||||
} catch (Exception e) {
|
||||
logger.error("搜索相似文档失败", e);
|
||||
throw new RuntimeException("搜索失败: " + e.getMessage(), e);
|
||||
logger.error("Milvus SDK vector search failed", e);
|
||||
throw new RuntimeException("Vector search failed: " + e.getMessage(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 搜索结果类
|
||||
*/
|
||||
private float toCompatibleL2Distance(Document document) {
|
||||
Double distance = extractDistance(document.getMetadata());
|
||||
if (distance != null) {
|
||||
return distance.floatValue();
|
||||
}
|
||||
return toCompatibleL2Distance(document.getScore());
|
||||
}
|
||||
|
||||
private float toCompatibleL2Distance(Double similarity) {
|
||||
if (similarity == null) {
|
||||
return (float) maxL2Distance;
|
||||
}
|
||||
double bounded = Math.max(0.0, Math.min(1.0, similarity));
|
||||
return (float) ((1.0 - bounded) * maxL2Distance);
|
||||
}
|
||||
|
||||
private Double extractDistance(Map<String, Object> metadata) {
|
||||
if (metadata == null) {
|
||||
return null;
|
||||
}
|
||||
Object value = metadata.get("distance");
|
||||
if (value instanceof Number number) {
|
||||
return number.doubleValue();
|
||||
}
|
||||
if (value instanceof String text) {
|
||||
try {
|
||||
return Double.parseDouble(text);
|
||||
} catch (NumberFormatException ignored) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private String toJson(Map<String, Object> metadata) {
|
||||
if (metadata == null || metadata.isEmpty()) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
return objectMapper.writeValueAsString(metadata);
|
||||
} catch (JsonProcessingException e) {
|
||||
return metadata.toString();
|
||||
}
|
||||
}
|
||||
|
||||
private String escapeFilterValue(String value) {
|
||||
return value.replace("'", "\\'");
|
||||
}
|
||||
|
||||
@Setter
|
||||
@Getter
|
||||
public static class SearchResult {
|
||||
private String id;
|
||||
private String content;
|
||||
/**
|
||||
* Compatibility score used by existing lookup relevance normalization.
|
||||
* SDK mode keeps L2 distance; VectorStore mode prefers the Milvus
|
||||
* distance metadata and falls back to similarity mapping.
|
||||
*/
|
||||
private float score;
|
||||
private Double rawScore;
|
||||
private String scoreLabel;
|
||||
private String metadata;
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,8 +13,11 @@ import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -35,13 +38,13 @@ public class LookupKnowledgeTool {
|
||||
private static final String HINT_REFERENCE = "当前结果为相关参考,如需更精准信息请明确缺少的具体维度";
|
||||
|
||||
@Value("${retrieval.normalization.max-l2-distance:2.0}")
|
||||
private double maxL2Distance;
|
||||
private double maxL2Distance = 2.0;
|
||||
|
||||
@Value("${retrieval.normalization.highly-relevant-threshold:0.75}")
|
||||
private double highlyRelevantThreshold;
|
||||
private double highlyRelevantThreshold = 0.75;
|
||||
|
||||
@Value("${retrieval.normalization.reference-threshold:0.5}")
|
||||
private double referenceThreshold;
|
||||
private double referenceThreshold = 0.5;
|
||||
|
||||
@Autowired
|
||||
private KnowledgeIndexService knowledgeIndexService;
|
||||
@@ -83,30 +86,28 @@ public class LookupKnowledgeTool {
|
||||
log.info(">>> RequestId: {}", requestId);
|
||||
log.info("----------------------------------------");
|
||||
|
||||
// Step 1: L0 精确匹配
|
||||
// Step 1: L0 hint 分析
|
||||
long l0Start = System.currentTimeMillis();
|
||||
List<KnowledgeEntry> l0Matches = knowledgeIndexService.exactMatch(query);
|
||||
KnowledgeIndexService.L0Hint l0Hint = knowledgeIndexService.analyzeQuery(query);
|
||||
List<KnowledgeEntry> l0Matches = l0Hint.matches();
|
||||
long l0Time = System.currentTimeMillis() - l0Start;
|
||||
log.info("[L0 精确匹配] 完成: matches={}, time={}ms", l0Matches.size(), l0Time);
|
||||
log.info("[L0 Hint] 完成: matches={}, domains={}, keywords={}, time={}ms",
|
||||
l0Matches.size(), l0Hint.domains(), l0Hint.matchedKeywords(), l0Time);
|
||||
if (!l0Matches.isEmpty()) {
|
||||
log.info("[L0 精确匹配] 找到文档:");
|
||||
log.info("[L0 Hint] 找到文档:");
|
||||
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
|
||||
KnowledgeEntry entry = l0Matches.get(i);
|
||||
log.info(" - [{}] 标题: {}, 路径: {}, 域: {}", i+1, entry.getTitle(), entry.getFilePath(), entry.getCategory());
|
||||
}
|
||||
}
|
||||
|
||||
// Step 2: 判断是否高置信度(唯一匹配)
|
||||
boolean highConfidence = (l0Matches.size() == 1);
|
||||
log.info("[置信度判断] highConfidence={}, reason={}",
|
||||
highConfidence, highConfidence ? "唯一匹配" : "多个或零个匹配");
|
||||
|
||||
// Step 3: L1 条件调用
|
||||
List<VectorSearchService.SearchResult> l1Results = null;
|
||||
if (!highConfidence) {
|
||||
log.info("[L1 语义检索] L0非唯一匹配,触发L1语义检索...");
|
||||
// Step 2: L1 默认调用;L0 只提供可解释 hint 和可选 category filter
|
||||
List<VectorSearchService.SearchResult> l1Results = List.of();
|
||||
String l0CategoryFilter = l0Hint.singleDomainOrNull();
|
||||
try {
|
||||
log.info("[L1 语义检索] 触发L1语义检索, categoryFilter={}", l0CategoryFilter);
|
||||
long l1Start = System.currentTimeMillis();
|
||||
l1Results = vectorSearchService.searchSimilarDocuments(query, 3, null);
|
||||
l1Results = vectorSearchService.searchSimilarDocuments(query, 3, l0CategoryFilter);
|
||||
long l1Time = System.currentTimeMillis() - l1Start;
|
||||
log.info("[L1 语义检索] 完成: matches={}, time={}ms",
|
||||
l1Results != null ? l1Results.size() : 0, l1Time);
|
||||
@@ -117,25 +118,27 @@ public class LookupKnowledgeTool {
|
||||
log.info(" - [{}] 文档ID: {}, L2距离: {}", i+1, result.getId(), String.format("%.4f", result.getScore()));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
log.info("[L1 语义检索] L0唯一匹配,跳过L1检索");
|
||||
} catch (Exception e) {
|
||||
log.warn("[L1 语义检索] 调用失败,保留L0 fallback: {}", e.getMessage());
|
||||
l1Results = List.of();
|
||||
}
|
||||
|
||||
// Step 4: 归一化质量等级判定
|
||||
// Step 3: 归一化质量等级判定
|
||||
float l1TopScore = (l1Results != null && !l1Results.isEmpty()) ? l1Results.get(0).getScore() : Float.MAX_VALUE;
|
||||
RelevanceAssessment assessment = computeRelevance(l0Matches.size(), l1TopScore);
|
||||
boolean highConfidence = isHighConfidence(l0Matches.size(), l1TopScore);
|
||||
log.info("[归一化] relevanceLevel={}, completenessHint={}", assessment.level, assessment.hint);
|
||||
if (l1TopScore != Float.MAX_VALUE) {
|
||||
double similarity = normalizeL2(l1TopScore);
|
||||
log.info("[归一化] L2距离={}, similarity={}", String.format("%.4f", l1TopScore), String.format("%.4f", similarity));
|
||||
}
|
||||
|
||||
// Step 5: 组装结果
|
||||
// Step 4: 组装结果
|
||||
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
|
||||
result.setRelevanceLevel(assessment.level);
|
||||
result.setCompletenessHint(assessment.hint);
|
||||
|
||||
// Step 6: session 级去重过滤 + 域级行动记忆
|
||||
// Step 5: session 级去重过滤 + 域级行动记忆
|
||||
String sessionId = SessionContextHolder.getSessionId();
|
||||
String domain = extractDomain(l0Matches, l1Results);
|
||||
|
||||
@@ -144,7 +147,7 @@ public class LookupKnowledgeTool {
|
||||
if (docKey != null && retrievedDocTracker.isAlreadyRetrieved(sessionId, docKey)) {
|
||||
log.info("[去重] 文档已在本会话中检索过,跳过: {}", docKey);
|
||||
List<String> retrievedDomains = retrievedDocTracker.getRetrievedDomains(sessionId);
|
||||
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result, domain, "doc_retrieved");
|
||||
saveToolInvocation(query, l0Hint, l1Results, highConfidence, startTime, result, domain, "doc_retrieved");
|
||||
return LookupResult.builder()
|
||||
.found(false)
|
||||
.message("文档已在本会话中检索过,无需重复召回: " + docKey)
|
||||
@@ -197,7 +200,7 @@ public class LookupKnowledgeTool {
|
||||
log.info("========================================");
|
||||
|
||||
// 记录 tool_invocation
|
||||
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result, domain, null);
|
||||
saveToolInvocation(query, l0Hint, l1Results, highConfidence, startTime, result, domain, null);
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -225,11 +228,16 @@ public class LookupKnowledgeTool {
|
||||
RelevanceAssessment computeRelevance(int l0MatchCount, float l1TopScore) {
|
||||
double l1Similarity = (l1TopScore != Float.MAX_VALUE) ? normalizeL2(l1TopScore) : 0.0;
|
||||
|
||||
// L0 唯一匹配 → PRECISE
|
||||
if (l0MatchCount == 1) {
|
||||
// L0 唯一匹配 + L1 高分 → PRECISE
|
||||
if (l0MatchCount == 1 && l1Similarity >= highlyRelevantThreshold) {
|
||||
return new RelevanceAssessment(LEVEL_PRECISE, HINT_PRECISE);
|
||||
}
|
||||
|
||||
// L0 唯一匹配但缺少 L1 支持 → REFERENCE
|
||||
if (l0MatchCount == 1) {
|
||||
return new RelevanceAssessment(LEVEL_REFERENCE, HINT_REFERENCE);
|
||||
}
|
||||
|
||||
// L0 命中 + L1 高分 → HIGHLY_RELEVANT
|
||||
if (l0MatchCount > 1 && l1Similarity >= highlyRelevantThreshold) {
|
||||
return new RelevanceAssessment(LEVEL_HIGHLY_RELEVANT, HINT_HIGHLY_RELEVANT);
|
||||
@@ -259,6 +267,16 @@ public class LookupKnowledgeTool {
|
||||
return new RelevanceAssessment(null, null);
|
||||
}
|
||||
|
||||
boolean isHighConfidence(int l0MatchCount, float l1TopScore) {
|
||||
if (l0MatchCount != 1) {
|
||||
return false;
|
||||
}
|
||||
if (l1TopScore == Float.MAX_VALUE) {
|
||||
return true;
|
||||
}
|
||||
return normalizeL2(l1TopScore) >= highlyRelevantThreshold;
|
||||
}
|
||||
|
||||
/**
|
||||
* 归一化评估结果
|
||||
*/
|
||||
@@ -302,7 +320,7 @@ public class LookupKnowledgeTool {
|
||||
/**
|
||||
* 保存工具调用明细到 tool_invocation 表
|
||||
*/
|
||||
private void saveToolInvocation(String query, List<KnowledgeEntry> l0Matches,
|
||||
private void saveToolInvocation(String query, KnowledgeIndexService.L0Hint l0Hint,
|
||||
List<VectorSearchService.SearchResult> l1Results,
|
||||
boolean highConfidence, long startTime,
|
||||
LookupResult result, String domain, String dedupReason) {
|
||||
@@ -317,7 +335,7 @@ public class LookupKnowledgeTool {
|
||||
|
||||
ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.from(
|
||||
query,
|
||||
l0Matches,
|
||||
l0Hint,
|
||||
l1Results,
|
||||
highConfidence,
|
||||
result,
|
||||
@@ -382,12 +400,154 @@ public class LookupKnowledgeTool {
|
||||
}
|
||||
builder.supplement(supplement);
|
||||
|
||||
EvidencePostprocessResult evidence = buildEvidenceBlocks(l0Matches, l1Results);
|
||||
builder.evidenceBlocks(evidence.blocks());
|
||||
builder.evidenceCandidateCount(evidence.candidateCount());
|
||||
builder.evidenceBlockCount(evidence.blocks().size());
|
||||
|
||||
boolean found = (primary != null) || (supplement != null);
|
||||
builder.found(found);
|
||||
|
||||
return builder.build();
|
||||
}
|
||||
|
||||
private EvidencePostprocessResult buildEvidenceBlocks(
|
||||
List<KnowledgeEntry> l0Matches,
|
||||
List<VectorSearchService.SearchResult> l1Results) {
|
||||
Map<String, EvidenceBlock> deduped = new LinkedHashMap<>();
|
||||
int candidateCount = 0;
|
||||
|
||||
if (l0Matches != null) {
|
||||
for (int i = 0; i < l0Matches.size(); i++) {
|
||||
KnowledgeEntry entry = l0Matches.get(i);
|
||||
candidateCount++;
|
||||
EvidenceBlock block = EvidenceBlock.builder()
|
||||
.source(entry.getFilePath())
|
||||
.title(entry.getTitle())
|
||||
.breadcrumb(null)
|
||||
.retrievalLayer("L0")
|
||||
.content(buildMetadataOnlySummary(entry))
|
||||
.score(null)
|
||||
.hitReasons(buildL0HitReasons(entry, i + 1))
|
||||
.build();
|
||||
mergeEvidence(deduped, sourceKey(block, "l0-" + i), block);
|
||||
}
|
||||
}
|
||||
|
||||
if (l1Results != null) {
|
||||
for (int i = 0; i < l1Results.size(); i++) {
|
||||
VectorSearchService.SearchResult result = l1Results.get(i);
|
||||
candidateCount++;
|
||||
Map<String, String> metadata = parseMetadata(result.getMetadata());
|
||||
String source = firstNonBlank(
|
||||
metadata.get("_source"),
|
||||
metadata.get("docId"),
|
||||
result.getMetadata(),
|
||||
result.getId()
|
||||
);
|
||||
EvidenceBlock block = EvidenceBlock.builder()
|
||||
.source(source)
|
||||
.title(metadata.get("title"))
|
||||
.breadcrumb(metadata.get("breadcrumb"))
|
||||
.retrievalLayer("L1")
|
||||
.content(truncate(result.getContent(), 800))
|
||||
.score((double) result.getScore())
|
||||
.hitReasons(List.of("semantic_rank:" + (i + 1)))
|
||||
.build();
|
||||
mergeEvidence(deduped, sourceKey(block, "l1-" + i), block);
|
||||
}
|
||||
}
|
||||
|
||||
return new EvidencePostprocessResult(candidateCount, new ArrayList<>(deduped.values()));
|
||||
}
|
||||
|
||||
private void mergeEvidence(Map<String, EvidenceBlock> deduped, String key, EvidenceBlock incoming) {
|
||||
EvidenceBlock existing = deduped.get(key);
|
||||
if (existing == null) {
|
||||
deduped.put(key, incoming);
|
||||
return;
|
||||
}
|
||||
|
||||
List<String> mergedReasons = new ArrayList<>();
|
||||
if (existing.getHitReasons() != null) {
|
||||
mergedReasons.addAll(existing.getHitReasons());
|
||||
}
|
||||
if (incoming.getHitReasons() != null) {
|
||||
for (String reason : incoming.getHitReasons()) {
|
||||
if (!mergedReasons.contains(reason)) {
|
||||
mergedReasons.add(reason);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
String mergedLayer = existing.getRetrievalLayer();
|
||||
if (incoming.getRetrievalLayer() != null && !incoming.getRetrievalLayer().equals(mergedLayer)) {
|
||||
mergedLayer = "L0+L1";
|
||||
}
|
||||
|
||||
existing.setRetrievalLayer(mergedLayer);
|
||||
existing.setHitReasons(mergedReasons);
|
||||
if (existing.getScore() == null && incoming.getScore() != null) {
|
||||
existing.setScore(incoming.getScore());
|
||||
}
|
||||
if ((existing.getBreadcrumb() == null || existing.getBreadcrumb().isBlank())
|
||||
&& incoming.getBreadcrumb() != null) {
|
||||
existing.setBreadcrumb(incoming.getBreadcrumb());
|
||||
}
|
||||
}
|
||||
|
||||
private List<String> buildL0HitReasons(KnowledgeEntry entry, int rank) {
|
||||
List<String> reasons = new ArrayList<>();
|
||||
reasons.add("l0_rank:" + rank);
|
||||
if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) {
|
||||
reasons.add("l0_keywords:" + String.join(",", entry.getKeywords()));
|
||||
}
|
||||
if (entry.getCategory() != null && !entry.getCategory().isBlank()) {
|
||||
reasons.add("domain:" + entry.getCategory());
|
||||
}
|
||||
return reasons;
|
||||
}
|
||||
|
||||
private String sourceKey(EvidenceBlock block, String fallback) {
|
||||
return firstNonBlank(block.getSource(), block.getTitle(), block.getBreadcrumb(), fallback);
|
||||
}
|
||||
|
||||
private Map<String, String> parseMetadata(String metadata) {
|
||||
if (metadata == null || metadata.isBlank()) {
|
||||
return Map.of();
|
||||
}
|
||||
try {
|
||||
Map<?, ?> raw = objectMapper.readValue(metadata, Map.class);
|
||||
Map<String, String> result = new LinkedHashMap<>();
|
||||
for (Map.Entry<?, ?> entry : raw.entrySet()) {
|
||||
if (entry.getKey() != null && entry.getValue() != null) {
|
||||
result.put(String.valueOf(entry.getKey()), String.valueOf(entry.getValue()));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
} catch (Exception e) {
|
||||
return Map.of();
|
||||
}
|
||||
}
|
||||
|
||||
private String firstNonBlank(String... values) {
|
||||
for (String value : values) {
|
||||
if (value != null && !value.isBlank()) {
|
||||
return value;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private String truncate(String text, int maxLength) {
|
||||
if (text == null || text.length() <= maxLength) {
|
||||
return text;
|
||||
}
|
||||
return text.substring(0, maxLength) + "...";
|
||||
}
|
||||
|
||||
private record EvidencePostprocessResult(int candidateCount, List<EvidenceBlock> blocks) {}
|
||||
|
||||
private int countMdHeadings(String content) {
|
||||
if (content == null) return 0;
|
||||
return (int) content.lines()
|
||||
|
||||
@@ -93,6 +93,30 @@ spring:
|
||||
min-idle: 0
|
||||
|
||||
ai:
|
||||
vectorstore:
|
||||
type: milvus
|
||||
milvus:
|
||||
initialize-schema: false
|
||||
database-name: ${milvus.database}
|
||||
collection-name: biz
|
||||
embedding-dimension: ${milvus.vector-dim}
|
||||
index-type: IVF_FLAT
|
||||
metric-type: L2
|
||||
index-parameters: '{"nlist":128}'
|
||||
id-field-name: id
|
||||
auto-id: false
|
||||
content-field-name: content
|
||||
metadata-field-name: metadata
|
||||
embedding-field-name: vector
|
||||
client:
|
||||
host: ${milvus.host}
|
||||
port: ${milvus.port}
|
||||
token: ${milvus.token}
|
||||
username: ${milvus.username}
|
||||
password: ${milvus.password}
|
||||
secure: ${milvus.secure}
|
||||
connect-timeout-ms: ${milvus.timeout}
|
||||
|
||||
# --- Chat: DeepSeek (原生) ---
|
||||
deepseek:
|
||||
api-key: sk-1f44696abe644bd684f09cc43f12c557
|
||||
@@ -126,9 +150,15 @@ document:
|
||||
# RAG 配置
|
||||
rag:
|
||||
top-k: 3 # 检索返回的最相似文档数量
|
||||
sidecar:
|
||||
spring-ai:
|
||||
enabled: false
|
||||
content-preview-limit: 300
|
||||
|
||||
# 检索归一化配置
|
||||
retrieval:
|
||||
vector-store:
|
||||
mode: auto # auto | spring-ai | sdk
|
||||
normalization:
|
||||
max-l2-distance: 2.0 # L2 距离上界(BGE-M3 单位向量 = 2.0)
|
||||
highly-relevant-threshold: 0.75 # similarity >= 0.75 → HIGHLY_RELEVANT
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||
import com.superbiz.agent.dto.AIOpsRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
class AiOpsRuleEvaluationServiceTest {
|
||||
|
||||
private final AiOpsRuleEvaluationService service = new AiOpsRuleEvaluationService();
|
||||
|
||||
@Test
|
||||
void evaluatePassesWhenReportFocusesPayloadAndHasEvidenceTools() {
|
||||
AIOpsRequest request = new AIOpsRequest();
|
||||
request.setAlertName("HighCPUUsage");
|
||||
request.setService("payment-service");
|
||||
|
||||
ToolInvocation invocation = ToolInvocation.builder()
|
||||
.toolName("lookup_knowledge")
|
||||
.build();
|
||||
|
||||
Map<String, Object> evaluation = service.evaluate(
|
||||
request,
|
||||
"HighCPUUsage alert on payment-service was diagnosed using metrics and knowledge evidence.",
|
||||
List.of(invocation)
|
||||
);
|
||||
|
||||
assertEquals("PASS", evaluation.get("verdict"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void evaluateWarnsWhenEvidenceToolsAreMissing() {
|
||||
AIOpsRequest request = new AIOpsRequest();
|
||||
request.setAlertName("HighCPUUsage");
|
||||
request.setService("payment-service");
|
||||
|
||||
Map<String, Object> evaluation = service.evaluate(
|
||||
request,
|
||||
"HighCPUUsage alert on payment-service has a likely resource saturation issue.",
|
||||
List.of()
|
||||
);
|
||||
|
||||
assertEquals("WARN", evaluation.get("verdict"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void evaluateFailsWhenReportIsMissing() {
|
||||
Map<String, Object> evaluation = service.evaluate(null, "too short", List.of());
|
||||
|
||||
assertEquals("FAIL", evaluation.get("verdict"));
|
||||
}
|
||||
}
|
||||
@@ -28,6 +28,8 @@ class AiOpsServiceTest {
|
||||
ReflectionTestUtils.setField(service, "diagnosisSessionRepository", diagnosisSessionRepository);
|
||||
ReflectionTestUtils.setField(service, "agentStepRepository", agentStepRepository);
|
||||
ReflectionTestUtils.setField(service, "toolInvocationRepository", toolInvocationRepository);
|
||||
ReflectionTestUtils.setField(service, "aiOpsRuleEvaluationService", new AiOpsRuleEvaluationService());
|
||||
ReflectionTestUtils.setField(service, "selfEvaluationMergeService", new SelfEvaluationMergeService());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -95,11 +97,28 @@ class AiOpsServiceTest {
|
||||
assertTrue(prompt.contains("queryPrometheusAlerts only to verify"));
|
||||
assertTrue(prompt.contains("do not create full root-cause or remediation sections"));
|
||||
assertTrue(prompt.contains("Related Risk"));
|
||||
assertTrue(prompt.contains("Recommended lookup_knowledge query: HighCPUUsage payment-service P1 CPU usage is above 80% last_15m"));
|
||||
assertTrue(prompt.contains("preserves alertName and service"));
|
||||
assertTrue(prompt.contains("告警: HighCPUUsage"));
|
||||
assertTrue(prompt.contains("服务: payment-service"));
|
||||
assertFalse(prompt.contains("AIOps scope mode: AUTO_DISCOVERY"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void buildKnowledgeRetrievalQueryUsesPayloadFieldsAndSkipsBlankValues() {
|
||||
AIOpsRequest request = new AIOpsRequest();
|
||||
request.setAlertName("HighLatency");
|
||||
request.setService(" payment-service ");
|
||||
request.setSeverity(" ");
|
||||
request.setDescription("P95 latency above threshold");
|
||||
request.setTimeRange("last_10m");
|
||||
request.setUserRequest("结合日志和指标排查");
|
||||
|
||||
String query = service.buildKnowledgeRetrievalQuery(request);
|
||||
|
||||
assertEquals("HighLatency payment-service P95 latency above threshold last_10m 结合日志和指标排查", query);
|
||||
}
|
||||
|
||||
@Test
|
||||
void buildTaskPromptUsesAutoDiscoveryModeWhenAlertPayloadIsMissing() {
|
||||
String nullRequestPrompt = service.buildTaskPrompt(null);
|
||||
@@ -116,6 +135,7 @@ class AiOpsServiceTest {
|
||||
|
||||
assertTrue(userRequestOnlyPrompt.contains("AIOps scope mode: AUTO_DISCOVERY"));
|
||||
assertTrue(userRequestOnlyPrompt.contains("First call queryPrometheusAlerts"));
|
||||
assertFalse(userRequestOnlyPrompt.contains("Recommended lookup_knowledge query"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -127,10 +147,12 @@ class AiOpsServiceTest {
|
||||
.agentFlow("AI_OPS")
|
||||
.build();
|
||||
when(diagnosisSessionRepository.findBySessionId("aiops-session-001")).thenReturn(Optional.of(session));
|
||||
when(toolInvocationRepository.findBySessionIdOrderByIdAsc("aiops-session-001")).thenReturn(List.of());
|
||||
|
||||
service.persistFinalReport("aiops-session-001", "# 告警分析报告");
|
||||
service.persistFinalReport("aiops-session-001", "# 告警分析报告\nHighCPUUsage payment-service analysis with evidence summary.");
|
||||
|
||||
assertEquals("# 告警分析报告", session.getAnswer());
|
||||
assertEquals("# 告警分析报告\nHighCPUUsage payment-service analysis with evidence summary.", session.getAnswer());
|
||||
assertTrue(session.getSelfEvaluation().contains("aiops_rule_evaluation"));
|
||||
verify(diagnosisSessionRepository).save(session);
|
||||
}
|
||||
|
||||
|
||||
@@ -45,7 +45,7 @@ class DiagnosisTraceServiceTest {
|
||||
.stepCount(2)
|
||||
.toolCallCount(1)
|
||||
.answer("restart payment gateway pool")
|
||||
.selfEvaluation("{\"verifier_evaluation\":{\"verdict\":\"PASS\"}}")
|
||||
.selfEvaluation("{\"verifier_evaluation\":{\"verdict\":\"PASS\"},\"aiops_rule_evaluation\":{\"verdict\":\"WARN\"}}")
|
||||
.feedback("useful")
|
||||
.createdAt(now)
|
||||
.updatedAt(now)
|
||||
@@ -103,6 +103,7 @@ class DiagnosisTraceServiceTest {
|
||||
assertEquals(1, response.getSummary().getPersistedToolCallCount());
|
||||
assertEquals(1, response.getSummary().getReturnedToolCallCount());
|
||||
assertTrue(response.getSummary().isHasVerifierEvaluation());
|
||||
assertTrue(response.getSummary().isHasAiOpsRuleEvaluation());
|
||||
assertTrue(response.getSummary().isHasFeedback());
|
||||
}
|
||||
|
||||
|
||||
@@ -98,6 +98,46 @@ class KnowledgeIndexServiceTest {
|
||||
assertEquals(2, results.size());
|
||||
}
|
||||
|
||||
@Test
|
||||
void testAnalyzeQuery_returnsStructuredHint() {
|
||||
KnowledgeEntry entry = KnowledgeEntry.builder()
|
||||
.filePath("mysql.md")
|
||||
.title("MySQL Doc")
|
||||
.keywords(List.of("mysql", "connection pool"))
|
||||
.category("database")
|
||||
.build();
|
||||
|
||||
service.addToIndex(entry);
|
||||
|
||||
KnowledgeIndexService.L0Hint hint = service.analyzeQuery("mysql connection pool timeout");
|
||||
|
||||
assertEquals(1, hint.matches().size());
|
||||
assertEquals(List.of("mysql", "connection pool"), hint.matchedKeywords());
|
||||
assertEquals(List.of("database"), hint.domains());
|
||||
assertEquals(List.of("mysql", "connection pool"), hint.entities());
|
||||
assertEquals(List.of("MySQL Doc"), hint.titles());
|
||||
assertEquals("database", hint.singleDomainOrNull());
|
||||
}
|
||||
|
||||
@Test
|
||||
void testAnalyzeQuery_multipleDomainsHasNoSingleDomain() {
|
||||
service.addToIndex(KnowledgeEntry.builder()
|
||||
.filePath("mysql.md")
|
||||
.keywords(List.of("timeout"))
|
||||
.category("database")
|
||||
.build());
|
||||
service.addToIndex(KnowledgeEntry.builder()
|
||||
.filePath("api.md")
|
||||
.keywords(List.of("timeout"))
|
||||
.category("api")
|
||||
.build());
|
||||
|
||||
KnowledgeIndexService.L0Hint hint = service.analyzeQuery("timeout");
|
||||
|
||||
assertEquals(2, hint.matches().size());
|
||||
assertNull(hint.singleDomainOrNull());
|
||||
}
|
||||
|
||||
@Test
|
||||
void testExactMatch_noMatch() {
|
||||
KnowledgeEntry entry = KnowledgeEntry.builder()
|
||||
|
||||
+117
@@ -0,0 +1,117 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.config.RagSidecarProperties;
|
||||
import com.superbiz.agent.dto.ComparableRetrievalResult;
|
||||
import com.superbiz.agent.dto.RetrievalComparisonCase;
|
||||
import com.superbiz.agent.dto.RetrievalComparisonReport;
|
||||
import com.superbiz.agent.dto.SidecarRetrievalResponse;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.mockito.Mockito.mock;
|
||||
import static org.mockito.Mockito.when;
|
||||
|
||||
class RagRetrievalSidecarComparisonServiceTest {
|
||||
|
||||
@TempDir
|
||||
Path tempDir;
|
||||
|
||||
@Test
|
||||
void compareWritesSeparateSidecarReports() throws Exception {
|
||||
VectorSearchService vectorSearchService = mock(VectorSearchService.class);
|
||||
SpringAiVectorStoreSidecarService sidecarService = mock(SpringAiVectorStoreSidecarService.class);
|
||||
RagSidecarProperties properties = new RagSidecarProperties();
|
||||
RetrievalResultNormalizer normalizer = new RetrievalResultNormalizer(new ObjectMapper());
|
||||
RagRetrievalSidecarComparisonService comparisonService = new RagRetrievalSidecarComparisonService(
|
||||
vectorSearchService,
|
||||
sidecarService,
|
||||
normalizer,
|
||||
properties,
|
||||
new ObjectMapper()
|
||||
);
|
||||
|
||||
VectorSearchService.SearchResult current = new VectorSearchService.SearchResult();
|
||||
current.setId("current-1");
|
||||
current.setMetadata("{\"_source\":\"current.md\",\"breadcrumb\":\"A\",\"category\":\"api\"}");
|
||||
current.setContent("current content");
|
||||
current.setScore(0.1f);
|
||||
when(vectorSearchService.searchSimilarDocuments("timeout", 3, "api"))
|
||||
.thenReturn(List.of(current));
|
||||
when(sidecarService.search("timeout", 3, "api"))
|
||||
.thenReturn(SidecarRetrievalResponse.builder()
|
||||
.enabled(true)
|
||||
.available(true)
|
||||
.status("available")
|
||||
.results(List.of(ComparableRetrievalResult.builder()
|
||||
.path("sidecar")
|
||||
.rank(1)
|
||||
.source("sidecar.md")
|
||||
.breadcrumb("B")
|
||||
.scoreLabel("similarity")
|
||||
.scoreValue(0.9)
|
||||
.build()))
|
||||
.build());
|
||||
|
||||
RetrievalComparisonReport report = comparisonService.compare(List.of(
|
||||
RetrievalComparisonCase.builder()
|
||||
.caseId("case-1")
|
||||
.scenario("aiops")
|
||||
.query("timeout")
|
||||
.category("api")
|
||||
.build()
|
||||
), 3);
|
||||
|
||||
assertEquals(1, report.getCaseCount());
|
||||
assertEquals("available", report.getSidecarStatus());
|
||||
assertTrue(report.getResults().get(0).getDifferences().contains("top_source_differs"));
|
||||
Path json = tempDir.resolve("sidecar.json");
|
||||
Path markdown = tempDir.resolve("sidecar.md");
|
||||
comparisonService.writeReports(report, json, markdown);
|
||||
|
||||
assertTrue(Files.readString(json).contains("\"sidecarStatus\""));
|
||||
assertTrue(Files.readString(markdown).contains("RAG Sidecar Retrieval Comparison"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void compareGoldenCasesLoadsExistingCaseShape() throws Exception {
|
||||
VectorSearchService vectorSearchService = mock(VectorSearchService.class);
|
||||
SpringAiVectorStoreSidecarService sidecarService = mock(SpringAiVectorStoreSidecarService.class);
|
||||
RagRetrievalSidecarComparisonService comparisonService = new RagRetrievalSidecarComparisonService(
|
||||
vectorSearchService,
|
||||
sidecarService,
|
||||
new RetrievalResultNormalizer(new ObjectMapper()),
|
||||
new RagSidecarProperties(),
|
||||
new ObjectMapper()
|
||||
);
|
||||
when(vectorSearchService.searchSimilarDocuments("query", 2, null)).thenReturn(List.of());
|
||||
when(sidecarService.search("query", 2, null))
|
||||
.thenReturn(SidecarRetrievalResponse.builder()
|
||||
.enabled(false)
|
||||
.available(false)
|
||||
.status("disabled")
|
||||
.results(List.of())
|
||||
.build());
|
||||
Path cases = tempDir.resolve("cases.json");
|
||||
Files.writeString(cases, """
|
||||
{
|
||||
"topK": 2,
|
||||
"cases": [
|
||||
{"caseId": "case-1", "scenario": "chat", "query": "query"}
|
||||
]
|
||||
}
|
||||
""");
|
||||
|
||||
RetrievalComparisonReport report = comparisonService.compareGoldenCases(cases);
|
||||
|
||||
assertEquals(1, report.getCaseCount());
|
||||
assertEquals(2, report.getTopK());
|
||||
assertEquals("disabled", report.getSidecarStatus());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.dto.ComparableRetrievalResult;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.springframework.ai.document.Document;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
class RetrievalResultNormalizerTest {
|
||||
|
||||
private final RetrievalResultNormalizer normalizer = new RetrievalResultNormalizer(new ObjectMapper());
|
||||
|
||||
@Test
|
||||
void fromCurrentParsesMetadataAndLabelsDistanceScore() {
|
||||
VectorSearchService.SearchResult result = new VectorSearchService.SearchResult();
|
||||
result.setId("vec-1");
|
||||
result.setMetadata("{\"docId\":\"doc-1\",\"_source\":\"docs/api.md\",\"title\":\"API\",\"breadcrumb\":\"A > B\",\"category\":\"api\"}");
|
||||
result.setContent("abcdef");
|
||||
result.setScore(0.25f);
|
||||
|
||||
ComparableRetrievalResult comparable = normalizer.fromCurrent(result, 1, 3);
|
||||
|
||||
assertEquals("current", comparable.getPath());
|
||||
assertEquals("docs/api.md", comparable.getSource());
|
||||
assertEquals("doc-1", comparable.getDocId());
|
||||
assertEquals("API", comparable.getTitle());
|
||||
assertEquals("A > B", comparable.getBreadcrumb());
|
||||
assertEquals("api", comparable.getCategory());
|
||||
assertEquals("abc...", comparable.getContentPreview());
|
||||
assertEquals("l2_distance", comparable.getScoreLabel());
|
||||
assertEquals(0.25, comparable.getScoreValue(), 0.0001);
|
||||
}
|
||||
|
||||
@Test
|
||||
void fromSidecarNormalizesDocumentMetadataAndLabelsSimilarityScore() {
|
||||
Document document = Document.builder()
|
||||
.id("doc-vector")
|
||||
.text("sidecar content")
|
||||
.metadata(Map.of(
|
||||
"docId", "doc-2",
|
||||
"_source", "docs/sidecar.md",
|
||||
"title", "Sidecar",
|
||||
"breadcrumb", "Root > Sidecar",
|
||||
"category", "rag"
|
||||
))
|
||||
.score(0.91)
|
||||
.build();
|
||||
|
||||
ComparableRetrievalResult comparable = normalizer.fromSidecar(document, 2, 100);
|
||||
|
||||
assertEquals("sidecar", comparable.getPath());
|
||||
assertEquals(2, comparable.getRank());
|
||||
assertEquals("docs/sidecar.md", comparable.getSource());
|
||||
assertEquals("similarity", comparable.getScoreLabel());
|
||||
assertEquals(0.91, comparable.getScoreValue(), 0.0001);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.config.RagSidecarProperties;
|
||||
import com.superbiz.agent.dto.SidecarRetrievalResponse;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.springframework.ai.vectorstore.VectorStore;
|
||||
import org.springframework.beans.factory.ObjectProvider;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.mockito.Mockito.mock;
|
||||
import static org.mockito.Mockito.never;
|
||||
import static org.mockito.Mockito.verify;
|
||||
import static org.mockito.Mockito.when;
|
||||
|
||||
class SpringAiVectorStoreSidecarServiceTest {
|
||||
|
||||
@Test
|
||||
void disabledSidecarDoesNotRequestVectorStore() {
|
||||
RagSidecarProperties properties = new RagSidecarProperties();
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
SpringAiVectorStoreSidecarService service = new SpringAiVectorStoreSidecarService(
|
||||
properties,
|
||||
provider,
|
||||
new RetrievalResultNormalizer(new ObjectMapper())
|
||||
);
|
||||
|
||||
SidecarRetrievalResponse response = service.search("query", 3, null);
|
||||
|
||||
assertFalse(response.isEnabled());
|
||||
assertFalse(response.isAvailable());
|
||||
assertEquals("disabled", response.getStatus());
|
||||
verify(provider, never()).getIfAvailable();
|
||||
}
|
||||
|
||||
@Test
|
||||
void enabledSidecarReportsMissingVectorStore() {
|
||||
RagSidecarProperties properties = new RagSidecarProperties();
|
||||
properties.setEnabled(true);
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
when(provider.getIfAvailable()).thenReturn(null);
|
||||
SpringAiVectorStoreSidecarService service = new SpringAiVectorStoreSidecarService(
|
||||
properties,
|
||||
provider,
|
||||
new RetrievalResultNormalizer(new ObjectMapper())
|
||||
);
|
||||
|
||||
SidecarRetrievalResponse response = service.search("query", 3, "api");
|
||||
|
||||
assertEquals("missing_vector_store", response.getStatus());
|
||||
assertFalse(response.isAvailable());
|
||||
assertEquals(0, response.getResults().size());
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||||
import com.superbiz.agent.dto.EvidenceBlock;
|
||||
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||||
import com.superbiz.agent.util.SessionContextHolder;
|
||||
import org.junit.jupiter.api.Test;
|
||||
@@ -75,6 +76,17 @@ class ToolInvocationRecorderTest {
|
||||
.success(true)
|
||||
.evidenceStatus(ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED)
|
||||
.l0Titles(List.of("payment/errors.md"))
|
||||
.l0MatchedKeywords(List.of("ERR_TIMEOUT"))
|
||||
.l0Domains(List.of("payment"))
|
||||
.l0Entities(List.of("ERR_TIMEOUT"))
|
||||
.evidenceCandidateCount(2)
|
||||
.evidenceBlockCount(1)
|
||||
.evidenceBlocks(List.of(Map.of(
|
||||
"source", "payment/errors.md",
|
||||
"title", "payment/errors.md",
|
||||
"retrieval_layer", "L0+L1",
|
||||
"hit_reasons", List.of("l0_keywords:ERR_TIMEOUT", "semantic_rank:1")
|
||||
)))
|
||||
.build();
|
||||
|
||||
try {
|
||||
@@ -92,5 +104,48 @@ class ToolInvocationRecorderTest {
|
||||
assertEquals("doc_retrieved", saved.getDedupReason());
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"evidence_status\":\"deduped\""));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"retrieved_domains\":[\"payment\"]"));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"l0_matched_keywords\":[\"ERR_TIMEOUT\"]"));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"l0_domains\":[\"payment\"]"));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"l0_entities\":[\"ERR_TIMEOUT\"]"));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"evidence_candidate_count\":2"));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"evidence_block_count\":1"));
|
||||
assertTrue(saved.getRetrievalDetails().contains("\"evidence_blocks\""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void lookupKnowledgeRecordFromSummarizesEvidenceBlocks() {
|
||||
EvidenceBlock block = EvidenceBlock.builder()
|
||||
.source("doc.md")
|
||||
.title("Doc")
|
||||
.breadcrumb("A > B")
|
||||
.retrievalLayer("L1")
|
||||
.score(0.42)
|
||||
.hitReasons(List.of("semantic_rank:1"))
|
||||
.content("x".repeat(220))
|
||||
.build();
|
||||
|
||||
com.superbiz.agent.dto.LookupResult result = com.superbiz.agent.dto.LookupResult.builder()
|
||||
.found(true)
|
||||
.evidenceCandidateCount(3)
|
||||
.evidenceBlockCount(1)
|
||||
.evidenceBlocks(List.of(block))
|
||||
.build();
|
||||
|
||||
ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.from(
|
||||
"query",
|
||||
KnowledgeIndexService.L0Hint.empty(),
|
||||
List.of(),
|
||||
false,
|
||||
result,
|
||||
null,
|
||||
null,
|
||||
10,
|
||||
-1
|
||||
);
|
||||
|
||||
assertEquals(3, record.evidenceCandidateCount());
|
||||
assertEquals(1, record.evidenceBlockCount());
|
||||
assertEquals(1, record.evidenceBlocks().size());
|
||||
assertTrue(String.valueOf(record.evidenceBlocks().get(0).get("content_preview")).endsWith("..."));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.dto.DocumentChunk;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
class VectorIndexServiceTest {
|
||||
|
||||
@Test
|
||||
void buildEmbeddingTextIncludesTitleAndBreadcrumb() {
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.title("Connection Pool")
|
||||
.breadcrumb("Database > MySQL > Connection Pool")
|
||||
.content("Check active connections and leak detection.")
|
||||
.build();
|
||||
|
||||
String embeddingText = VectorIndexService.buildEmbeddingText(chunk);
|
||||
|
||||
assertEquals("""
|
||||
Title: Connection Pool
|
||||
Path: Database > MySQL > Connection Pool
|
||||
Content:
|
||||
Check active connections and leak detection.""", embeddingText);
|
||||
}
|
||||
|
||||
@Test
|
||||
void buildEmbeddingTextKeepsPlainContentWhenNoStructureExists() {
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.title(" ")
|
||||
.breadcrumb(null)
|
||||
.content("Plain chunk content.")
|
||||
.build();
|
||||
|
||||
assertEquals("Plain chunk content.", VectorIndexService.buildEmbeddingText(chunk));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.mockito.ArgumentCaptor;
|
||||
import org.springframework.ai.document.Document;
|
||||
import org.springframework.ai.vectorstore.SearchRequest;
|
||||
import org.springframework.ai.vectorstore.VectorStore;
|
||||
import org.springframework.beans.factory.ObjectProvider;
|
||||
import org.springframework.test.util.ReflectionTestUtils;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.mockito.ArgumentMatchers.any;
|
||||
import static org.mockito.ArgumentMatchers.eq;
|
||||
import static org.mockito.Mockito.doReturn;
|
||||
import static org.mockito.Mockito.mock;
|
||||
import static org.mockito.Mockito.never;
|
||||
import static org.mockito.Mockito.spy;
|
||||
import static org.mockito.Mockito.verify;
|
||||
import static org.mockito.Mockito.when;
|
||||
|
||||
class VectorSearchServiceTest {
|
||||
|
||||
@Test
|
||||
void sdkModeBypassesVectorStore() {
|
||||
VectorSearchService service = spy(new VectorSearchService());
|
||||
setMode(service, "sdk");
|
||||
VectorSearchService.SearchResult expected = result("sdk-doc", 0.2f);
|
||||
doReturn(List.of(expected))
|
||||
.when(service).searchSimilarDocumentsWithSdk("query", 3, null);
|
||||
|
||||
List<VectorSearchService.SearchResult> results = service.searchSimilarDocuments("query", 3, null);
|
||||
|
||||
assertEquals(List.of(expected), results);
|
||||
verify(service, never()).searchSimilarDocumentsWithVectorStore(any(), eq(3), any());
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoModeUsesVectorStoreWhenAvailable() {
|
||||
VectorStore vectorStore = mock(VectorStore.class);
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
when(provider.getIfAvailable()).thenReturn(vectorStore);
|
||||
when(vectorStore.similaritySearch(any(SearchRequest.class))).thenReturn(List.of(
|
||||
Document.builder()
|
||||
.id("spring-doc")
|
||||
.text("spring content")
|
||||
.metadata(Map.of("_source", "spring.md", "category", "api"))
|
||||
.score(0.8)
|
||||
.build()
|
||||
));
|
||||
|
||||
VectorSearchService service = new VectorSearchService();
|
||||
setMode(service, "auto");
|
||||
setVectorStore(service, provider);
|
||||
|
||||
List<VectorSearchService.SearchResult> results = service.searchSimilarDocuments("query", 3, null);
|
||||
|
||||
assertEquals(1, results.size());
|
||||
assertEquals("spring-doc", results.get(0).getId());
|
||||
assertEquals("similarity", results.get(0).getScoreLabel());
|
||||
assertEquals(0.8, results.get(0).getRawScore(), 0.0001);
|
||||
assertEquals(0.4f, results.get(0).getScore(), 0.0001);
|
||||
assertTrue(results.get(0).getMetadata().contains("spring.md"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void vectorStoreSearchUsesDistanceMetadataAsCompatibleScore() {
|
||||
VectorStore vectorStore = mock(VectorStore.class);
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
when(provider.getIfAvailable()).thenReturn(vectorStore);
|
||||
when(vectorStore.similaritySearch(any(SearchRequest.class))).thenReturn(List.of(
|
||||
Document.builder()
|
||||
.id("spring-doc")
|
||||
.text("spring content")
|
||||
.metadata(Map.of("distance", 0.5659486, "category", "api"))
|
||||
.score(0.4340514)
|
||||
.build()
|
||||
));
|
||||
|
||||
VectorSearchService service = new VectorSearchService();
|
||||
setMode(service, "auto");
|
||||
setVectorStore(service, provider);
|
||||
|
||||
List<VectorSearchService.SearchResult> results = service.searchSimilarDocuments("query", 3, null);
|
||||
|
||||
assertEquals(1, results.size());
|
||||
assertEquals("similarity", results.get(0).getScoreLabel());
|
||||
assertEquals(0.4340514, results.get(0).getRawScore(), 0.0001);
|
||||
assertEquals(0.5659486f, results.get(0).getScore(), 0.0001);
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoModeFallsBackToSdkWhenVectorStoreFails() {
|
||||
VectorStore vectorStore = mock(VectorStore.class);
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
when(provider.getIfAvailable()).thenReturn(vectorStore);
|
||||
when(vectorStore.similaritySearch(any(SearchRequest.class))).thenThrow(new RuntimeException("vectorstore down"));
|
||||
|
||||
VectorSearchService service = spy(new VectorSearchService());
|
||||
setMode(service, "auto");
|
||||
setVectorStore(service, provider);
|
||||
VectorSearchService.SearchResult fallback = result("sdk-doc", 0.3f);
|
||||
doReturn(List.of(fallback))
|
||||
.when(service).searchSimilarDocumentsWithSdk("query", 3, null);
|
||||
|
||||
List<VectorSearchService.SearchResult> results = service.searchSimilarDocuments("query", 3, null);
|
||||
|
||||
assertEquals(List.of(fallback), results);
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoModeFallsBackToSdkWhenVectorStoreUnavailable() {
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
when(provider.getIfAvailable()).thenReturn(null);
|
||||
|
||||
VectorSearchService service = spy(new VectorSearchService());
|
||||
setMode(service, "auto");
|
||||
setVectorStore(service, provider);
|
||||
VectorSearchService.SearchResult fallback = result("sdk-doc", 0.3f);
|
||||
doReturn(List.of(fallback))
|
||||
.when(service).searchSimilarDocumentsWithSdk("query", 3, null);
|
||||
|
||||
List<VectorSearchService.SearchResult> results = service.searchSimilarDocuments("query", 3, null);
|
||||
|
||||
assertEquals(List.of(fallback), results);
|
||||
}
|
||||
|
||||
@Test
|
||||
void vectorStoreSearchUsesCategoryFilter() {
|
||||
VectorStore vectorStore = mock(VectorStore.class);
|
||||
ObjectProvider<VectorStore> provider = mock(ObjectProvider.class);
|
||||
when(provider.getIfAvailable()).thenReturn(vectorStore);
|
||||
when(vectorStore.similaritySearch(any(SearchRequest.class))).thenReturn(List.of());
|
||||
VectorSearchService service = new VectorSearchService();
|
||||
setMode(service, "spring-ai");
|
||||
setVectorStore(service, provider);
|
||||
|
||||
service.searchSimilarDocuments("query", 5, "api");
|
||||
|
||||
ArgumentCaptor<SearchRequest> requestCaptor = ArgumentCaptor.forClass(SearchRequest.class);
|
||||
verify(vectorStore).similaritySearch(requestCaptor.capture());
|
||||
SearchRequest request = requestCaptor.getValue();
|
||||
assertEquals("query", request.getQuery());
|
||||
assertEquals(5, request.getTopK());
|
||||
assertTrue(request.hasFilterExpression());
|
||||
assertTrue(request.toString().contains("category"));
|
||||
assertTrue(request.toString().contains("api"));
|
||||
}
|
||||
|
||||
private static void setMode(VectorSearchService service, String mode) {
|
||||
ReflectionTestUtils.setField(service, "vectorStoreMode", mode);
|
||||
}
|
||||
|
||||
private static void setVectorStore(VectorSearchService service, ObjectProvider<VectorStore> provider) {
|
||||
ReflectionTestUtils.setField(service, "vectorStoreProvider", provider);
|
||||
ReflectionTestUtils.setField(service, "objectMapper", new ObjectMapper());
|
||||
ReflectionTestUtils.setField(service, "maxL2Distance", 2.0);
|
||||
}
|
||||
|
||||
private static VectorSearchService.SearchResult result(String id, float score) {
|
||||
VectorSearchService.SearchResult result = new VectorSearchService.SearchResult();
|
||||
result.setId(id);
|
||||
result.setScore(score);
|
||||
result.setRawScore((double) score);
|
||||
result.setScoreLabel("l2_distance");
|
||||
result.setContent("content");
|
||||
result.setMetadata("{}");
|
||||
return result;
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user