docs(rag): add comments on knowledge retrieval pipeline

Document the lookup_knowledge flow from L0 hints through L1 retrieval,
post-processing, packing, and Agent projection so the boundaries and
current limitations are easier to follow.
This commit is contained in:
zhuyongxin
2026-07-27 16:26:06 +08:00
parent edb6153fd6
commit 99d4f6f216
19 changed files with 480 additions and 37 deletions
@@ -7,35 +7,57 @@ import java.util.List;
import java.util.Map;
/**
* Normalized vector retrieval candidate before evidence post-processing.
* L1 向量命中后、后处理前的统一候选结构。
*
* <p>由 {@code KnowledgeDocumentRetriever} 从 {@code VectorSearchService.SearchResult} 映射而来。
* 后处理会基于它做归一化、规则 boost、去重并生成 {@link EvidenceBlock}。</p>
*/
@Data
@Builder
public class RetrievedEvidenceCandidate {
/** 向量库记录 id。 */
private String id;
/**
* 来源标识(_source / source / filePath / docId 等)。
* 当前后处理去重主要依赖该字段,粒度偏文档级。
*/
private String source;
private String title;
private String breadcrumb;
/** chunk 正文原文(后处理前未截断或仅底层原样)。 */
private String content;
/** 固定为 L1(向量层);预留多路召回标记。 */
private String retrievalLayer;
/** 所属 attempt 名,如 FILTERED_VECTOR。 */
private String retrievalAttempt;
/**
* 兼容 L2 距离分(越小越相似),后处理会 normalize 成 baseScore。
*/
private Double score;
/** 底层原始分。 */
private Double rawScore;
/** rawScore 语义标签:l2_distance / similarity。 */
private String scoreLabel;
/** 向量召回顺序(从 1 起),规则 rerank 前的名次。 */
private Integer originalRank;
/**
* 扁平化 metadata(string map)。
* 可能含 docId、chunkIndex、category、kb_scope 等;chunkIndex 尚未提升为一等字段。
*/
private Map<String, String> metadata;
/** 初步命中原因,后处理会追加 boost reasons。 */
private List<String> hitReasons;
}