docs(rag): add comments on knowledge retrieval pipeline
Document the lookup_knowledge flow from L0 hints through L1 retrieval, post-processing, packing, and Agent projection so the boundaries and current limitations are easier to follow.
This commit is contained in:
@@ -7,35 +7,57 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Normalized vector retrieval candidate before evidence post-processing.
|
||||
* L1 向量命中后、后处理前的统一候选结构。
|
||||
*
|
||||
* <p>由 {@code KnowledgeDocumentRetriever} 从 {@code VectorSearchService.SearchResult} 映射而来。
|
||||
* 后处理会基于它做归一化、规则 boost、去重并生成 {@link EvidenceBlock}。</p>
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
public class RetrievedEvidenceCandidate {
|
||||
|
||||
/** 向量库记录 id。 */
|
||||
private String id;
|
||||
|
||||
/**
|
||||
* 来源标识(_source / source / filePath / docId 等)。
|
||||
* 当前后处理去重主要依赖该字段,粒度偏文档级。
|
||||
*/
|
||||
private String source;
|
||||
|
||||
private String title;
|
||||
|
||||
private String breadcrumb;
|
||||
|
||||
/** chunk 正文原文(后处理前未截断或仅底层原样)。 */
|
||||
private String content;
|
||||
|
||||
/** 固定为 L1(向量层);预留多路召回标记。 */
|
||||
private String retrievalLayer;
|
||||
|
||||
/** 所属 attempt 名,如 FILTERED_VECTOR。 */
|
||||
private String retrievalAttempt;
|
||||
|
||||
/**
|
||||
* 兼容 L2 距离分(越小越相似),后处理会 normalize 成 baseScore。
|
||||
*/
|
||||
private Double score;
|
||||
|
||||
/** 底层原始分。 */
|
||||
private Double rawScore;
|
||||
|
||||
/** rawScore 语义标签:l2_distance / similarity。 */
|
||||
private String scoreLabel;
|
||||
|
||||
/** 向量召回顺序(从 1 起),规则 rerank 前的名次。 */
|
||||
private Integer originalRank;
|
||||
|
||||
/**
|
||||
* 扁平化 metadata(string map)。
|
||||
* 可能含 docId、chunkIndex、category、kb_scope 等;chunkIndex 尚未提升为一等字段。
|
||||
*/
|
||||
private Map<String, String> metadata;
|
||||
|
||||
/** 初步命中原因,后处理会追加 boost reasons。 */
|
||||
private List<String> hitReasons;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user