feat(rag): chunk evidence identity, dedup, and search port
Preserve same-document multi-chunk evidence with evidenceKey identity, per-document caps, retrieve-k/return-n split, and a dense KnowledgeSearchPort. Archives Delivery 1 OpenSpec change as the foundation for hybrid retrieval.
This commit is contained in:
@@ -9,8 +9,8 @@ import java.util.Map;
|
||||
/**
|
||||
* L1 向量命中后、后处理前的统一候选结构。
|
||||
*
|
||||
* <p>由 {@code KnowledgeDocumentRetriever} 从 {@code VectorSearchService.SearchResult} 映射而来。
|
||||
* 后处理会基于它做归一化、规则 boost、去重并生成 {@link EvidenceBlock}。</p>
|
||||
* <p>由检索适配器从 {@code KnowledgeSearchHit} / 向量结果映射而来。
|
||||
* 后处理会基于它做归一化、规则 boost、chunk 级去重并生成 {@link EvidenceBlock}。</p>
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@@ -19,9 +19,24 @@ public class RetrievedEvidenceCandidate {
|
||||
/** 向量库记录 id。 */
|
||||
private String id;
|
||||
|
||||
/**
|
||||
* 文档级 id(metadata.docId 等)。
|
||||
* 用于每文档 chunk 上限;不等于 evidenceKey。
|
||||
*/
|
||||
private String docId;
|
||||
|
||||
/** 文档内切片序号;可能为空(老数据)。 */
|
||||
private Integer chunkIndex;
|
||||
|
||||
/**
|
||||
* 片段级去重/投影主键。
|
||||
* 通常为 docId#chunk-N,fallback 为 vector:{id}。
|
||||
*/
|
||||
private String evidenceKey;
|
||||
|
||||
/**
|
||||
* 来源标识(_source / source / filePath / docId 等)。
|
||||
* 当前后处理去重主要依赖该字段,粒度偏文档级。
|
||||
* 可与同文档其他 chunk 重复;不再作为唯一去重键。
|
||||
*/
|
||||
private String source;
|
||||
|
||||
@@ -54,7 +69,7 @@ public class RetrievedEvidenceCandidate {
|
||||
|
||||
/**
|
||||
* 扁平化 metadata(string map)。
|
||||
* 可能含 docId、chunkIndex、category、kb_scope 等;chunkIndex 尚未提升为一等字段。
|
||||
* 可能含 docId、chunkIndex、category、kb_scope 等。
|
||||
*/
|
||||
private Map<String, String> metadata;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user