docs(rag): add comments on knowledge retrieval pipeline

Document the lookup_knowledge flow from L0 hints through L1 retrieval,
post-processing, packing, and Agent projection so the boundaries and
current limitations are easier to follow.
This commit is contained in:
zhuyongxin
2026-07-27 16:26:06 +08:00
parent edb6153fd6
commit 99d4f6f216
19 changed files with 480 additions and 37 deletions
@@ -11,9 +11,22 @@ import com.superbiz.agent.harness.tool.projection.RagResultProjector;
import java.util.Objects;
/** Bridges a typed RAG request and a legacy knowledge executor through ToolBoundary. */
/**
* Harness 侧 RAG 工具适配器。
*
* <p>连接三层:</p>
* <ol>
* <li>解析 Agent 的 typed request({@link RagToolRequest})</li>
* <li>经 {@link ToolBoundary} 执行预算/审计等边界控制</li>
* <li>调用 legacy {@code LookupKnowledgeTool},再用 {@link RagResultProjector}
* 投影成冻结的 Agent 可见契约</li>
* </ol>
*
* <p>这样检索实现可演进,而 Agent tool schema 与 EvidenceGuard 契约保持稳定。</p>
*/
public final class RagToolAdapter {
/** 兼容旧检索后端:只接收 query,返回可序列化的 LookupResult(或等价 Map)。 */
@FunctionalInterface
public interface LegacyExecutor {
Object execute(String query) throws Exception;
@@ -32,6 +45,11 @@ public final class RagToolAdapter {
this.legacyExecutor = Objects.requireNonNull(legacyExecutor, "legacyExecutor must not be null");
}
/**
* 执行一次 lookup_knowledge 工具调用。
*
* <p>流程:校验 request.query -&gt; boundary.execute(legacy) -&gt; projector.project。</p>
*/
public ToolBoundaryResult execute(RunContext context, ToolCallRequestEnvelope envelope) {
try {
RagToolRequest request = objectMapper.readValue(envelope.requestJson(), RagToolRequest.class);
@@ -39,7 +57,9 @@ public final class RagToolAdapter {
return ToolBoundaryResult.error(envelope.toolCallId(), ToolBoundaryErrorCode.INVALID_REQUEST);
}
return boundary.execute(context, envelope,
// legacy 原始 JSON(内部 LookupResult)
ignored -> objectMapper.writeValueAsString(legacyExecutor.execute(request.query())),
// 投影为 Agent 契约(RagToolResult)
raw -> projector.project(request, envelope.toolCallId(), raw));
} catch (Exception e) {
return ToolBoundaryResult.error(envelope == null ? null : envelope.toolCallId(),
@@ -2,10 +2,21 @@ package com.superbiz.agent.harness.tool.contract;
import com.fasterxml.jackson.annotation.JsonProperty;
/**
* Agent 可见的单条 RAG 证据(冻结契约)。
*
* <p>由 {@code RagResultProjector} 从内部 {@code EvidenceBlock} 投影而来。
* EvidenceGuard / 诊断报告引用时使用 {@code document_id}。</p>
*
* <p>注意:当 legacy block 未提供 document_id 时,投影层常把 source 当作 document_id,
* 因此同 source 的多个 chunk 在 Agent 侧也会去重成一条。</p>
*/
public record RagEvidence(
/** 证据引用 id;当前实现常等于 source(文档级)。 */
@JsonProperty("document_id") String documentId,
@JsonProperty("source") String source,
@JsonProperty("title") String title,
@JsonProperty("breadcrumb") String breadcrumb,
/** 截断后的正文摘录(对应内部 content)。 */
@JsonProperty("excerpt") String excerpt) {
}
@@ -15,7 +15,31 @@ import java.util.HashSet;
import java.util.List;
import java.util.Set;
/** Projects legacy knowledge retrieval JSON into the frozen RAG contract. */
/**
* 把 legacy {@code LookupResult} JSON 投影成冻结的 Agent 可见 RAG 契约。
*
* <h3>为什么需要投影</h3>
* 内部检索结果字段较多(retrievalTrace、rerankTrace、score、hitReasons、contextPack…),
* Agent / EvidenceGuard 只应看到受控、有界、可引用的子集。
*
* <h3>保留给 Agent 的字段</h3>
* <ul>
* <li>evidenceStatus / toolCallId / query</li>
* <li>evidence[]:document_id, source, title, breadcrumb, excerpt</li>
* <li>relevanceLevel、truncated、returned_count</li>
* </ul>
*
* <h3>刻意丢弃</h3>
* score、hitReasons、retrievalTrace、rerankTrace、contextPack 等内部可观测细节。
*
* <h3>去重与预算(读代码关键)</h3>
* <ul>
* <li>按 {@code document_id} 去重;若 block 无 document_id,则回退 source/title</li>
* <li>因此同 source 的多个 chunk 在此也会被压成 1 条(与后处理文档级去重叠加)</li>
* <li>条数上限 {@link ToolProjectionLimits#maxEvidence()},excerpt 字符上限,
* 以及总 UTF-8 字节预算(超限从后往前删 evidence)</li>
* </ul>
*/
public final class RagResultProjector {
private final ObjectMapper objectMapper;
@@ -26,6 +50,11 @@ public final class RagResultProjector {
this.limits = limits;
}
/**
* @param request Agent 原始请求(用于回填/截断 query)
* @param toolCallId 框架分配的本次工具调用 id,进入结果供证据引用
* @param rawResponse legacy LookupResult 的 JSON 字符串
*/
public ProjectedToolResult project(RagToolRequest request, String toolCallId,
String rawResponse) throws Exception {
if (request == null || toolCallId == null || toolCallId.isBlank()) {
@@ -40,6 +69,7 @@ public final class RagResultProjector {
String query = bounded(request.query(), limits.maxQueryChars());
truncated = !query.equals(request.query());
List<RagEvidence> evidence = new ArrayList<>();
// 文档级唯一集合:相同 documentId 只保留首次出现
Set<String> documentIds = new HashSet<>();
JsonNode blocks = root.has("evidenceBlocks") ? root.get("evidenceBlocks") : root.get("evidence_blocks");
if (blocks != null && blocks.isArray()) {
@@ -59,9 +89,11 @@ public final class RagResultProjector {
}
String source = text(block, "source");
String title = text(block, "title");
// legacy EvidenceBlock 通常没有 document_id,实际常退化为 source
String documentId = firstNonBlank(text(block, "document_id"), source, title,
"legacy-document-" + ordinal);
if (!documentIds.add(documentId)) {
// 同 documentId 重复:丢弃后续条,并标记 truncated
truncated = true;
continue;
}
@@ -86,10 +118,12 @@ public final class RagResultProjector {
? relevanceLevel(root) : null;
RagToolResult result = new RagToolResult(
status, toolCallId, query, evidence, evidence.size(), relevanceLevel, truncated);
// 总字节预算:仍超限则从尾部删 evidence,直到放得下或变 no_evidence
result = fitBudget(result, truncated);
return new ProjectedToolResult(objectMapper.writeValueAsString(result), result.evidenceStatus());
}
/** 按 maxAgentUtf8Bytes 从后往前删 evidence,保证 Agent 侧 payload 有界。 */
private RagToolResult fitBudget(RagToolResult result, boolean truncated) throws Exception {
RagToolResult current = result;
while (bytes(objectMapper.writeValueAsString(current)) > limits.maxAgentUtf8Bytes()