Files
SuperBizAgent-java/src/main/java/com/superbiz/agent/tool/LookupKnowledgeTool.java
T
zhuyongxin bb44140901 feat(knowledge): 会话级去重 + 知识域地图注入 Planner 解决 ISS-001 重复检索
- RetrievedDocTracker: sessionId → Set<filePath> 会话级去重,LookupKnowledgeTool Step 5 过滤已检索文档
- KnowledgeDomainService: 域级聚合,LLM 生成 when_to_retrieve,构建 knowledge map YAML
- DocumentFieldEnricher: 上传时 LLM 补全 covers + whenToRetrieve(含同域文档排除上下文)
- KnowledgeDomain entity + V009 迁移: 域级元数据持久化,避免重启重复 LLM 调用
- ChatService: 注入 knowledge map 到 Planner prompt,会话结束时清理去重状态
- KnowledgeIndexService: 手写 JSON 解析替换为 Jackson ObjectMapper,启动时补建缺失域记录
- chat-planner-prompt: 新增知识库检索规则(按域 when_to_retrieve 判断,每域最多一次检索)
- doc-field-enricher-prompt / domain-summary-prompt: 外部化 LLM 提示词
2026-07-01 10:47:46 +08:00

464 lines
19 KiB
Java
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package com.superbiz.agent.tool;
import com.superbiz.agent.domain.entity.ToolInvocation;
import com.superbiz.agent.dto.*;
import com.superbiz.agent.repository.ToolInvocationRepository;
import com.superbiz.agent.service.KnowledgeIndexService;
import com.superbiz.agent.service.VectorSearchService;
import com.superbiz.agent.util.SessionContextHolder;
import lombok.extern.slf4j.Slf4j;
import org.springframework.ai.tool.annotation.Tool;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import java.util.List;
import java.util.stream.Collectors;
/**
* 知识库查询工具
* 提供给 Agent 的混合检索工具(L0 + L1)
*/
@Slf4j
@Component
public class LookupKnowledgeTool {
@Autowired
private KnowledgeIndexService knowledgeIndexService;
@Autowired
private VectorSearchService vectorSearchService;
@Autowired
private ToolInvocationRepository toolInvocationRepository;
@Autowired
private RetrievedDocTracker retrievedDocTracker;
/**
* 查询知识库文档
*
* @param query 查询关键词
* @return 查询结果
*/
@Tool(description = "查询内部知识库文档,获取错误码定义、接口文档、排障步骤、配置说明等背景信息。" +
"采用两阶段检索:L0 精确匹配关键词(< 10ms),L1 语义检索补充(200-500ms)。" +
"IMPORTANT: 遇到错误码、接口名、配置项、排障问题时,优先使用此工具。" +
"支持的查询场景:" +
"1) 错误码定义 - 查询错误码的含义和处理方法,例如 'ERR_TIMEOUT'、'ERR_CONNECTION_REFUSED';" +
"2) 接口文档 - 查询 API 接口定义、参数说明、返回格式,例如 'payment-gateway'、'/api/v1/orders';" +
"3) 排障步骤 - 查询故障诊断流程、最佳实践,例如 '支付超时排查'、'数据库连接池配置';" +
"4) 配置说明 - 查询系统配置、中间件参数,例如 'HikariCP'、'Redis 集群配置'。" +
"参数 query: 查询关键词或描述")
public LookupResult lookupKnowledge(String query) {
// 生成请求ID用于追踪
String requestId = java.util.UUID.randomUUID().toString().substring(0, 8);
long startTime = System.currentTimeMillis();
log.info("========================================");
log.info(">>> [工具调用] lookup_knowledge");
log.info(">>> 参数: query = \"{}\"", query);
log.info(">>> RequestId: {}", requestId);
log.info("----------------------------------------");
// Step 1: L0 精确匹配
long l0Start = System.currentTimeMillis();
List<KnowledgeEntry> l0Matches = knowledgeIndexService.exactMatch(query);
long l0Time = System.currentTimeMillis() - l0Start;
log.info("[L0 精确匹配] 完成: matches={}, time={}ms", l0Matches.size(), l0Time);
if (!l0Matches.isEmpty()) {
log.info("[L0 精确匹配] 找到文档:");
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
KnowledgeEntry entry = l0Matches.get(i);
log.info(" - [{}] 标题: {}, 路径: {}", i+1, entry.getTitle(), entry.getFilePath());
}
}
// Step 2: 判断是否高置信度(唯一匹配)
boolean highConfidence = (l0Matches.size() == 1);
log.info("[置信度判断] highConfidence={}, reason={}",
highConfidence, highConfidence ? "唯一匹配" : "多个或零个匹配");
// Step 3: L1 条件调用
List<VectorSearchService.SearchResult> l1Results = null;
if (!highConfidence) {
log.info("[L1 语义检索] L0非唯一匹配,触发L1语义检索...");
long l1Start = System.currentTimeMillis();
l1Results = vectorSearchService.searchSimilarDocuments(query, 3, null);
long l1Time = System.currentTimeMillis() - l1Start;
log.info("[L1 语义检索] 完成: matches={}, time={}ms",
l1Results != null ? l1Results.size() : 0, l1Time);
if (l1Results != null && !l1Results.isEmpty()) {
log.info("[L1 语义检索] 找到文档:");
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
VectorSearchService.SearchResult result = l1Results.get(i);
log.info(" - [{}] 文档ID: {}, 相似度得分: {}", i+1, result.getId(), result.getScore());
}
}
} else {
log.info("[L1 语义检索] L0唯一匹配,跳过L1检索");
}
// Step 4: 组装结果
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
// Step 5: session 级去重过滤
String sessionId = SessionContextHolder.getSessionId();
if (sessionId != null && result.isFound()) {
String docKey = extractDocKey(result);
if (docKey != null && retrievedDocTracker.isAlreadyRetrieved(sessionId, docKey)) {
log.info("[去重] 文档已在本会话中检索过,跳过: {}", docKey);
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result);
return LookupResult.builder()
.found(false)
.message("文档已在本会话中检索过,无需重复召回: " + docKey)
.build();
}
if (docKey != null) {
retrievedDocTracker.markRetrieved(sessionId, docKey);
}
}
// 记录结构化结果摘要(替代原始 MD 内容预览)
long totalTime = System.currentTimeMillis() - startTime;
log.info("----------------------------------------");
log.info("<<< [工具返回] lookup_knowledge");
log.info("<<< 结果: found={}, 耗时: {}ms (L0={}ms, L1={}ms)",
result.isFound(), totalTime, l0Time,
l1Results != null ? System.currentTimeMillis() - startTime - l0Time : 0);
// L0 精确匹配摘要
if (!l0Matches.isEmpty()) {
KnowledgeEntry top = l0Matches.get(0);
log.info("<<< [L0 主结果] 标题: {}", top.getTitle());
log.info("<<< [L0 主结果] 来源: {}", top.getFilePath());
if (top.getSummary() != null) {
log.info("<<< [L0 主结果] 摘要: {}", top.getSummary());
}
if (top.getKeywords() != null && !top.getKeywords().isEmpty()) {
log.info("<<< [L0 主结果] 关键词: {}", String.join(", ", top.getKeywords()));
}
// 内容概况:长度 + 章节数
String content = result.getPrimary() != null ? result.getPrimary().getContent() : null;
if (content != null) {
int headingCount = countMdHeadings(content);
log.info("<<< [L0 主结果] 内容: {} 字符, {} 个章节",
content.length(), headingCount);
}
}
// L1 语义检索摘要
if (l1Results != null && !l1Results.isEmpty()) {
VectorSearchService.SearchResult topL1 = l1Results.get(0);
log.info("<<< [L1 补充] 来源: {}", topL1.getMetadata() != null ? topL1.getMetadata() : topL1.getId());
log.info("<<< [L1 补充] 相似度: {}", String.format("%.4f", topL1.getScore()));
if (topL1.getContent() != null) {
String snippet = extractFirstMeaningfulLine(topL1.getContent(), 120);
log.info("<<< [L1 补充] 内容片段: {}", snippet);
log.info("<<< [L1 补充] 片段长度: {} 字符", topL1.getContent().length());
}
}
log.info("========================================");
// 记录 tool_invocation(持久化检索明细)
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result);
return result;
}
/**
* 保存工具调用明细到 tool_invocation 表
*/
private void saveToolInvocation(String query, List<KnowledgeEntry> l0Matches,
List<VectorSearchService.SearchResult> l1Results,
boolean highConfidence, long startTime, LookupResult result) {
try {
String sessionId = SessionContextHolder.getSessionId();
if (sessionId == null) return; // 非会话上下文不记录
boolean hasL0 = l0Matches != null && !l0Matches.isEmpty();
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
long duration = System.currentTimeMillis() - startTime;
String layer;
String outputPreview = null;
int outputLength = 0;
int l0Count = 0;
int l1Count = 0;
boolean truncated = false;
if (hasL0 && !highConfidence) {
layer = "L0+L1";
l0Count = l0Matches.size();
l1Count = l1Results.size();
} else if (hasL0) {
layer = "L0";
l0Count = l0Matches.size();
} else if (hasL1) {
layer = "L1";
l1Count = l1Results.size();
} else {
layer = null;
}
// 拼接 output_preview(前500字符)
if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) {
String content = result.getPrimary().getContent();
outputLength = content.length();
if (content.length() > 500) {
outputPreview = content.substring(0, 500) + "...";
truncated = true;
} else {
outputPreview = content;
}
} else if (l1Results != null && !l1Results.isEmpty() && l1Results.get(0).getContent() != null) {
String content = l1Results.get(0).getContent();
outputLength = content.length();
if (content.length() > 500) {
outputPreview = content.substring(0, 500) + "...";
truncated = true;
} else {
outputPreview = content;
}
}
// 构建检索明细 JSON
StringBuilder details = new StringBuilder("{");
if (hasL0) {
details.append("\"l0_titles\":[");
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
if (i > 0) details.append(",");
details.append("\"").append(escapeJson(l0Matches.get(i).getTitle())).append("\"");
}
details.append("]");
}
if (hasL1) {
if (hasL0) details.append(",");
details.append("\"l1_scores\":[");
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
if (i > 0) details.append(",");
details.append(l1Results.get(i).getScore());
}
details.append("]");
}
details.append("}");
ToolInvocation inv = ToolInvocation.builder()
.sessionId(sessionId)
.toolName("lookup_knowledge")
.inputParams("{\"query\":\"" + escapeJson(query) + "\"}")
.outputPreview(outputPreview)
.outputLength(outputLength)
.retrievalLayer(layer)
.l0MatchCount(hasL0 ? l0Count : null)
.l1MatchCount(hasL1 ? l1Count : null)
.isTruncated(truncated)
.retrievalDetails(details.toString())
.durationMs((int) duration)
.success(true)
.build();
toolInvocationRepository.save(inv);
log.debug("tool_invocation 已保存: sessionId={}, layer={}, duration={}ms", sessionId, layer, duration);
} catch (Exception e) {
log.error("保存 tool_invocation 失败", e);
}
}
private String escapeJson(String s) {
if (s == null) return "";
return s.replace("\\", "\\\\")
.replace("\"", "\\\"")
.replace("\n", "\\n")
.replace("\r", "\\r")
.replace("\t", "\\t");
}
/**
* 组装查询结果
*
* @param l0Matches L0 匹配结果
* @param l1Results L1 检索结果
* @param highConfidence 是否高置信度
* @return 组装后的结果
*/
private LookupResult buildResult(
List<KnowledgeEntry> l0Matches,
List<VectorSearchService.SearchResult> l1Results,
boolean highConfidence
) {
LookupResult.LookupResultBuilder builder = LookupResult.builder();
// 构建 primary(L0 结果)
PrimaryResult primary = null;
if (l0Matches != null && !l0Matches.isEmpty()) {
KnowledgeEntry first = l0Matches.get(0);
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
// 场景决策:唯一匹配或 L1 无结果 → LLM 需要正文内容;多匹配且有 L1 → 只需元数据
boolean needFullContent = highConfidence || !hasL1;
String content = needFullContent
? buildCompactSummary(first)
: buildMetadataOnlySummary(first);
if (content != null) {
primary = PrimaryResult.builder()
.content(content)
.source(first.getFilePath())
.matchType("exact_L0")
.confidence(highConfidence ? "high" : "low")
.availableSections(null) // MVP 返回 null
.build();
log.debug("L0结果已构建: source={}, contentLength={}", first.getFilePath(), content.length());
} else {
log.warn("L0匹配但文件读取失败: {}", first.getFilePath());
}
}
builder.primary(primary);
// 构建 supplement(L1 结果)
SupplementResult supplement = null;
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
if (hasL1) {
VectorSearchService.SearchResult firstL1 = l1Results.get(0);
supplement = SupplementResult.builder()
.content(firstL1.getContent())
.source(firstL1.getMetadata())
.matchType("semantic_L1")
.build();
log.debug("L1结果已构建: source={}, score={}", firstL1.getMetadata(), firstL1.getScore());
}
builder.supplement(supplement);
// 判断是否找到结果(primary 或 supplement 至少有一个)
boolean found = (primary != null) || (supplement != null);
builder.found(found);
return builder.build();
}
/**
* 统计 MD 文档中的章节数(二级标题 ## 数量)
*/
private int countMdHeadings(String content) {
if (content == null) return 0;
return (int) content.lines()
.filter(l -> l.trim().startsWith("##"))
.count();
}
/**
* 构建紧凑文档摘要(替代原始 MD 全文,节省上下文窗口)
* 组合:title/summary + 章节结构 + 正文片段(~500 字符)
*/
private String buildCompactSummary(KnowledgeEntry entry) {
String rawContent = knowledgeIndexService.readDocument(entry.getFilePath(), 2000);
if (rawContent == null) return null;
// 跳过 YAML frontmatter 得到正文
String body = rawContent;
if (body.startsWith("---")) {
int end = body.indexOf("---", 3);
if (end != -1) {
body = body.substring(end + 3).trim();
}
}
StringBuilder sb = new StringBuilder();
// 1. 元数据头(始终包含)
sb.append("文档: ").append(entry.getTitle()).append("\n");
if (entry.getSummary() != null) {
sb.append("摘要: ").append(entry.getSummary()).append("\n");
}
// 2. 章节结构(## 标题列表)
String headings = body.lines()
.filter(l -> l.trim().startsWith("##"))
.map(l -> " - " + l.trim().replaceAll("^#+\\s*", ""))
.collect(Collectors.joining("\n"));
if (!headings.isEmpty()) {
sb.append("章节:\n").append(headings).append("\n");
}
sb.append("---\n");
// 3. 正文片段(去标题行、去空行,智能截断)
String textContent = body.lines()
.filter(l -> !l.trim().startsWith("#") && !l.trim().isEmpty())
.collect(Collectors.joining("\n"))
.trim();
// 短文档保留更多内容,长文档节省上下文
int maxBodyChars = body.length() < 500 ? 800 : 500;
if (textContent.length() > maxBodyChars) {
sb.append(textContent, 0, maxBodyChars).append("...");
} else {
sb.append(textContent);
}
return sb.toString();
}
/**
* 构建纯元数据摘要(不读文件,仅用内存索引信息)
* 多匹配且有 L1 补充时使用,L0 只需告知 LLM 命中了哪些文档
*/
private String buildMetadataOnlySummary(KnowledgeEntry entry) {
StringBuilder sb = new StringBuilder();
sb.append("文档: ").append(entry.getTitle()).append("\n");
if (entry.getSummary() != null) {
sb.append("摘要: ").append(entry.getSummary()).append("\n");
}
if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) {
sb.append("关键词: ").append(String.join(", ", entry.getKeywords())).append("\n");
}
sb.append("来源: ").append(entry.getFilePath()).append("\n");
return sb.toString();
}
/**
* 提取 MD 内容中第一个有意义的文本行(跳过 frontmatter 和标题行)
*/
private String extractFirstMeaningfulLine(String content, int maxLen) {
if (content == null || content.isBlank()) return "(空)";
String text = content.trim();
// 跳过 YAML frontmatter (--- ... ---)
if (text.startsWith("---")) {
int end = text.indexOf("---", 3);
if (end != -1) {
text = text.substring(end + 3);
}
}
// 查找第一个非空、非标题行
String[] lines = text.split("\n");
for (String line : lines) {
String tl = line.trim();
if (!tl.isEmpty() && !tl.startsWith("#")) {
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
}
}
// 兜底:第一行非空行
for (String line : lines) {
if (!line.trim().isEmpty()) {
String tl = line.trim();
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
}
}
return "(无有效内容)";
}
private String extractDocKey(LookupResult result) {
if (result.getPrimary() != null && result.getPrimary().getSource() != null) {
return result.getPrimary().getSource();
}
if (result.getSupplement() != null && result.getSupplement().getSource() != null) {
return result.getSupplement().getSource();
}
return null;
}
}