- RetrievedDocTracker: sessionId → Set<filePath> 会话级去重,LookupKnowledgeTool Step 5 过滤已检索文档 - KnowledgeDomainService: 域级聚合,LLM 生成 when_to_retrieve,构建 knowledge map YAML - DocumentFieldEnricher: 上传时 LLM 补全 covers + whenToRetrieve(含同域文档排除上下文) - KnowledgeDomain entity + V009 迁移: 域级元数据持久化,避免重启重复 LLM 调用 - ChatService: 注入 knowledge map 到 Planner prompt,会话结束时清理去重状态 - KnowledgeIndexService: 手写 JSON 解析替换为 Jackson ObjectMapper,启动时补建缺失域记录 - chat-planner-prompt: 新增知识库检索规则(按域 when_to_retrieve 判断,每域最多一次检索) - doc-field-enricher-prompt / domain-summary-prompt: 外部化 LLM 提示词
464 lines
19 KiB
Java
464 lines
19 KiB
Java
package com.superbiz.agent.tool;
|
||
|
||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||
import com.superbiz.agent.dto.*;
|
||
import com.superbiz.agent.repository.ToolInvocationRepository;
|
||
import com.superbiz.agent.service.KnowledgeIndexService;
|
||
import com.superbiz.agent.service.VectorSearchService;
|
||
import com.superbiz.agent.util.SessionContextHolder;
|
||
import lombok.extern.slf4j.Slf4j;
|
||
import org.springframework.ai.tool.annotation.Tool;
|
||
import org.springframework.beans.factory.annotation.Autowired;
|
||
import org.springframework.stereotype.Component;
|
||
|
||
import java.util.List;
|
||
import java.util.stream.Collectors;
|
||
|
||
/**
|
||
* 知识库查询工具
|
||
* 提供给 Agent 的混合检索工具(L0 + L1)
|
||
*/
|
||
@Slf4j
|
||
@Component
|
||
public class LookupKnowledgeTool {
|
||
|
||
@Autowired
|
||
private KnowledgeIndexService knowledgeIndexService;
|
||
|
||
@Autowired
|
||
private VectorSearchService vectorSearchService;
|
||
|
||
@Autowired
|
||
private ToolInvocationRepository toolInvocationRepository;
|
||
|
||
@Autowired
|
||
private RetrievedDocTracker retrievedDocTracker;
|
||
|
||
/**
|
||
* 查询知识库文档
|
||
*
|
||
* @param query 查询关键词
|
||
* @return 查询结果
|
||
*/
|
||
@Tool(description = "查询内部知识库文档,获取错误码定义、接口文档、排障步骤、配置说明等背景信息。" +
|
||
"采用两阶段检索:L0 精确匹配关键词(< 10ms),L1 语义检索补充(200-500ms)。" +
|
||
"IMPORTANT: 遇到错误码、接口名、配置项、排障问题时,优先使用此工具。" +
|
||
"支持的查询场景:" +
|
||
"1) 错误码定义 - 查询错误码的含义和处理方法,例如 'ERR_TIMEOUT'、'ERR_CONNECTION_REFUSED';" +
|
||
"2) 接口文档 - 查询 API 接口定义、参数说明、返回格式,例如 'payment-gateway'、'/api/v1/orders';" +
|
||
"3) 排障步骤 - 查询故障诊断流程、最佳实践,例如 '支付超时排查'、'数据库连接池配置';" +
|
||
"4) 配置说明 - 查询系统配置、中间件参数,例如 'HikariCP'、'Redis 集群配置'。" +
|
||
"参数 query: 查询关键词或描述")
|
||
public LookupResult lookupKnowledge(String query) {
|
||
// 生成请求ID用于追踪
|
||
String requestId = java.util.UUID.randomUUID().toString().substring(0, 8);
|
||
long startTime = System.currentTimeMillis();
|
||
|
||
log.info("========================================");
|
||
log.info(">>> [工具调用] lookup_knowledge");
|
||
log.info(">>> 参数: query = \"{}\"", query);
|
||
log.info(">>> RequestId: {}", requestId);
|
||
log.info("----------------------------------------");
|
||
|
||
// Step 1: L0 精确匹配
|
||
long l0Start = System.currentTimeMillis();
|
||
List<KnowledgeEntry> l0Matches = knowledgeIndexService.exactMatch(query);
|
||
long l0Time = System.currentTimeMillis() - l0Start;
|
||
log.info("[L0 精确匹配] 完成: matches={}, time={}ms", l0Matches.size(), l0Time);
|
||
if (!l0Matches.isEmpty()) {
|
||
log.info("[L0 精确匹配] 找到文档:");
|
||
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
|
||
KnowledgeEntry entry = l0Matches.get(i);
|
||
log.info(" - [{}] 标题: {}, 路径: {}", i+1, entry.getTitle(), entry.getFilePath());
|
||
}
|
||
}
|
||
|
||
// Step 2: 判断是否高置信度(唯一匹配)
|
||
boolean highConfidence = (l0Matches.size() == 1);
|
||
log.info("[置信度判断] highConfidence={}, reason={}",
|
||
highConfidence, highConfidence ? "唯一匹配" : "多个或零个匹配");
|
||
|
||
// Step 3: L1 条件调用
|
||
List<VectorSearchService.SearchResult> l1Results = null;
|
||
if (!highConfidence) {
|
||
log.info("[L1 语义检索] L0非唯一匹配,触发L1语义检索...");
|
||
long l1Start = System.currentTimeMillis();
|
||
l1Results = vectorSearchService.searchSimilarDocuments(query, 3, null);
|
||
long l1Time = System.currentTimeMillis() - l1Start;
|
||
log.info("[L1 语义检索] 完成: matches={}, time={}ms",
|
||
l1Results != null ? l1Results.size() : 0, l1Time);
|
||
if (l1Results != null && !l1Results.isEmpty()) {
|
||
log.info("[L1 语义检索] 找到文档:");
|
||
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
|
||
VectorSearchService.SearchResult result = l1Results.get(i);
|
||
log.info(" - [{}] 文档ID: {}, 相似度得分: {}", i+1, result.getId(), result.getScore());
|
||
}
|
||
}
|
||
} else {
|
||
log.info("[L1 语义检索] L0唯一匹配,跳过L1检索");
|
||
}
|
||
|
||
// Step 4: 组装结果
|
||
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
|
||
|
||
// Step 5: session 级去重过滤
|
||
String sessionId = SessionContextHolder.getSessionId();
|
||
if (sessionId != null && result.isFound()) {
|
||
String docKey = extractDocKey(result);
|
||
if (docKey != null && retrievedDocTracker.isAlreadyRetrieved(sessionId, docKey)) {
|
||
log.info("[去重] 文档已在本会话中检索过,跳过: {}", docKey);
|
||
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result);
|
||
return LookupResult.builder()
|
||
.found(false)
|
||
.message("文档已在本会话中检索过,无需重复召回: " + docKey)
|
||
.build();
|
||
}
|
||
if (docKey != null) {
|
||
retrievedDocTracker.markRetrieved(sessionId, docKey);
|
||
}
|
||
}
|
||
|
||
// 记录结构化结果摘要(替代原始 MD 内容预览)
|
||
long totalTime = System.currentTimeMillis() - startTime;
|
||
log.info("----------------------------------------");
|
||
log.info("<<< [工具返回] lookup_knowledge");
|
||
log.info("<<< 结果: found={}, 耗时: {}ms (L0={}ms, L1={}ms)",
|
||
result.isFound(), totalTime, l0Time,
|
||
l1Results != null ? System.currentTimeMillis() - startTime - l0Time : 0);
|
||
|
||
// L0 精确匹配摘要
|
||
if (!l0Matches.isEmpty()) {
|
||
KnowledgeEntry top = l0Matches.get(0);
|
||
log.info("<<< [L0 主结果] 标题: {}", top.getTitle());
|
||
log.info("<<< [L0 主结果] 来源: {}", top.getFilePath());
|
||
if (top.getSummary() != null) {
|
||
log.info("<<< [L0 主结果] 摘要: {}", top.getSummary());
|
||
}
|
||
if (top.getKeywords() != null && !top.getKeywords().isEmpty()) {
|
||
log.info("<<< [L0 主结果] 关键词: {}", String.join(", ", top.getKeywords()));
|
||
}
|
||
// 内容概况:长度 + 章节数
|
||
String content = result.getPrimary() != null ? result.getPrimary().getContent() : null;
|
||
if (content != null) {
|
||
int headingCount = countMdHeadings(content);
|
||
log.info("<<< [L0 主结果] 内容: {} 字符, {} 个章节",
|
||
content.length(), headingCount);
|
||
}
|
||
}
|
||
|
||
// L1 语义检索摘要
|
||
if (l1Results != null && !l1Results.isEmpty()) {
|
||
VectorSearchService.SearchResult topL1 = l1Results.get(0);
|
||
log.info("<<< [L1 补充] 来源: {}", topL1.getMetadata() != null ? topL1.getMetadata() : topL1.getId());
|
||
log.info("<<< [L1 补充] 相似度: {}", String.format("%.4f", topL1.getScore()));
|
||
if (topL1.getContent() != null) {
|
||
String snippet = extractFirstMeaningfulLine(topL1.getContent(), 120);
|
||
log.info("<<< [L1 补充] 内容片段: {}", snippet);
|
||
log.info("<<< [L1 补充] 片段长度: {} 字符", topL1.getContent().length());
|
||
}
|
||
}
|
||
|
||
log.info("========================================");
|
||
|
||
// 记录 tool_invocation(持久化检索明细)
|
||
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result);
|
||
|
||
return result;
|
||
}
|
||
|
||
/**
|
||
* 保存工具调用明细到 tool_invocation 表
|
||
*/
|
||
private void saveToolInvocation(String query, List<KnowledgeEntry> l0Matches,
|
||
List<VectorSearchService.SearchResult> l1Results,
|
||
boolean highConfidence, long startTime, LookupResult result) {
|
||
try {
|
||
String sessionId = SessionContextHolder.getSessionId();
|
||
if (sessionId == null) return; // 非会话上下文不记录
|
||
|
||
boolean hasL0 = l0Matches != null && !l0Matches.isEmpty();
|
||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||
long duration = System.currentTimeMillis() - startTime;
|
||
|
||
String layer;
|
||
String outputPreview = null;
|
||
int outputLength = 0;
|
||
int l0Count = 0;
|
||
int l1Count = 0;
|
||
boolean truncated = false;
|
||
|
||
if (hasL0 && !highConfidence) {
|
||
layer = "L0+L1";
|
||
l0Count = l0Matches.size();
|
||
l1Count = l1Results.size();
|
||
} else if (hasL0) {
|
||
layer = "L0";
|
||
l0Count = l0Matches.size();
|
||
} else if (hasL1) {
|
||
layer = "L1";
|
||
l1Count = l1Results.size();
|
||
} else {
|
||
layer = null;
|
||
}
|
||
|
||
// 拼接 output_preview(前500字符)
|
||
if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) {
|
||
String content = result.getPrimary().getContent();
|
||
outputLength = content.length();
|
||
if (content.length() > 500) {
|
||
outputPreview = content.substring(0, 500) + "...";
|
||
truncated = true;
|
||
} else {
|
||
outputPreview = content;
|
||
}
|
||
} else if (l1Results != null && !l1Results.isEmpty() && l1Results.get(0).getContent() != null) {
|
||
String content = l1Results.get(0).getContent();
|
||
outputLength = content.length();
|
||
if (content.length() > 500) {
|
||
outputPreview = content.substring(0, 500) + "...";
|
||
truncated = true;
|
||
} else {
|
||
outputPreview = content;
|
||
}
|
||
}
|
||
|
||
// 构建检索明细 JSON
|
||
StringBuilder details = new StringBuilder("{");
|
||
if (hasL0) {
|
||
details.append("\"l0_titles\":[");
|
||
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
|
||
if (i > 0) details.append(",");
|
||
details.append("\"").append(escapeJson(l0Matches.get(i).getTitle())).append("\"");
|
||
}
|
||
details.append("]");
|
||
}
|
||
if (hasL1) {
|
||
if (hasL0) details.append(",");
|
||
details.append("\"l1_scores\":[");
|
||
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
|
||
if (i > 0) details.append(",");
|
||
details.append(l1Results.get(i).getScore());
|
||
}
|
||
details.append("]");
|
||
}
|
||
details.append("}");
|
||
|
||
ToolInvocation inv = ToolInvocation.builder()
|
||
.sessionId(sessionId)
|
||
.toolName("lookup_knowledge")
|
||
.inputParams("{\"query\":\"" + escapeJson(query) + "\"}")
|
||
.outputPreview(outputPreview)
|
||
.outputLength(outputLength)
|
||
.retrievalLayer(layer)
|
||
.l0MatchCount(hasL0 ? l0Count : null)
|
||
.l1MatchCount(hasL1 ? l1Count : null)
|
||
.isTruncated(truncated)
|
||
.retrievalDetails(details.toString())
|
||
.durationMs((int) duration)
|
||
.success(true)
|
||
.build();
|
||
|
||
toolInvocationRepository.save(inv);
|
||
log.debug("tool_invocation 已保存: sessionId={}, layer={}, duration={}ms", sessionId, layer, duration);
|
||
} catch (Exception e) {
|
||
log.error("保存 tool_invocation 失败", e);
|
||
}
|
||
}
|
||
|
||
private String escapeJson(String s) {
|
||
if (s == null) return "";
|
||
return s.replace("\\", "\\\\")
|
||
.replace("\"", "\\\"")
|
||
.replace("\n", "\\n")
|
||
.replace("\r", "\\r")
|
||
.replace("\t", "\\t");
|
||
}
|
||
|
||
/**
|
||
* 组装查询结果
|
||
*
|
||
* @param l0Matches L0 匹配结果
|
||
* @param l1Results L1 检索结果
|
||
* @param highConfidence 是否高置信度
|
||
* @return 组装后的结果
|
||
*/
|
||
private LookupResult buildResult(
|
||
List<KnowledgeEntry> l0Matches,
|
||
List<VectorSearchService.SearchResult> l1Results,
|
||
boolean highConfidence
|
||
) {
|
||
LookupResult.LookupResultBuilder builder = LookupResult.builder();
|
||
|
||
// 构建 primary(L0 结果)
|
||
PrimaryResult primary = null;
|
||
if (l0Matches != null && !l0Matches.isEmpty()) {
|
||
KnowledgeEntry first = l0Matches.get(0);
|
||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||
|
||
// 场景决策:唯一匹配或 L1 无结果 → LLM 需要正文内容;多匹配且有 L1 → 只需元数据
|
||
boolean needFullContent = highConfidence || !hasL1;
|
||
String content = needFullContent
|
||
? buildCompactSummary(first)
|
||
: buildMetadataOnlySummary(first);
|
||
|
||
if (content != null) {
|
||
primary = PrimaryResult.builder()
|
||
.content(content)
|
||
.source(first.getFilePath())
|
||
.matchType("exact_L0")
|
||
.confidence(highConfidence ? "high" : "low")
|
||
.availableSections(null) // MVP 返回 null
|
||
.build();
|
||
log.debug("L0结果已构建: source={}, contentLength={}", first.getFilePath(), content.length());
|
||
} else {
|
||
log.warn("L0匹配但文件读取失败: {}", first.getFilePath());
|
||
}
|
||
}
|
||
builder.primary(primary);
|
||
|
||
// 构建 supplement(L1 结果)
|
||
SupplementResult supplement = null;
|
||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||
if (hasL1) {
|
||
VectorSearchService.SearchResult firstL1 = l1Results.get(0);
|
||
supplement = SupplementResult.builder()
|
||
.content(firstL1.getContent())
|
||
.source(firstL1.getMetadata())
|
||
.matchType("semantic_L1")
|
||
.build();
|
||
log.debug("L1结果已构建: source={}, score={}", firstL1.getMetadata(), firstL1.getScore());
|
||
}
|
||
builder.supplement(supplement);
|
||
|
||
// 判断是否找到结果(primary 或 supplement 至少有一个)
|
||
boolean found = (primary != null) || (supplement != null);
|
||
builder.found(found);
|
||
|
||
return builder.build();
|
||
}
|
||
|
||
/**
|
||
* 统计 MD 文档中的章节数(二级标题 ## 数量)
|
||
*/
|
||
private int countMdHeadings(String content) {
|
||
if (content == null) return 0;
|
||
return (int) content.lines()
|
||
.filter(l -> l.trim().startsWith("##"))
|
||
.count();
|
||
}
|
||
|
||
/**
|
||
* 构建紧凑文档摘要(替代原始 MD 全文,节省上下文窗口)
|
||
* 组合:title/summary + 章节结构 + 正文片段(~500 字符)
|
||
*/
|
||
private String buildCompactSummary(KnowledgeEntry entry) {
|
||
String rawContent = knowledgeIndexService.readDocument(entry.getFilePath(), 2000);
|
||
if (rawContent == null) return null;
|
||
|
||
// 跳过 YAML frontmatter 得到正文
|
||
String body = rawContent;
|
||
if (body.startsWith("---")) {
|
||
int end = body.indexOf("---", 3);
|
||
if (end != -1) {
|
||
body = body.substring(end + 3).trim();
|
||
}
|
||
}
|
||
|
||
StringBuilder sb = new StringBuilder();
|
||
|
||
// 1. 元数据头(始终包含)
|
||
sb.append("文档: ").append(entry.getTitle()).append("\n");
|
||
if (entry.getSummary() != null) {
|
||
sb.append("摘要: ").append(entry.getSummary()).append("\n");
|
||
}
|
||
|
||
// 2. 章节结构(## 标题列表)
|
||
String headings = body.lines()
|
||
.filter(l -> l.trim().startsWith("##"))
|
||
.map(l -> " - " + l.trim().replaceAll("^#+\\s*", ""))
|
||
.collect(Collectors.joining("\n"));
|
||
if (!headings.isEmpty()) {
|
||
sb.append("章节:\n").append(headings).append("\n");
|
||
}
|
||
sb.append("---\n");
|
||
|
||
// 3. 正文片段(去标题行、去空行,智能截断)
|
||
String textContent = body.lines()
|
||
.filter(l -> !l.trim().startsWith("#") && !l.trim().isEmpty())
|
||
.collect(Collectors.joining("\n"))
|
||
.trim();
|
||
|
||
// 短文档保留更多内容,长文档节省上下文
|
||
int maxBodyChars = body.length() < 500 ? 800 : 500;
|
||
if (textContent.length() > maxBodyChars) {
|
||
sb.append(textContent, 0, maxBodyChars).append("...");
|
||
} else {
|
||
sb.append(textContent);
|
||
}
|
||
|
||
return sb.toString();
|
||
}
|
||
|
||
/**
|
||
* 构建纯元数据摘要(不读文件,仅用内存索引信息)
|
||
* 多匹配且有 L1 补充时使用,L0 只需告知 LLM 命中了哪些文档
|
||
*/
|
||
private String buildMetadataOnlySummary(KnowledgeEntry entry) {
|
||
StringBuilder sb = new StringBuilder();
|
||
sb.append("文档: ").append(entry.getTitle()).append("\n");
|
||
if (entry.getSummary() != null) {
|
||
sb.append("摘要: ").append(entry.getSummary()).append("\n");
|
||
}
|
||
if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) {
|
||
sb.append("关键词: ").append(String.join(", ", entry.getKeywords())).append("\n");
|
||
}
|
||
sb.append("来源: ").append(entry.getFilePath()).append("\n");
|
||
return sb.toString();
|
||
}
|
||
|
||
/**
|
||
* 提取 MD 内容中第一个有意义的文本行(跳过 frontmatter 和标题行)
|
||
*/
|
||
private String extractFirstMeaningfulLine(String content, int maxLen) {
|
||
if (content == null || content.isBlank()) return "(空)";
|
||
|
||
String text = content.trim();
|
||
// 跳过 YAML frontmatter (--- ... ---)
|
||
if (text.startsWith("---")) {
|
||
int end = text.indexOf("---", 3);
|
||
if (end != -1) {
|
||
text = text.substring(end + 3);
|
||
}
|
||
}
|
||
|
||
// 查找第一个非空、非标题行
|
||
String[] lines = text.split("\n");
|
||
for (String line : lines) {
|
||
String tl = line.trim();
|
||
if (!tl.isEmpty() && !tl.startsWith("#")) {
|
||
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
|
||
}
|
||
}
|
||
|
||
// 兜底:第一行非空行
|
||
for (String line : lines) {
|
||
if (!line.trim().isEmpty()) {
|
||
String tl = line.trim();
|
||
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
|
||
}
|
||
}
|
||
|
||
return "(无有效内容)";
|
||
}
|
||
|
||
private String extractDocKey(LookupResult result) {
|
||
if (result.getPrimary() != null && result.getPrimary().getSource() != null) {
|
||
return result.getPrimary().getSource();
|
||
}
|
||
if (result.getSupplement() != null && result.getSupplement().getSource() != null) {
|
||
return result.getSupplement().getSource();
|
||
}
|
||
return null;
|
||
}
|
||
}
|