package com.superbiz.agent.tool; import com.superbiz.agent.domain.entity.ToolInvocation; import com.superbiz.agent.dto.*; import com.superbiz.agent.repository.ToolInvocationRepository; import com.superbiz.agent.service.KnowledgeIndexService; import com.superbiz.agent.service.VectorSearchService; import com.superbiz.agent.util.SessionContextHolder; import lombok.extern.slf4j.Slf4j; import org.springframework.ai.tool.annotation.Tool; import org.springframework.beans.factory.annotation.Autowired; import org.springframework.stereotype.Component; import java.util.List; import java.util.stream.Collectors; /** * 知识库查询工具 * 提供给 Agent 的混合检索工具(L0 + L1) */ @Slf4j @Component public class LookupKnowledgeTool { @Autowired private KnowledgeIndexService knowledgeIndexService; @Autowired private VectorSearchService vectorSearchService; @Autowired private ToolInvocationRepository toolInvocationRepository; @Autowired private RetrievedDocTracker retrievedDocTracker; /** * 查询知识库文档 * * @param query 查询关键词 * @return 查询结果 */ @Tool(description = "查询内部知识库文档,获取错误码定义、接口文档、排障步骤、配置说明等背景信息。" + "采用两阶段检索:L0 精确匹配关键词(< 10ms),L1 语义检索补充(200-500ms)。" + "IMPORTANT: 遇到错误码、接口名、配置项、排障问题时,优先使用此工具。" + "支持的查询场景:" + "1) 错误码定义 - 查询错误码的含义和处理方法,例如 'ERR_TIMEOUT'、'ERR_CONNECTION_REFUSED';" + "2) 接口文档 - 查询 API 接口定义、参数说明、返回格式,例如 'payment-gateway'、'/api/v1/orders';" + "3) 排障步骤 - 查询故障诊断流程、最佳实践,例如 '支付超时排查'、'数据库连接池配置';" + "4) 配置说明 - 查询系统配置、中间件参数,例如 'HikariCP'、'Redis 集群配置'。" + "参数 query: 查询关键词或描述") public LookupResult lookupKnowledge(String query) { // 生成请求ID用于追踪 String requestId = java.util.UUID.randomUUID().toString().substring(0, 8); long startTime = System.currentTimeMillis(); log.info("========================================"); log.info(">>> [工具调用] lookup_knowledge"); log.info(">>> 参数: query = \"{}\"", query); log.info(">>> RequestId: {}", requestId); log.info("----------------------------------------"); // Step 1: L0 精确匹配 long l0Start = System.currentTimeMillis(); List l0Matches = knowledgeIndexService.exactMatch(query); long l0Time = System.currentTimeMillis() - l0Start; log.info("[L0 精确匹配] 完成: matches={}, time={}ms", l0Matches.size(), l0Time); if (!l0Matches.isEmpty()) { log.info("[L0 精确匹配] 找到文档:"); for (int i = 0; i < Math.min(3, l0Matches.size()); i++) { KnowledgeEntry entry = l0Matches.get(i); log.info(" - [{}] 标题: {}, 路径: {}", i+1, entry.getTitle(), entry.getFilePath()); } } // Step 2: 判断是否高置信度(唯一匹配) boolean highConfidence = (l0Matches.size() == 1); log.info("[置信度判断] highConfidence={}, reason={}", highConfidence, highConfidence ? "唯一匹配" : "多个或零个匹配"); // Step 3: L1 条件调用 List l1Results = null; if (!highConfidence) { log.info("[L1 语义检索] L0非唯一匹配,触发L1语义检索..."); long l1Start = System.currentTimeMillis(); l1Results = vectorSearchService.searchSimilarDocuments(query, 3, null); long l1Time = System.currentTimeMillis() - l1Start; log.info("[L1 语义检索] 完成: matches={}, time={}ms", l1Results != null ? l1Results.size() : 0, l1Time); if (l1Results != null && !l1Results.isEmpty()) { log.info("[L1 语义检索] 找到文档:"); for (int i = 0; i < Math.min(3, l1Results.size()); i++) { VectorSearchService.SearchResult result = l1Results.get(i); log.info(" - [{}] 文档ID: {}, 相似度得分: {}", i+1, result.getId(), result.getScore()); } } } else { log.info("[L1 语义检索] L0唯一匹配,跳过L1检索"); } // Step 4: 组装结果 LookupResult result = buildResult(l0Matches, l1Results, highConfidence); // Step 5: session 级去重过滤 String sessionId = SessionContextHolder.getSessionId(); if (sessionId != null && result.isFound()) { String docKey = extractDocKey(result); if (docKey != null && retrievedDocTracker.isAlreadyRetrieved(sessionId, docKey)) { log.info("[去重] 文档已在本会话中检索过,跳过: {}", docKey); saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result); return LookupResult.builder() .found(false) .message("文档已在本会话中检索过,无需重复召回: " + docKey) .build(); } if (docKey != null) { retrievedDocTracker.markRetrieved(sessionId, docKey); } } // 记录结构化结果摘要(替代原始 MD 内容预览) long totalTime = System.currentTimeMillis() - startTime; log.info("----------------------------------------"); log.info("<<< [工具返回] lookup_knowledge"); log.info("<<< 结果: found={}, 耗时: {}ms (L0={}ms, L1={}ms)", result.isFound(), totalTime, l0Time, l1Results != null ? System.currentTimeMillis() - startTime - l0Time : 0); // L0 精确匹配摘要 if (!l0Matches.isEmpty()) { KnowledgeEntry top = l0Matches.get(0); log.info("<<< [L0 主结果] 标题: {}", top.getTitle()); log.info("<<< [L0 主结果] 来源: {}", top.getFilePath()); if (top.getSummary() != null) { log.info("<<< [L0 主结果] 摘要: {}", top.getSummary()); } if (top.getKeywords() != null && !top.getKeywords().isEmpty()) { log.info("<<< [L0 主结果] 关键词: {}", String.join(", ", top.getKeywords())); } // 内容概况:长度 + 章节数 String content = result.getPrimary() != null ? result.getPrimary().getContent() : null; if (content != null) { int headingCount = countMdHeadings(content); log.info("<<< [L0 主结果] 内容: {} 字符, {} 个章节", content.length(), headingCount); } } // L1 语义检索摘要 if (l1Results != null && !l1Results.isEmpty()) { VectorSearchService.SearchResult topL1 = l1Results.get(0); log.info("<<< [L1 补充] 来源: {}", topL1.getMetadata() != null ? topL1.getMetadata() : topL1.getId()); log.info("<<< [L1 补充] 相似度: {}", String.format("%.4f", topL1.getScore())); if (topL1.getContent() != null) { String snippet = extractFirstMeaningfulLine(topL1.getContent(), 120); log.info("<<< [L1 补充] 内容片段: {}", snippet); log.info("<<< [L1 补充] 片段长度: {} 字符", topL1.getContent().length()); } } log.info("========================================"); // 记录 tool_invocation(持久化检索明细) saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result); return result; } /** * 保存工具调用明细到 tool_invocation 表 */ private void saveToolInvocation(String query, List l0Matches, List l1Results, boolean highConfidence, long startTime, LookupResult result) { try { String sessionId = SessionContextHolder.getSessionId(); if (sessionId == null) return; // 非会话上下文不记录 boolean hasL0 = l0Matches != null && !l0Matches.isEmpty(); boolean hasL1 = l1Results != null && !l1Results.isEmpty(); long duration = System.currentTimeMillis() - startTime; String layer; String outputPreview = null; int outputLength = 0; int l0Count = 0; int l1Count = 0; boolean truncated = false; if (hasL0 && !highConfidence) { layer = "L0+L1"; l0Count = l0Matches.size(); l1Count = l1Results.size(); } else if (hasL0) { layer = "L0"; l0Count = l0Matches.size(); } else if (hasL1) { layer = "L1"; l1Count = l1Results.size(); } else { layer = null; } // 拼接 output_preview(前500字符) if (result != null && result.getPrimary() != null && result.getPrimary().getContent() != null) { String content = result.getPrimary().getContent(); outputLength = content.length(); if (content.length() > 500) { outputPreview = content.substring(0, 500) + "..."; truncated = true; } else { outputPreview = content; } } else if (l1Results != null && !l1Results.isEmpty() && l1Results.get(0).getContent() != null) { String content = l1Results.get(0).getContent(); outputLength = content.length(); if (content.length() > 500) { outputPreview = content.substring(0, 500) + "..."; truncated = true; } else { outputPreview = content; } } // 构建检索明细 JSON StringBuilder details = new StringBuilder("{"); if (hasL0) { details.append("\"l0_titles\":["); for (int i = 0; i < Math.min(3, l0Matches.size()); i++) { if (i > 0) details.append(","); details.append("\"").append(escapeJson(l0Matches.get(i).getTitle())).append("\""); } details.append("]"); } if (hasL1) { if (hasL0) details.append(","); details.append("\"l1_scores\":["); for (int i = 0; i < Math.min(3, l1Results.size()); i++) { if (i > 0) details.append(","); details.append(l1Results.get(i).getScore()); } details.append("]"); } details.append("}"); ToolInvocation inv = ToolInvocation.builder() .sessionId(sessionId) .toolName("lookup_knowledge") .inputParams("{\"query\":\"" + escapeJson(query) + "\"}") .outputPreview(outputPreview) .outputLength(outputLength) .retrievalLayer(layer) .l0MatchCount(hasL0 ? l0Count : null) .l1MatchCount(hasL1 ? l1Count : null) .isTruncated(truncated) .retrievalDetails(details.toString()) .durationMs((int) duration) .success(true) .build(); toolInvocationRepository.save(inv); log.debug("tool_invocation 已保存: sessionId={}, layer={}, duration={}ms", sessionId, layer, duration); } catch (Exception e) { log.error("保存 tool_invocation 失败", e); } } private String escapeJson(String s) { if (s == null) return ""; return s.replace("\\", "\\\\") .replace("\"", "\\\"") .replace("\n", "\\n") .replace("\r", "\\r") .replace("\t", "\\t"); } /** * 组装查询结果 * * @param l0Matches L0 匹配结果 * @param l1Results L1 检索结果 * @param highConfidence 是否高置信度 * @return 组装后的结果 */ private LookupResult buildResult( List l0Matches, List l1Results, boolean highConfidence ) { LookupResult.LookupResultBuilder builder = LookupResult.builder(); // 构建 primary(L0 结果) PrimaryResult primary = null; if (l0Matches != null && !l0Matches.isEmpty()) { KnowledgeEntry first = l0Matches.get(0); boolean hasL1 = l1Results != null && !l1Results.isEmpty(); // 场景决策:唯一匹配或 L1 无结果 → LLM 需要正文内容;多匹配且有 L1 → 只需元数据 boolean needFullContent = highConfidence || !hasL1; String content = needFullContent ? buildCompactSummary(first) : buildMetadataOnlySummary(first); if (content != null) { primary = PrimaryResult.builder() .content(content) .source(first.getFilePath()) .matchType("exact_L0") .confidence(highConfidence ? "high" : "low") .availableSections(null) // MVP 返回 null .build(); log.debug("L0结果已构建: source={}, contentLength={}", first.getFilePath(), content.length()); } else { log.warn("L0匹配但文件读取失败: {}", first.getFilePath()); } } builder.primary(primary); // 构建 supplement(L1 结果) SupplementResult supplement = null; boolean hasL1 = l1Results != null && !l1Results.isEmpty(); if (hasL1) { VectorSearchService.SearchResult firstL1 = l1Results.get(0); supplement = SupplementResult.builder() .content(firstL1.getContent()) .source(firstL1.getMetadata()) .matchType("semantic_L1") .build(); log.debug("L1结果已构建: source={}, score={}", firstL1.getMetadata(), firstL1.getScore()); } builder.supplement(supplement); // 判断是否找到结果(primary 或 supplement 至少有一个) boolean found = (primary != null) || (supplement != null); builder.found(found); return builder.build(); } /** * 统计 MD 文档中的章节数(二级标题 ## 数量) */ private int countMdHeadings(String content) { if (content == null) return 0; return (int) content.lines() .filter(l -> l.trim().startsWith("##")) .count(); } /** * 构建紧凑文档摘要(替代原始 MD 全文,节省上下文窗口) * 组合:title/summary + 章节结构 + 正文片段(~500 字符) */ private String buildCompactSummary(KnowledgeEntry entry) { String rawContent = knowledgeIndexService.readDocument(entry.getFilePath(), 2000); if (rawContent == null) return null; // 跳过 YAML frontmatter 得到正文 String body = rawContent; if (body.startsWith("---")) { int end = body.indexOf("---", 3); if (end != -1) { body = body.substring(end + 3).trim(); } } StringBuilder sb = new StringBuilder(); // 1. 元数据头(始终包含) sb.append("文档: ").append(entry.getTitle()).append("\n"); if (entry.getSummary() != null) { sb.append("摘要: ").append(entry.getSummary()).append("\n"); } // 2. 章节结构(## 标题列表) String headings = body.lines() .filter(l -> l.trim().startsWith("##")) .map(l -> " - " + l.trim().replaceAll("^#+\\s*", "")) .collect(Collectors.joining("\n")); if (!headings.isEmpty()) { sb.append("章节:\n").append(headings).append("\n"); } sb.append("---\n"); // 3. 正文片段(去标题行、去空行,智能截断) String textContent = body.lines() .filter(l -> !l.trim().startsWith("#") && !l.trim().isEmpty()) .collect(Collectors.joining("\n")) .trim(); // 短文档保留更多内容,长文档节省上下文 int maxBodyChars = body.length() < 500 ? 800 : 500; if (textContent.length() > maxBodyChars) { sb.append(textContent, 0, maxBodyChars).append("..."); } else { sb.append(textContent); } return sb.toString(); } /** * 构建纯元数据摘要(不读文件,仅用内存索引信息) * 多匹配且有 L1 补充时使用,L0 只需告知 LLM 命中了哪些文档 */ private String buildMetadataOnlySummary(KnowledgeEntry entry) { StringBuilder sb = new StringBuilder(); sb.append("文档: ").append(entry.getTitle()).append("\n"); if (entry.getSummary() != null) { sb.append("摘要: ").append(entry.getSummary()).append("\n"); } if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) { sb.append("关键词: ").append(String.join(", ", entry.getKeywords())).append("\n"); } sb.append("来源: ").append(entry.getFilePath()).append("\n"); return sb.toString(); } /** * 提取 MD 内容中第一个有意义的文本行(跳过 frontmatter 和标题行) */ private String extractFirstMeaningfulLine(String content, int maxLen) { if (content == null || content.isBlank()) return "(空)"; String text = content.trim(); // 跳过 YAML frontmatter (--- ... ---) if (text.startsWith("---")) { int end = text.indexOf("---", 3); if (end != -1) { text = text.substring(end + 3); } } // 查找第一个非空、非标题行 String[] lines = text.split("\n"); for (String line : lines) { String tl = line.trim(); if (!tl.isEmpty() && !tl.startsWith("#")) { return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "..."; } } // 兜底:第一行非空行 for (String line : lines) { if (!line.trim().isEmpty()) { String tl = line.trim(); return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "..."; } } return "(无有效内容)"; } private String extractDocKey(LookupResult result) { if (result.getPrimary() != null && result.getPrimary().getSource() != null) { return result.getPrimary().getSource(); } if (result.getSupplement() != null && result.getSupplement().getSource() != null) { return result.getSupplement().getSource(); } return null; } }