492 lines
20 KiB
Java
492 lines
20 KiB
Java
package com.superbiz.agent.tool;
|
||
|
||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||
import com.superbiz.agent.domain.entity.ToolInvocation;
|
||
import com.superbiz.agent.dto.*;
|
||
import com.superbiz.agent.service.KnowledgeIndexService;
|
||
import com.superbiz.agent.service.ToolInvocationRecorder;
|
||
import com.superbiz.agent.service.VectorSearchService;
|
||
import com.superbiz.agent.util.SessionContextHolder;
|
||
import lombok.extern.slf4j.Slf4j;
|
||
import org.springframework.ai.tool.annotation.Tool;
|
||
import org.springframework.beans.factory.annotation.Autowired;
|
||
import org.springframework.beans.factory.annotation.Value;
|
||
import org.springframework.stereotype.Component;
|
||
|
||
import java.util.List;
|
||
import java.util.Locale;
|
||
import java.util.stream.Collectors;
|
||
|
||
/**
|
||
* 知识库查询工具
|
||
* 提供给 Agent 的混合检索工具(L0 + L1)
|
||
* 内置归一化层:将 L0 匹配数 + L1 L2 距离归一化为统一质量等级
|
||
*/
|
||
@Slf4j
|
||
@Component
|
||
public class LookupKnowledgeTool {
|
||
|
||
private static final String LEVEL_PRECISE = "PRECISE";
|
||
private static final String LEVEL_HIGHLY_RELEVANT = "HIGHLY_RELEVANT";
|
||
private static final String LEVEL_REFERENCE = "REFERENCE";
|
||
|
||
private static final String HINT_PRECISE = "知识库中不存在比上述结果更精准的文档";
|
||
private static final String HINT_HIGHLY_RELEVANT = "当前结果已高度相关,继续检索不太可能找到更精准的文档";
|
||
private static final String HINT_REFERENCE = "当前结果为相关参考,如需更精准信息请明确缺少的具体维度";
|
||
|
||
@Value("${retrieval.normalization.max-l2-distance:2.0}")
|
||
private double maxL2Distance;
|
||
|
||
@Value("${retrieval.normalization.highly-relevant-threshold:0.75}")
|
||
private double highlyRelevantThreshold;
|
||
|
||
@Value("${retrieval.normalization.reference-threshold:0.5}")
|
||
private double referenceThreshold;
|
||
|
||
@Autowired
|
||
private KnowledgeIndexService knowledgeIndexService;
|
||
|
||
@Autowired
|
||
private VectorSearchService vectorSearchService;
|
||
|
||
@Autowired
|
||
private ToolInvocationRecorder toolInvocationRecorder;
|
||
|
||
@Autowired
|
||
private RetrievedDocTracker retrievedDocTracker;
|
||
|
||
@Autowired
|
||
private ObjectMapper objectMapper;
|
||
|
||
/**
|
||
* 查询知识库文档
|
||
*
|
||
* @param query 查询关键词
|
||
* @return 查询结果
|
||
*/
|
||
@Tool(description = "查询内部知识库文档,获取错误码定义、接口文档、排障步骤、配置说明等背景信息。" +
|
||
"采用两阶段检索:L0 精确匹配关键词(< 10ms),L1 语义检索补充(200-500ms)。" +
|
||
"IMPORTANT: 遇到错误码、接口名、配置项、排障问题时,优先使用此工具。" +
|
||
"支持的查询场景:" +
|
||
"1) 错误码定义 - 查询错误码的含义和处理方法,例如 'ERR_TIMEOUT'、'ERR_CONNECTION_REFUSED';" +
|
||
"2) 接口文档 - 查询 API 接口定义、参数说明、返回格式,例如 'payment-gateway'、'/api/v1/orders';" +
|
||
"3) 排障步骤 - 查询故障诊断流程、最佳实践,例如 '支付超时排查'、'数据库连接池配置';" +
|
||
"4) 配置说明 - 查询系统配置、中间件参数,例如 'HikariCP'、'Redis 集群配置'。" +
|
||
"参数 query: 查询关键词或描述")
|
||
public LookupResult lookupKnowledge(String query) {
|
||
String requestId = java.util.UUID.randomUUID().toString().substring(0, 8);
|
||
long startTime = System.currentTimeMillis();
|
||
|
||
log.info("========================================");
|
||
log.info(">>> [工具调用] lookup_knowledge");
|
||
log.info(">>> 参数: query = \"{}\"", query);
|
||
log.info(">>> RequestId: {}", requestId);
|
||
log.info("----------------------------------------");
|
||
|
||
// Step 1: L0 精确匹配
|
||
long l0Start = System.currentTimeMillis();
|
||
List<KnowledgeEntry> l0Matches = knowledgeIndexService.exactMatch(query);
|
||
long l0Time = System.currentTimeMillis() - l0Start;
|
||
log.info("[L0 精确匹配] 完成: matches={}, time={}ms", l0Matches.size(), l0Time);
|
||
if (!l0Matches.isEmpty()) {
|
||
log.info("[L0 精确匹配] 找到文档:");
|
||
for (int i = 0; i < Math.min(3, l0Matches.size()); i++) {
|
||
KnowledgeEntry entry = l0Matches.get(i);
|
||
log.info(" - [{}] 标题: {}, 路径: {}, 域: {}", i+1, entry.getTitle(), entry.getFilePath(), entry.getCategory());
|
||
}
|
||
}
|
||
|
||
// Step 2: 判断是否高置信度(唯一匹配)
|
||
boolean highConfidence = (l0Matches.size() == 1);
|
||
log.info("[置信度判断] highConfidence={}, reason={}",
|
||
highConfidence, highConfidence ? "唯一匹配" : "多个或零个匹配");
|
||
|
||
// Step 3: L1 条件调用
|
||
List<VectorSearchService.SearchResult> l1Results = null;
|
||
if (!highConfidence) {
|
||
log.info("[L1 语义检索] L0非唯一匹配,触发L1语义检索...");
|
||
long l1Start = System.currentTimeMillis();
|
||
l1Results = vectorSearchService.searchSimilarDocuments(query, 3, null);
|
||
long l1Time = System.currentTimeMillis() - l1Start;
|
||
log.info("[L1 语义检索] 完成: matches={}, time={}ms",
|
||
l1Results != null ? l1Results.size() : 0, l1Time);
|
||
if (l1Results != null && !l1Results.isEmpty()) {
|
||
log.info("[L1 语义检索] 找到文档:");
|
||
for (int i = 0; i < Math.min(3, l1Results.size()); i++) {
|
||
VectorSearchService.SearchResult result = l1Results.get(i);
|
||
log.info(" - [{}] 文档ID: {}, L2距离: {}", i+1, result.getId(), String.format("%.4f", result.getScore()));
|
||
}
|
||
}
|
||
} else {
|
||
log.info("[L1 语义检索] L0唯一匹配,跳过L1检索");
|
||
}
|
||
|
||
// Step 4: 归一化质量等级判定
|
||
float l1TopScore = (l1Results != null && !l1Results.isEmpty()) ? l1Results.get(0).getScore() : Float.MAX_VALUE;
|
||
RelevanceAssessment assessment = computeRelevance(l0Matches.size(), l1TopScore);
|
||
log.info("[归一化] relevanceLevel={}, completenessHint={}", assessment.level, assessment.hint);
|
||
if (l1TopScore != Float.MAX_VALUE) {
|
||
double similarity = normalizeL2(l1TopScore);
|
||
log.info("[归一化] L2距离={}, similarity={}", String.format("%.4f", l1TopScore), String.format("%.4f", similarity));
|
||
}
|
||
|
||
// Step 5: 组装结果
|
||
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
|
||
result.setRelevanceLevel(assessment.level);
|
||
result.setCompletenessHint(assessment.hint);
|
||
|
||
// Step 6: session 级去重过滤 + 域级行动记忆
|
||
String sessionId = SessionContextHolder.getSessionId();
|
||
String domain = extractDomain(l0Matches, l1Results);
|
||
|
||
if (sessionId != null && result.isFound()) {
|
||
String docKey = extractDocKey(result);
|
||
if (docKey != null && retrievedDocTracker.isAlreadyRetrieved(sessionId, docKey)) {
|
||
log.info("[去重] 文档已在本会话中检索过,跳过: {}", docKey);
|
||
List<String> retrievedDomains = retrievedDocTracker.getRetrievedDomains(sessionId);
|
||
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result, domain, "doc_retrieved");
|
||
return LookupResult.builder()
|
||
.found(false)
|
||
.message("文档已在本会话中检索过,无需重复召回: " + docKey)
|
||
.relevanceLevel(assessment.level)
|
||
.completenessHint(assessment.hint)
|
||
.retrievedDomainsThisSession(retrievedDomains)
|
||
.build();
|
||
}
|
||
if (docKey != null) {
|
||
retrievedDocTracker.markRetrieved(sessionId, domain, docKey);
|
||
}
|
||
}
|
||
|
||
// 附加行动记忆
|
||
if (sessionId != null) {
|
||
result.setRetrievedDomainsThisSession(retrievedDocTracker.getRetrievedDomains(sessionId));
|
||
}
|
||
|
||
// 记录结构化结果摘要
|
||
long totalTime = System.currentTimeMillis() - startTime;
|
||
log.info("----------------------------------------");
|
||
log.info("<<< [工具返回] lookup_knowledge");
|
||
log.info("<<< 结果: found={}, relevanceLevel={}, 耗时: {}ms",
|
||
result.isFound(), result.getRelevanceLevel(), totalTime);
|
||
log.info("<<< 行动记忆: retrievedDomainsThisSession={}", result.getRetrievedDomainsThisSession());
|
||
|
||
if (!l0Matches.isEmpty()) {
|
||
KnowledgeEntry top = l0Matches.get(0);
|
||
log.info("<<< [L0 主结果] 标题: {}", top.getTitle());
|
||
log.info("<<< [L0 主结果] 来源: {}", top.getFilePath());
|
||
log.info("<<< [L0 主结果] 域: {}", top.getCategory());
|
||
if (top.getSummary() != null) {
|
||
log.info("<<< [L0 主结果] 摘要: {}", top.getSummary());
|
||
}
|
||
String content = result.getPrimary() != null ? result.getPrimary().getContent() : null;
|
||
if (content != null) {
|
||
int headingCount = countMdHeadings(content);
|
||
log.info("<<< [L0 主结果] 内容: {} 字符, {} 个章节", content.length(), headingCount);
|
||
}
|
||
}
|
||
|
||
if (l1Results != null && !l1Results.isEmpty()) {
|
||
VectorSearchService.SearchResult topL1 = l1Results.get(0);
|
||
log.info("<<< [L1 补充] 来源: {}", topL1.getMetadata() != null ? topL1.getMetadata() : topL1.getId());
|
||
log.info("<<< [L1 补充] L2距离: {}, similarity: {}",
|
||
String.format("%.4f", topL1.getScore()),
|
||
String.format("%.4f", normalizeL2(topL1.getScore())));
|
||
}
|
||
|
||
log.info("========================================");
|
||
|
||
// 记录 tool_invocation
|
||
saveToolInvocation(query, l0Matches, l1Results, highConfidence, startTime, result, domain, null);
|
||
|
||
return result;
|
||
}
|
||
|
||
// ==================== 归一化层 ====================
|
||
|
||
/**
|
||
* L2 距离 Min-Max 归一化到 [0,1] similarity
|
||
* BGE-M3 输出 L2 归一化单位向量,L2 距离硬上界 = 2.0
|
||
* similarity = 1 - min(score, maxL2Distance) / maxL2Distance
|
||
* score=0 → 1.0(完全相同),score=2.0 → 0.0(完全相反)
|
||
*/
|
||
double normalizeL2(float l2Score) {
|
||
double clamped = Math.min(l2Score, maxL2Distance);
|
||
return 1.0 - clamped / maxL2Distance;
|
||
}
|
||
|
||
/**
|
||
* 归一化质量等级判定
|
||
*
|
||
* @param l0MatchCount L0 匹配数
|
||
* @param l1TopScore L1 最高分(L2 距离),无 L1 结果时传 Float.MAX_VALUE
|
||
* @return RelevanceAssessment(level + hint)
|
||
*/
|
||
RelevanceAssessment computeRelevance(int l0MatchCount, float l1TopScore) {
|
||
double l1Similarity = (l1TopScore != Float.MAX_VALUE) ? normalizeL2(l1TopScore) : 0.0;
|
||
|
||
// L0 唯一匹配 → PRECISE
|
||
if (l0MatchCount == 1) {
|
||
return new RelevanceAssessment(LEVEL_PRECISE, HINT_PRECISE);
|
||
}
|
||
|
||
// L0 命中 + L1 高分 → HIGHLY_RELEVANT
|
||
if (l0MatchCount > 1 && l1Similarity >= highlyRelevantThreshold) {
|
||
return new RelevanceAssessment(LEVEL_HIGHLY_RELEVANT, HINT_HIGHLY_RELEVANT);
|
||
}
|
||
|
||
// 仅 L1 高分 → HIGHLY_RELEVANT
|
||
if (l0MatchCount == 0 && l1Similarity >= highlyRelevantThreshold) {
|
||
return new RelevanceAssessment(LEVEL_HIGHLY_RELEVANT, HINT_HIGHLY_RELEVANT);
|
||
}
|
||
|
||
// L0 多匹配 + L1 中分 → REFERENCE
|
||
if (l0MatchCount > 1 && l1Similarity >= referenceThreshold) {
|
||
return new RelevanceAssessment(LEVEL_REFERENCE, HINT_REFERENCE);
|
||
}
|
||
|
||
// 仅 L1 中分 → REFERENCE
|
||
if (l0MatchCount == 0 && l1Similarity >= referenceThreshold) {
|
||
return new RelevanceAssessment(LEVEL_REFERENCE, HINT_REFERENCE);
|
||
}
|
||
|
||
// L0 多匹配 + 无 L1 / L1 低分 → REFERENCE(L0 命中本身有价值)
|
||
if (l0MatchCount > 1) {
|
||
return new RelevanceAssessment(LEVEL_REFERENCE, HINT_REFERENCE);
|
||
}
|
||
|
||
// 无有效结果
|
||
return new RelevanceAssessment(null, null);
|
||
}
|
||
|
||
/**
|
||
* 归一化评估结果
|
||
*/
|
||
record RelevanceAssessment(String level, String hint) {}
|
||
|
||
// ==================== 域提取 ====================
|
||
|
||
/**
|
||
* 从检索结果中提取域信息
|
||
* 优先使用 L0 的 category,兜底从 L1 metadata 解析
|
||
*/
|
||
private String extractDomain(List<KnowledgeEntry> l0Matches, List<VectorSearchService.SearchResult> l1Results) {
|
||
// 优先 L0
|
||
if (l0Matches != null && !l0Matches.isEmpty()) {
|
||
String category = l0Matches.get(0).getCategory();
|
||
if (category != null && !category.isBlank()) {
|
||
return category;
|
||
}
|
||
}
|
||
|
||
// 兜底 L1:从 metadata JSON 中解析 category
|
||
if (l1Results != null && !l1Results.isEmpty()) {
|
||
try {
|
||
String metadata = l1Results.get(0).getMetadata();
|
||
if (metadata != null && metadata.contains("category")) {
|
||
var node = objectMapper.readTree(metadata);
|
||
if (node.has("category")) {
|
||
return node.get("category").asText();
|
||
}
|
||
}
|
||
} catch (Exception e) {
|
||
log.debug("L1 metadata 解析 category 失败: {}", e.getMessage());
|
||
}
|
||
}
|
||
|
||
return null;
|
||
}
|
||
|
||
// ==================== 入库 ====================
|
||
|
||
/**
|
||
* 保存工具调用明细到 tool_invocation 表
|
||
*/
|
||
private void saveToolInvocation(String query, List<KnowledgeEntry> l0Matches,
|
||
List<VectorSearchService.SearchResult> l1Results,
|
||
boolean highConfidence, long startTime,
|
||
LookupResult result, String domain, String dedupReason) {
|
||
try {
|
||
String sessionId = SessionContextHolder.getSessionId();
|
||
if (sessionId == null) return;
|
||
|
||
long duration = System.currentTimeMillis() - startTime;
|
||
double l1TopSimilarity = (l1Results != null && !l1Results.isEmpty())
|
||
? normalizeL2(l1Results.get(0).getScore())
|
||
: -1;
|
||
|
||
ToolInvocationRecorder.LookupKnowledgeRecord record = ToolInvocationRecorder.LookupKnowledgeRecord.from(
|
||
query,
|
||
l0Matches,
|
||
l1Results,
|
||
highConfidence,
|
||
result,
|
||
domain,
|
||
dedupReason,
|
||
(int) duration,
|
||
l1TopSimilarity
|
||
);
|
||
toolInvocationRecorder.recordLookupKnowledge(record);
|
||
log.debug("tool_invocation 已保存: sessionId={}, layer={}, relevanceLevel={}, duration={}ms",
|
||
sessionId, record.retrievalLayer(), record.relevanceLevel(), duration);
|
||
} catch (Exception e) {
|
||
log.error("保存 tool_invocation 失败", e);
|
||
}
|
||
}
|
||
|
||
// ==================== 结果组装 ====================
|
||
|
||
private LookupResult buildResult(
|
||
List<KnowledgeEntry> l0Matches,
|
||
List<VectorSearchService.SearchResult> l1Results,
|
||
boolean highConfidence
|
||
) {
|
||
LookupResult.LookupResultBuilder builder = LookupResult.builder();
|
||
|
||
// 构建 primary(L0 结果)
|
||
PrimaryResult primary = null;
|
||
if (l0Matches != null && !l0Matches.isEmpty()) {
|
||
KnowledgeEntry first = l0Matches.get(0);
|
||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||
boolean needFullContent = highConfidence || !hasL1;
|
||
String content = needFullContent
|
||
? buildCompactSummary(first)
|
||
: buildMetadataOnlySummary(first);
|
||
|
||
if (content != null) {
|
||
primary = PrimaryResult.builder()
|
||
.content(content)
|
||
.source(first.getFilePath())
|
||
.matchType("exact_L0")
|
||
.confidence(highConfidence ? "high" : "low")
|
||
.availableSections(null)
|
||
.build();
|
||
log.debug("L0结果已构建: source={}, contentLength={}", first.getFilePath(), content.length());
|
||
} else {
|
||
log.warn("L0匹配但文件读取失败: {}", first.getFilePath());
|
||
}
|
||
}
|
||
builder.primary(primary);
|
||
|
||
// 构建 supplement(L1 结果)
|
||
SupplementResult supplement = null;
|
||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||
if (hasL1) {
|
||
VectorSearchService.SearchResult firstL1 = l1Results.get(0);
|
||
supplement = SupplementResult.builder()
|
||
.content(firstL1.getContent())
|
||
.source(firstL1.getMetadata())
|
||
.matchType("semantic_L1")
|
||
.build();
|
||
log.debug("L1结果已构建: source={}, score={}", firstL1.getMetadata(), firstL1.getScore());
|
||
}
|
||
builder.supplement(supplement);
|
||
|
||
boolean found = (primary != null) || (supplement != null);
|
||
builder.found(found);
|
||
|
||
return builder.build();
|
||
}
|
||
|
||
private int countMdHeadings(String content) {
|
||
if (content == null) return 0;
|
||
return (int) content.lines()
|
||
.filter(l -> l.trim().startsWith("##"))
|
||
.count();
|
||
}
|
||
|
||
private String buildCompactSummary(KnowledgeEntry entry) {
|
||
String rawContent = knowledgeIndexService.readDocument(entry.getFilePath(), 2000);
|
||
if (rawContent == null) return null;
|
||
|
||
String body = rawContent;
|
||
if (body.startsWith("---")) {
|
||
int end = body.indexOf("---", 3);
|
||
if (end != -1) {
|
||
body = body.substring(end + 3).trim();
|
||
}
|
||
}
|
||
|
||
StringBuilder sb = new StringBuilder();
|
||
sb.append("文档: ").append(entry.getTitle()).append("\n");
|
||
if (entry.getSummary() != null) {
|
||
sb.append("摘要: ").append(entry.getSummary()).append("\n");
|
||
}
|
||
|
||
String headings = body.lines()
|
||
.filter(l -> l.trim().startsWith("##"))
|
||
.map(l -> " - " + l.trim().replaceAll("^#+\\s*", ""))
|
||
.collect(Collectors.joining("\n"));
|
||
if (!headings.isEmpty()) {
|
||
sb.append("章节:\n").append(headings).append("\n");
|
||
}
|
||
sb.append("---\n");
|
||
|
||
String textContent = body.lines()
|
||
.filter(l -> !l.trim().startsWith("#") && !l.trim().isEmpty())
|
||
.collect(Collectors.joining("\n"))
|
||
.trim();
|
||
|
||
int maxBodyChars = body.length() < 500 ? 800 : 500;
|
||
if (textContent.length() > maxBodyChars) {
|
||
sb.append(textContent, 0, maxBodyChars).append("...");
|
||
} else {
|
||
sb.append(textContent);
|
||
}
|
||
|
||
return sb.toString();
|
||
}
|
||
|
||
private String buildMetadataOnlySummary(KnowledgeEntry entry) {
|
||
StringBuilder sb = new StringBuilder();
|
||
sb.append("文档: ").append(entry.getTitle()).append("\n");
|
||
if (entry.getSummary() != null) {
|
||
sb.append("摘要: ").append(entry.getSummary()).append("\n");
|
||
}
|
||
if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) {
|
||
sb.append("关键词: ").append(String.join(", ", entry.getKeywords())).append("\n");
|
||
}
|
||
sb.append("来源: ").append(entry.getFilePath()).append("\n");
|
||
return sb.toString();
|
||
}
|
||
|
||
private String extractFirstMeaningfulLine(String content, int maxLen) {
|
||
if (content == null || content.isBlank()) return "(空)";
|
||
|
||
String text = content.trim();
|
||
if (text.startsWith("---")) {
|
||
int end = text.indexOf("---", 3);
|
||
if (end != -1) {
|
||
text = text.substring(end + 3);
|
||
}
|
||
}
|
||
|
||
String[] lines = text.split("\n");
|
||
for (String line : lines) {
|
||
String tl = line.trim();
|
||
if (!tl.isEmpty() && !tl.startsWith("#")) {
|
||
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
|
||
}
|
||
}
|
||
|
||
for (String line : lines) {
|
||
if (!line.trim().isEmpty()) {
|
||
String tl = line.trim();
|
||
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
|
||
}
|
||
}
|
||
|
||
return "(无有效内容)";
|
||
}
|
||
|
||
private String extractDocKey(LookupResult result) {
|
||
if (result.getPrimary() != null && result.getPrimary().getSource() != null) {
|
||
return result.getPrimary().getSource();
|
||
}
|
||
if (result.getSupplement() != null && result.getSupplement().getSource() != null) {
|
||
return result.getSupplement().getSource();
|
||
}
|
||
return null;
|
||
}
|
||
}
|