feat(knowledge): breadcrumb分块上下文 & LookupKnowledgeTool日志优化
- DocumentChunk新增breadcrumb字段,分块时构建完整标题层级路径 - DocumentChunkService splitByHeadings维护标题层级栈算法 - VectorIndexService 将breadcrumb写入Milvus metadata - LookupKnowledgeTool日志替换为结构化摘要,替代原始MD预览 - L0返回策略:唯一匹配用正文摘要,多匹配+L1有结果仅元数据(不读文件) - 新增buildCompactSummary / buildMetadataOnlySummary方法 - 安装frontend-design skill - 创建mvp/文档目录(架构设计+会话存储方案) - 更新测试适配新逻辑
This commit is contained in:
@@ -9,6 +9,7 @@ import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* 知识库查询工具
|
||||
@@ -91,23 +92,46 @@ public class LookupKnowledgeTool {
|
||||
// Step 4: 组装结果
|
||||
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
|
||||
|
||||
// 记录完整结果
|
||||
// 记录结构化结果摘要(替代原始 MD 内容预览)
|
||||
long totalTime = System.currentTimeMillis() - startTime;
|
||||
log.info("----------------------------------------");
|
||||
log.info("<<< [工具返回] lookup_knowledge");
|
||||
log.info("<<< 结果: found={}, matchType={}, confidence={}",
|
||||
result.isFound(),
|
||||
result.getPrimary() != null ? result.getPrimary().getMatchType() : "N/A",
|
||||
result.getPrimary() != null ? result.getPrimary().getConfidence() : "N/A");
|
||||
log.info("<<< 总耗时: {}ms (L0={}ms, L1={}ms)",
|
||||
totalTime, l0Time, l1Results != null ? (totalTime - l0Time) : 0);
|
||||
if (result.isFound() && result.getPrimary() != null) {
|
||||
String content = result.getPrimary().getContent();
|
||||
log.info("<<< 返回内容长度: {} 字符", content != null ? content.length() : 0);
|
||||
if (content != null && content.length() > 200) {
|
||||
log.info("<<< 内容预览: {}", content.substring(0, 200) + "...");
|
||||
log.info("<<< 结果: found={}, 耗时: {}ms (L0={}ms, L1={}ms)",
|
||||
result.isFound(), totalTime, l0Time,
|
||||
l1Results != null ? System.currentTimeMillis() - startTime - l0Time : 0);
|
||||
|
||||
// L0 精确匹配摘要
|
||||
if (!l0Matches.isEmpty()) {
|
||||
KnowledgeEntry top = l0Matches.get(0);
|
||||
log.info("<<< [L0 主结果] 标题: {}", top.getTitle());
|
||||
log.info("<<< [L0 主结果] 来源: {}", top.getFilePath());
|
||||
if (top.getSummary() != null) {
|
||||
log.info("<<< [L0 主结果] 摘要: {}", top.getSummary());
|
||||
}
|
||||
if (top.getKeywords() != null && !top.getKeywords().isEmpty()) {
|
||||
log.info("<<< [L0 主结果] 关键词: {}", String.join(", ", top.getKeywords()));
|
||||
}
|
||||
// 内容概况:长度 + 章节数
|
||||
String content = result.getPrimary() != null ? result.getPrimary().getContent() : null;
|
||||
if (content != null) {
|
||||
int headingCount = countMdHeadings(content);
|
||||
log.info("<<< [L0 主结果] 内容: {} 字符, {} 个章节",
|
||||
content.length(), headingCount);
|
||||
}
|
||||
}
|
||||
|
||||
// L1 语义检索摘要
|
||||
if (l1Results != null && !l1Results.isEmpty()) {
|
||||
VectorSearchService.SearchResult topL1 = l1Results.get(0);
|
||||
log.info("<<< [L1 补充] 来源: {}", topL1.getMetadata() != null ? topL1.getMetadata() : topL1.getId());
|
||||
log.info("<<< [L1 补充] 相似度: {}", String.format("%.4f", topL1.getScore()));
|
||||
if (topL1.getContent() != null) {
|
||||
String snippet = extractFirstMeaningfulLine(topL1.getContent(), 120);
|
||||
log.info("<<< [L1 补充] 内容片段: {}", snippet);
|
||||
log.info("<<< [L1 补充] 片段长度: {} 字符", topL1.getContent().length());
|
||||
}
|
||||
}
|
||||
|
||||
log.info("========================================");
|
||||
|
||||
return result;
|
||||
@@ -132,7 +156,13 @@ public class LookupKnowledgeTool {
|
||||
PrimaryResult primary = null;
|
||||
if (l0Matches != null && !l0Matches.isEmpty()) {
|
||||
KnowledgeEntry first = l0Matches.get(0);
|
||||
String content = knowledgeIndexService.readDocument(first.getFilePath(), 2000);
|
||||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||||
|
||||
// 场景决策:唯一匹配或 L1 无结果 → LLM 需要正文内容;多匹配且有 L1 → 只需元数据
|
||||
boolean needFullContent = highConfidence || !hasL1;
|
||||
String content = needFullContent
|
||||
? buildCompactSummary(first)
|
||||
: buildMetadataOnlySummary(first);
|
||||
|
||||
if (content != null) {
|
||||
primary = PrimaryResult.builder()
|
||||
@@ -169,4 +199,118 @@ public class LookupKnowledgeTool {
|
||||
|
||||
return builder.build();
|
||||
}
|
||||
|
||||
/**
|
||||
* 统计 MD 文档中的章节数(二级标题 ## 数量)
|
||||
*/
|
||||
private int countMdHeadings(String content) {
|
||||
if (content == null) return 0;
|
||||
return (int) content.lines()
|
||||
.filter(l -> l.trim().startsWith("##"))
|
||||
.count();
|
||||
}
|
||||
|
||||
/**
|
||||
* 构建紧凑文档摘要(替代原始 MD 全文,节省上下文窗口)
|
||||
* 组合:title/summary + 章节结构 + 正文片段(~500 字符)
|
||||
*/
|
||||
private String buildCompactSummary(KnowledgeEntry entry) {
|
||||
String rawContent = knowledgeIndexService.readDocument(entry.getFilePath(), 2000);
|
||||
if (rawContent == null) return null;
|
||||
|
||||
// 跳过 YAML frontmatter 得到正文
|
||||
String body = rawContent;
|
||||
if (body.startsWith("---")) {
|
||||
int end = body.indexOf("---", 3);
|
||||
if (end != -1) {
|
||||
body = body.substring(end + 3).trim();
|
||||
}
|
||||
}
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
|
||||
// 1. 元数据头(始终包含)
|
||||
sb.append("文档: ").append(entry.getTitle()).append("\n");
|
||||
if (entry.getSummary() != null) {
|
||||
sb.append("摘要: ").append(entry.getSummary()).append("\n");
|
||||
}
|
||||
|
||||
// 2. 章节结构(## 标题列表)
|
||||
String headings = body.lines()
|
||||
.filter(l -> l.trim().startsWith("##"))
|
||||
.map(l -> " - " + l.trim().replaceAll("^#+\\s*", ""))
|
||||
.collect(Collectors.joining("\n"));
|
||||
if (!headings.isEmpty()) {
|
||||
sb.append("章节:\n").append(headings).append("\n");
|
||||
}
|
||||
sb.append("---\n");
|
||||
|
||||
// 3. 正文片段(去标题行、去空行,智能截断)
|
||||
String textContent = body.lines()
|
||||
.filter(l -> !l.trim().startsWith("#") && !l.trim().isEmpty())
|
||||
.collect(Collectors.joining("\n"))
|
||||
.trim();
|
||||
|
||||
// 短文档保留更多内容,长文档节省上下文
|
||||
int maxBodyChars = body.length() < 500 ? 800 : 500;
|
||||
if (textContent.length() > maxBodyChars) {
|
||||
sb.append(textContent, 0, maxBodyChars).append("...");
|
||||
} else {
|
||||
sb.append(textContent);
|
||||
}
|
||||
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* 构建纯元数据摘要(不读文件,仅用内存索引信息)
|
||||
* 多匹配且有 L1 补充时使用,L0 只需告知 LLM 命中了哪些文档
|
||||
*/
|
||||
private String buildMetadataOnlySummary(KnowledgeEntry entry) {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("文档: ").append(entry.getTitle()).append("\n");
|
||||
if (entry.getSummary() != null) {
|
||||
sb.append("摘要: ").append(entry.getSummary()).append("\n");
|
||||
}
|
||||
if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) {
|
||||
sb.append("关键词: ").append(String.join(", ", entry.getKeywords())).append("\n");
|
||||
}
|
||||
sb.append("来源: ").append(entry.getFilePath()).append("\n");
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* 提取 MD 内容中第一个有意义的文本行(跳过 frontmatter 和标题行)
|
||||
*/
|
||||
private String extractFirstMeaningfulLine(String content, int maxLen) {
|
||||
if (content == null || content.isBlank()) return "(空)";
|
||||
|
||||
String text = content.trim();
|
||||
// 跳过 YAML frontmatter (--- ... ---)
|
||||
if (text.startsWith("---")) {
|
||||
int end = text.indexOf("---", 3);
|
||||
if (end != -1) {
|
||||
text = text.substring(end + 3);
|
||||
}
|
||||
}
|
||||
|
||||
// 查找第一个非空、非标题行
|
||||
String[] lines = text.split("\n");
|
||||
for (String line : lines) {
|
||||
String tl = line.trim();
|
||||
if (!tl.isEmpty() && !tl.startsWith("#")) {
|
||||
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
|
||||
}
|
||||
}
|
||||
|
||||
// 兜底:第一行非空行
|
||||
for (String line : lines) {
|
||||
if (!line.trim().isEmpty()) {
|
||||
String tl = line.trim();
|
||||
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
|
||||
}
|
||||
}
|
||||
|
||||
return "(无有效内容)";
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user