feat(knowledge): breadcrumb分块上下文 & LookupKnowledgeTool日志优化

- DocumentChunk新增breadcrumb字段,分块时构建完整标题层级路径
- DocumentChunkService splitByHeadings维护标题层级栈算法
- VectorIndexService 将breadcrumb写入Milvus metadata
- LookupKnowledgeTool日志替换为结构化摘要,替代原始MD预览
- L0返回策略:唯一匹配用正文摘要,多匹配+L1有结果仅元数据(不读文件)
- 新增buildCompactSummary / buildMetadataOnlySummary方法
- 安装frontend-design skill
- 创建mvp/文档目录(架构设计+会话存储方案)
- 更新测试适配新逻辑
This commit is contained in:
zhuyongxin
2026-06-26 13:56:10 +08:00
parent a1876286fd
commit a74ccea5be
11 changed files with 1599 additions and 378 deletions
@@ -38,4 +38,10 @@ public class DocumentChunk {
* 分片标题或上下文信息
*/
private String title;
/**
* 面包屑导航(完整标题层级路径)
* 例如: "故障诊断流程规范 > 应急响应流程 > 1. 初步评估"
*/
private String breadcrumb;
}
@@ -56,7 +56,7 @@ public class DocumentChunkService {
}
/**
* 按照 Markdown 标题分割文档
* 按照 Markdown 标题分割文档,同时构建面包屑层级路径
*/
private List<Section> splitByHeadings(String content) {
List<Section> sections = new ArrayList<>();
@@ -65,20 +65,34 @@ public class DocumentChunkService {
Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE);
Matcher matcher = headingPattern.matcher(content);
// 标题层级栈:维护当前标题的完整路径
List<String> headingStack = new ArrayList<>();
int lastEnd = 0;
String currentTitle = null;
String currentBreadcrumb = null;
while (matcher.find()) {
int level = matcher.group(1).length(); // #→1, ##→2, ###→3 ...
String title = matcher.group(2).trim();
// 保存上一个章节
if (lastEnd < matcher.start()) {
String sectionContent = content.substring(lastEnd, matcher.start()).trim();
if (!sectionContent.isEmpty()) {
sections.add(new Section(currentTitle, sectionContent, lastEnd));
sections.add(new Section(
headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1),
level,
currentBreadcrumb,
sectionContent,
lastEnd));
}
}
// 更新当前标题
currentTitle = matcher.group(2).trim();
// 维护层级栈:同级别或更高级别 → 弹出,低级 → 追加
while (!headingStack.isEmpty() && headingStack.size() >= level) {
headingStack.remove(headingStack.size() - 1);
}
headingStack.add(title);
currentBreadcrumb = String.join(" > ", headingStack);
lastEnd = matcher.start();
}
@@ -86,13 +100,18 @@ public class DocumentChunkService {
if (lastEnd < content.length()) {
String sectionContent = content.substring(lastEnd).trim();
if (!sectionContent.isEmpty()) {
sections.add(new Section(currentTitle, sectionContent, lastEnd));
sections.add(new Section(
headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1),
headingStack.size(),
currentBreadcrumb,
sectionContent,
lastEnd));
}
}
// 如果没有找到任何标题,将整个文档作为一个章节
if (sections.isEmpty()) {
sections.add(new Section(null, content, 0));
sections.add(new Section(null, 0, null, content, 0));
}
return sections;
@@ -111,6 +130,7 @@ public class DocumentChunkService {
List<DocumentChunk> chunks = new ArrayList<>();
String content = section.content;
String title = section.title;
String breadcrumb = section.breadcrumb;
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
if (content.length() <= chunkConfig.getMaxSize()
@@ -121,6 +141,7 @@ public class DocumentChunkService {
.endOffset(section.startIndex + content.length())
.chunkIndex(startChunkIndex)
.title(title)
.breadcrumb(breadcrumb)
.build();
chunks.add(chunk);
return chunks;
@@ -155,7 +176,7 @@ public class DocumentChunkService {
logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens);
chunkParaStart = saveChunkAndGetNextStart(
chunks, section, paraPositions,
chunkParaStart, i, title, chunkIndex);
chunkParaStart, i, title, breadcrumb, chunkIndex);
chunkIndex++;
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
@@ -168,7 +189,7 @@ public class DocumentChunkService {
// 安全切点:段落边界
chunkParaStart = saveChunkAndGetNextStart(
chunks, section, paraPositions,
chunkParaStart, i, title, chunkIndex);
chunkParaStart, i, title, breadcrumb, chunkIndex);
chunkIndex++;
// 新分片以重叠文本开头
@@ -194,6 +215,7 @@ public class DocumentChunkService {
.endOffset(section.startIndex + actualEnd)
.chunkIndex(chunkIndex)
.title(title)
.breadcrumb(breadcrumb)
.build();
chunks.add(chunk);
}
@@ -213,6 +235,7 @@ public class DocumentChunkService {
int fromPara,
int toPara,
String title,
String breadcrumb,
int chunkIndex) {
int actualStart = paraPositions.get(fromPara).start;
@@ -225,6 +248,7 @@ public class DocumentChunkService {
.endOffset(section.startIndex + actualEnd)
.chunkIndex(chunkIndex)
.title(title)
.breadcrumb(breadcrumb)
.build();
chunks.add(chunk);
@@ -392,12 +416,16 @@ public class DocumentChunkService {
* 章节数据类
*/
private static class Section {
String title;
String content;
int startIndex;
String title; // 最近一级标题名称
int level; // 标题级别(1-6),0=无标题
String breadcrumb; // 完整面包屑路径
String content; // 章节内容
int startIndex; // 在原文中的起始偏移
Section(String title, String content, int startIndex) {
Section(String title, int level, String breadcrumb, String content, int startIndex) {
this.title = title;
this.level = level;
this.breadcrumb = breadcrumb;
this.content = content;
this.startIndex = startIndex;
}
@@ -269,6 +269,11 @@ public class VectorIndexService {
metadata.put("title", chunk.getTitle());
}
// 面包屑导航(完整标题层级路径)
if (chunk.getBreadcrumb() != null && !chunk.getBreadcrumb().isEmpty()) {
metadata.put("breadcrumb", chunk.getBreadcrumb());
}
// 文档类别
metadata.put("category", category != null && !category.isBlank() ? category : "upload");
@@ -360,6 +365,11 @@ public class VectorIndexService {
metadata.put("title", chunk.getTitle());
}
// 面包屑导航(完整标题层级路径)
if (chunk.getBreadcrumb() != null && !chunk.getBreadcrumb().isEmpty()) {
metadata.put("breadcrumb", chunk.getBreadcrumb());
}
return metadata;
}
@@ -9,6 +9,7 @@ import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import java.util.List;
import java.util.stream.Collectors;
/**
* 知识库查询工具
@@ -91,23 +92,46 @@ public class LookupKnowledgeTool {
// Step 4: 组装结果
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
// 记录完整结果
// 记录结构化结果摘要(替代原始 MD 内容预览)
long totalTime = System.currentTimeMillis() - startTime;
log.info("----------------------------------------");
log.info("<<< [工具返回] lookup_knowledge");
log.info("<<< 结果: found={}, matchType={}, confidence={}",
result.isFound(),
result.getPrimary() != null ? result.getPrimary().getMatchType() : "N/A",
result.getPrimary() != null ? result.getPrimary().getConfidence() : "N/A");
log.info("<<< 总耗时: {}ms (L0={}ms, L1={}ms)",
totalTime, l0Time, l1Results != null ? (totalTime - l0Time) : 0);
if (result.isFound() && result.getPrimary() != null) {
String content = result.getPrimary().getContent();
log.info("<<< 返回内容长度: {} 字符", content != null ? content.length() : 0);
if (content != null && content.length() > 200) {
log.info("<<< 内容预览: {}", content.substring(0, 200) + "...");
log.info("<<< 结果: found={}, 耗时: {}ms (L0={}ms, L1={}ms)",
result.isFound(), totalTime, l0Time,
l1Results != null ? System.currentTimeMillis() - startTime - l0Time : 0);
// L0 精确匹配摘要
if (!l0Matches.isEmpty()) {
KnowledgeEntry top = l0Matches.get(0);
log.info("<<< [L0 主结果] 标题: {}", top.getTitle());
log.info("<<< [L0 主结果] 来源: {}", top.getFilePath());
if (top.getSummary() != null) {
log.info("<<< [L0 主结果] 摘要: {}", top.getSummary());
}
if (top.getKeywords() != null && !top.getKeywords().isEmpty()) {
log.info("<<< [L0 主结果] 关键词: {}", String.join(", ", top.getKeywords()));
}
// 内容概况:长度 + 章节数
String content = result.getPrimary() != null ? result.getPrimary().getContent() : null;
if (content != null) {
int headingCount = countMdHeadings(content);
log.info("<<< [L0 主结果] 内容: {} 字符, {} 个章节",
content.length(), headingCount);
}
}
// L1 语义检索摘要
if (l1Results != null && !l1Results.isEmpty()) {
VectorSearchService.SearchResult topL1 = l1Results.get(0);
log.info("<<< [L1 补充] 来源: {}", topL1.getMetadata() != null ? topL1.getMetadata() : topL1.getId());
log.info("<<< [L1 补充] 相似度: {}", String.format("%.4f", topL1.getScore()));
if (topL1.getContent() != null) {
String snippet = extractFirstMeaningfulLine(topL1.getContent(), 120);
log.info("<<< [L1 补充] 内容片段: {}", snippet);
log.info("<<< [L1 补充] 片段长度: {} 字符", topL1.getContent().length());
}
}
log.info("========================================");
return result;
@@ -132,7 +156,13 @@ public class LookupKnowledgeTool {
PrimaryResult primary = null;
if (l0Matches != null && !l0Matches.isEmpty()) {
KnowledgeEntry first = l0Matches.get(0);
String content = knowledgeIndexService.readDocument(first.getFilePath(), 2000);
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
// 场景决策:唯一匹配或 L1 无结果 → LLM 需要正文内容;多匹配且有 L1 → 只需元数据
boolean needFullContent = highConfidence || !hasL1;
String content = needFullContent
? buildCompactSummary(first)
: buildMetadataOnlySummary(first);
if (content != null) {
primary = PrimaryResult.builder()
@@ -169,4 +199,118 @@ public class LookupKnowledgeTool {
return builder.build();
}
/**
* 统计 MD 文档中的章节数(二级标题 ## 数量)
*/
private int countMdHeadings(String content) {
if (content == null) return 0;
return (int) content.lines()
.filter(l -> l.trim().startsWith("##"))
.count();
}
/**
* 构建紧凑文档摘要(替代原始 MD 全文,节省上下文窗口)
* 组合:title/summary + 章节结构 + 正文片段(~500 字符)
*/
private String buildCompactSummary(KnowledgeEntry entry) {
String rawContent = knowledgeIndexService.readDocument(entry.getFilePath(), 2000);
if (rawContent == null) return null;
// 跳过 YAML frontmatter 得到正文
String body = rawContent;
if (body.startsWith("---")) {
int end = body.indexOf("---", 3);
if (end != -1) {
body = body.substring(end + 3).trim();
}
}
StringBuilder sb = new StringBuilder();
// 1. 元数据头(始终包含)
sb.append("文档: ").append(entry.getTitle()).append("\n");
if (entry.getSummary() != null) {
sb.append("摘要: ").append(entry.getSummary()).append("\n");
}
// 2. 章节结构(## 标题列表)
String headings = body.lines()
.filter(l -> l.trim().startsWith("##"))
.map(l -> " - " + l.trim().replaceAll("^#+\\s*", ""))
.collect(Collectors.joining("\n"));
if (!headings.isEmpty()) {
sb.append("章节:\n").append(headings).append("\n");
}
sb.append("---\n");
// 3. 正文片段(去标题行、去空行,智能截断)
String textContent = body.lines()
.filter(l -> !l.trim().startsWith("#") && !l.trim().isEmpty())
.collect(Collectors.joining("\n"))
.trim();
// 短文档保留更多内容,长文档节省上下文
int maxBodyChars = body.length() < 500 ? 800 : 500;
if (textContent.length() > maxBodyChars) {
sb.append(textContent, 0, maxBodyChars).append("...");
} else {
sb.append(textContent);
}
return sb.toString();
}
/**
* 构建纯元数据摘要(不读文件,仅用内存索引信息)
* 多匹配且有 L1 补充时使用,L0 只需告知 LLM 命中了哪些文档
*/
private String buildMetadataOnlySummary(KnowledgeEntry entry) {
StringBuilder sb = new StringBuilder();
sb.append("文档: ").append(entry.getTitle()).append("\n");
if (entry.getSummary() != null) {
sb.append("摘要: ").append(entry.getSummary()).append("\n");
}
if (entry.getKeywords() != null && !entry.getKeywords().isEmpty()) {
sb.append("关键词: ").append(String.join(", ", entry.getKeywords())).append("\n");
}
sb.append("来源: ").append(entry.getFilePath()).append("\n");
return sb.toString();
}
/**
* 提取 MD 内容中第一个有意义的文本行(跳过 frontmatter 和标题行)
*/
private String extractFirstMeaningfulLine(String content, int maxLen) {
if (content == null || content.isBlank()) return "(空)";
String text = content.trim();
// 跳过 YAML frontmatter (--- ... ---)
if (text.startsWith("---")) {
int end = text.indexOf("---", 3);
if (end != -1) {
text = text.substring(end + 3);
}
}
// 查找第一个非空、非标题行
String[] lines = text.split("\n");
for (String line : lines) {
String tl = line.trim();
if (!tl.isEmpty() && !tl.startsWith("#")) {
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
}
}
// 兜底:第一行非空行
for (String line : lines) {
if (!line.trim().isEmpty()) {
String tl = line.trim();
return tl.length() <= maxLen ? tl : tl.substring(0, maxLen) + "...";
}
}
return "(无有效内容)";
}
}