feat(knowledge): breadcrumb分块上下文 & LookupKnowledgeTool日志优化
- DocumentChunk新增breadcrumb字段,分块时构建完整标题层级路径 - DocumentChunkService splitByHeadings维护标题层级栈算法 - VectorIndexService 将breadcrumb写入Milvus metadata - LookupKnowledgeTool日志替换为结构化摘要,替代原始MD预览 - L0返回策略:唯一匹配用正文摘要,多匹配+L1有结果仅元数据(不读文件) - 新增buildCompactSummary / buildMetadataOnlySummary方法 - 安装frontend-design skill - 创建mvp/文档目录(架构设计+会话存储方案) - 更新测试适配新逻辑
This commit is contained in:
@@ -56,7 +56,7 @@ public class DocumentChunkService {
|
||||
}
|
||||
|
||||
/**
|
||||
* 按照 Markdown 标题分割文档
|
||||
* 按照 Markdown 标题分割文档,同时构建面包屑层级路径
|
||||
*/
|
||||
private List<Section> splitByHeadings(String content) {
|
||||
List<Section> sections = new ArrayList<>();
|
||||
@@ -65,20 +65,34 @@ public class DocumentChunkService {
|
||||
Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE);
|
||||
Matcher matcher = headingPattern.matcher(content);
|
||||
|
||||
// 标题层级栈:维护当前标题的完整路径
|
||||
List<String> headingStack = new ArrayList<>();
|
||||
int lastEnd = 0;
|
||||
String currentTitle = null;
|
||||
String currentBreadcrumb = null;
|
||||
|
||||
while (matcher.find()) {
|
||||
int level = matcher.group(1).length(); // #→1, ##→2, ###→3 ...
|
||||
String title = matcher.group(2).trim();
|
||||
|
||||
// 保存上一个章节
|
||||
if (lastEnd < matcher.start()) {
|
||||
String sectionContent = content.substring(lastEnd, matcher.start()).trim();
|
||||
if (!sectionContent.isEmpty()) {
|
||||
sections.add(new Section(currentTitle, sectionContent, lastEnd));
|
||||
sections.add(new Section(
|
||||
headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1),
|
||||
level,
|
||||
currentBreadcrumb,
|
||||
sectionContent,
|
||||
lastEnd));
|
||||
}
|
||||
}
|
||||
|
||||
// 更新当前标题
|
||||
currentTitle = matcher.group(2).trim();
|
||||
// 维护层级栈:同级别或更高级别 → 弹出,低级 → 追加
|
||||
while (!headingStack.isEmpty() && headingStack.size() >= level) {
|
||||
headingStack.remove(headingStack.size() - 1);
|
||||
}
|
||||
headingStack.add(title);
|
||||
currentBreadcrumb = String.join(" > ", headingStack);
|
||||
lastEnd = matcher.start();
|
||||
}
|
||||
|
||||
@@ -86,13 +100,18 @@ public class DocumentChunkService {
|
||||
if (lastEnd < content.length()) {
|
||||
String sectionContent = content.substring(lastEnd).trim();
|
||||
if (!sectionContent.isEmpty()) {
|
||||
sections.add(new Section(currentTitle, sectionContent, lastEnd));
|
||||
sections.add(new Section(
|
||||
headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1),
|
||||
headingStack.size(),
|
||||
currentBreadcrumb,
|
||||
sectionContent,
|
||||
lastEnd));
|
||||
}
|
||||
}
|
||||
|
||||
// 如果没有找到任何标题,将整个文档作为一个章节
|
||||
if (sections.isEmpty()) {
|
||||
sections.add(new Section(null, content, 0));
|
||||
sections.add(new Section(null, 0, null, content, 0));
|
||||
}
|
||||
|
||||
return sections;
|
||||
@@ -111,6 +130,7 @@ public class DocumentChunkService {
|
||||
List<DocumentChunk> chunks = new ArrayList<>();
|
||||
String content = section.content;
|
||||
String title = section.title;
|
||||
String breadcrumb = section.breadcrumb;
|
||||
|
||||
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
|
||||
if (content.length() <= chunkConfig.getMaxSize()
|
||||
@@ -121,6 +141,7 @@ public class DocumentChunkService {
|
||||
.endOffset(section.startIndex + content.length())
|
||||
.chunkIndex(startChunkIndex)
|
||||
.title(title)
|
||||
.breadcrumb(breadcrumb)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
return chunks;
|
||||
@@ -155,7 +176,7 @@ public class DocumentChunkService {
|
||||
logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens);
|
||||
chunkParaStart = saveChunkAndGetNextStart(
|
||||
chunks, section, paraPositions,
|
||||
chunkParaStart, i, title, chunkIndex);
|
||||
chunkParaStart, i, title, breadcrumb, chunkIndex);
|
||||
chunkIndex++;
|
||||
|
||||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||||
@@ -168,7 +189,7 @@ public class DocumentChunkService {
|
||||
// 安全切点:段落边界
|
||||
chunkParaStart = saveChunkAndGetNextStart(
|
||||
chunks, section, paraPositions,
|
||||
chunkParaStart, i, title, chunkIndex);
|
||||
chunkParaStart, i, title, breadcrumb, chunkIndex);
|
||||
chunkIndex++;
|
||||
|
||||
// 新分片以重叠文本开头
|
||||
@@ -194,6 +215,7 @@ public class DocumentChunkService {
|
||||
.endOffset(section.startIndex + actualEnd)
|
||||
.chunkIndex(chunkIndex)
|
||||
.title(title)
|
||||
.breadcrumb(breadcrumb)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
}
|
||||
@@ -213,6 +235,7 @@ public class DocumentChunkService {
|
||||
int fromPara,
|
||||
int toPara,
|
||||
String title,
|
||||
String breadcrumb,
|
||||
int chunkIndex) {
|
||||
|
||||
int actualStart = paraPositions.get(fromPara).start;
|
||||
@@ -225,6 +248,7 @@ public class DocumentChunkService {
|
||||
.endOffset(section.startIndex + actualEnd)
|
||||
.chunkIndex(chunkIndex)
|
||||
.title(title)
|
||||
.breadcrumb(breadcrumb)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
|
||||
@@ -392,12 +416,16 @@ public class DocumentChunkService {
|
||||
* 章节数据类
|
||||
*/
|
||||
private static class Section {
|
||||
String title;
|
||||
String content;
|
||||
int startIndex;
|
||||
String title; // 最近一级标题名称
|
||||
int level; // 标题级别(1-6),0=无标题
|
||||
String breadcrumb; // 完整面包屑路径
|
||||
String content; // 章节内容
|
||||
int startIndex; // 在原文中的起始偏移
|
||||
|
||||
Section(String title, String content, int startIndex) {
|
||||
Section(String title, int level, String breadcrumb, String content, int startIndex) {
|
||||
this.title = title;
|
||||
this.level = level;
|
||||
this.breadcrumb = breadcrumb;
|
||||
this.content = content;
|
||||
this.startIndex = startIndex;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user