Document the lookup_knowledge flow from L0 hints through L1 retrieval, post-processing, packing, and Agent projection so the boundaries and current limitations are easier to follow.
447 lines
16 KiB
Java
447 lines
16 KiB
Java
package com.superbiz.agent.service;
|
||
|
||
import com.superbiz.agent.config.DocumentChunkConfig;
|
||
import com.superbiz.agent.dto.DocumentChunk;
|
||
import org.slf4j.Logger;
|
||
import org.slf4j.LoggerFactory;
|
||
import org.springframework.beans.factory.annotation.Autowired;
|
||
import org.springframework.stereotype.Service;
|
||
|
||
import java.util.ArrayList;
|
||
import java.util.List;
|
||
import java.util.regex.Matcher;
|
||
import java.util.regex.Pattern;
|
||
|
||
/**
|
||
* 文档切片服务(RAG 入库前处理)。
|
||
*
|
||
* <p>把长 Markdown/文本切成带 title/breadcrumb 的 {@link com.superbiz.agent.dto.DocumentChunk},
|
||
* 供 {@link VectorIndexService} 向量化。</p>
|
||
*
|
||
* <h3>策略摘要</h3>
|
||
* <ol>
|
||
* <li>先按 Markdown 标题分 section,并维护 breadcrumb 层级</li>
|
||
* <li>section 过长再按段落累积;用 token 估算做软边界 / 硬上限</li>
|
||
* <li>尽量不在有序/无序列表或未闭合代码块中间切断</li>
|
||
* <li>相邻 chunk 保留 overlap,减轻边界语义断裂</li>
|
||
* </ol>
|
||
*
|
||
* <p>检索命中单个 chunk 后,当前主链路不会自动回补同章节相邻 chunk
|
||
* (上下文重建仍是后续增强点)。</p>
|
||
*/
|
||
@Service
|
||
public class DocumentChunkService {
|
||
|
||
private static final Logger logger = LoggerFactory.getLogger(DocumentChunkService.class);
|
||
|
||
@Autowired
|
||
private DocumentChunkConfig chunkConfig;
|
||
|
||
/**
|
||
* 智能分片文档
|
||
* 优先按照标题、段落边界进行分割,保持语义完整性
|
||
*
|
||
* @param content 文档内容
|
||
* @param filePath 文件路径(用于日志)
|
||
* @return 文档分片列表
|
||
*/
|
||
public List<DocumentChunk> chunkDocument(String content, String filePath) {
|
||
List<DocumentChunk> chunks = new ArrayList<>();
|
||
|
||
if (content == null || content.trim().isEmpty()) {
|
||
logger.warn("文档内容为空: {}", filePath);
|
||
return chunks;
|
||
}
|
||
|
||
// 1. 首先尝试按标题分割(Markdown格式)
|
||
List<Section> sections = splitByHeadings(content);
|
||
|
||
// 2. 对每个章节进行进一步分片
|
||
int globalChunkIndex = 0;
|
||
for (Section section : sections) {
|
||
List<DocumentChunk> sectionChunks = chunkSection(section, globalChunkIndex);
|
||
chunks.addAll(sectionChunks);
|
||
globalChunkIndex += sectionChunks.size();
|
||
}
|
||
|
||
logger.info("文档分片完成: {} -> {} 个分片", filePath, chunks.size());
|
||
return chunks;
|
||
}
|
||
|
||
/**
|
||
* 按照 Markdown 标题分割文档,同时构建面包屑层级路径
|
||
*/
|
||
private List<Section> splitByHeadings(String content) {
|
||
List<Section> sections = new ArrayList<>();
|
||
|
||
// 匹配 Markdown 标题:# 标题, ## 标题, ### 标题等
|
||
Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE);
|
||
Matcher matcher = headingPattern.matcher(content);
|
||
|
||
// 标题层级栈:维护当前标题的完整路径
|
||
List<String> headingStack = new ArrayList<>();
|
||
int lastEnd = 0;
|
||
String currentBreadcrumb = null;
|
||
|
||
while (matcher.find()) {
|
||
int level = matcher.group(1).length(); // #→1, ##→2, ###→3 ...
|
||
String title = matcher.group(2).trim();
|
||
|
||
// 保存上一个章节
|
||
if (lastEnd < matcher.start()) {
|
||
String sectionContent = content.substring(lastEnd, matcher.start()).trim();
|
||
if (!sectionContent.isEmpty()) {
|
||
sections.add(new Section(
|
||
headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1),
|
||
level,
|
||
currentBreadcrumb,
|
||
sectionContent,
|
||
lastEnd));
|
||
}
|
||
}
|
||
|
||
// 维护层级栈:同级别或更高级别 → 弹出,低级 → 追加
|
||
while (!headingStack.isEmpty() && headingStack.size() >= level) {
|
||
headingStack.remove(headingStack.size() - 1);
|
||
}
|
||
headingStack.add(title);
|
||
currentBreadcrumb = String.join(" > ", headingStack);
|
||
lastEnd = matcher.start();
|
||
}
|
||
|
||
// 添加最后一个章节
|
||
if (lastEnd < content.length()) {
|
||
String sectionContent = content.substring(lastEnd).trim();
|
||
if (!sectionContent.isEmpty()) {
|
||
sections.add(new Section(
|
||
headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1),
|
||
headingStack.size(),
|
||
currentBreadcrumb,
|
||
sectionContent,
|
||
lastEnd));
|
||
}
|
||
}
|
||
|
||
// 如果没有找到任何标题,将整个文档作为一个章节
|
||
if (sections.isEmpty()) {
|
||
sections.add(new Section(null, 0, null, content, 0));
|
||
}
|
||
|
||
return sections;
|
||
}
|
||
|
||
/**
|
||
* 对单个章节进行分片
|
||
* <p>
|
||
* 核心改造(Phase 1):
|
||
* - Token 估算替代字符计数
|
||
* - 感知有序/无序列表结构,不在列表中间切断
|
||
* - 软边界(maxTokens)+ 硬上限(maxTokensHard)双重控制
|
||
* - 修复 currentStartIndex 漂移:用段落原始位置而非手工推算
|
||
*/
|
||
private List<DocumentChunk> chunkSection(Section section, int startChunkIndex) {
|
||
List<DocumentChunk> chunks = new ArrayList<>();
|
||
String content = section.content;
|
||
String title = section.title;
|
||
String breadcrumb = section.breadcrumb;
|
||
|
||
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
|
||
if (content.length() <= chunkConfig.getMaxSize()
|
||
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) {
|
||
DocumentChunk chunk = DocumentChunk.builder()
|
||
.content(content)
|
||
.startOffset(section.startIndex)
|
||
.endOffset(section.startIndex + content.length())
|
||
.chunkIndex(startChunkIndex)
|
||
.title(title)
|
||
.breadcrumb(breadcrumb)
|
||
.build();
|
||
chunks.add(chunk);
|
||
return chunks;
|
||
}
|
||
|
||
// 章节内容较长,需要进一步分片
|
||
List<String> paragraphs = splitByParagraphs(content);
|
||
if (paragraphs.isEmpty()) {
|
||
return chunks;
|
||
}
|
||
|
||
// 定位每个段落在 section.content 中的位置(修复 index 漂移)
|
||
List<ParagraphPos> paraPositions = locateParagraphPositions(paragraphs, content);
|
||
|
||
// 当前分片的段落范围
|
||
int chunkParaStart = 0; // 当前分片第一个段落的索引(在 paragraphs 中)
|
||
StringBuilder buffer = new StringBuilder();
|
||
int tokenCount = 0;
|
||
int chunkIndex = startChunkIndex;
|
||
|
||
for (int i = 0; i < paragraphs.size(); i++) {
|
||
String paragraph = paragraphs.get(i);
|
||
int paraTokens = estimateTokens(paragraph);
|
||
|
||
// 判断是否需要切分
|
||
if (buffer.length() > 0 && tokenCount + paraTokens > chunkConfig.getMaxTokens()) {
|
||
|
||
// 检查是否处于不可中断的上下文中
|
||
if (isInUnbreakableContext(buffer.toString(), paragraph)) {
|
||
// 硬上限保护:即使不可中断也不能无限膨胀
|
||
if (tokenCount + paraTokens > chunkConfig.getMaxTokensHard()) {
|
||
logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens);
|
||
chunkParaStart = saveChunkAndGetNextStart(
|
||
chunks, section, paraPositions,
|
||
chunkParaStart, i, title, breadcrumb, chunkIndex);
|
||
chunkIndex++;
|
||
|
||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||
String overlap = getOverlapText(prevChunkContent);
|
||
buffer = new StringBuilder(overlap);
|
||
tokenCount = estimateTokens(overlap);
|
||
}
|
||
// 否则:容忍超出(软边界)
|
||
} else {
|
||
// 安全切点:段落边界
|
||
chunkParaStart = saveChunkAndGetNextStart(
|
||
chunks, section, paraPositions,
|
||
chunkParaStart, i, title, breadcrumb, chunkIndex);
|
||
chunkIndex++;
|
||
|
||
// 新分片以重叠文本开头
|
||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||
String overlap = getOverlapText(prevChunkContent);
|
||
buffer = new StringBuilder(overlap);
|
||
tokenCount = estimateTokens(overlap);
|
||
}
|
||
}
|
||
|
||
buffer.append(paragraph).append("\n\n");
|
||
tokenCount += paraTokens;
|
||
}
|
||
|
||
// 保存最后一个分片
|
||
if (buffer.length() > 0 && chunkParaStart < paragraphs.size()) {
|
||
String chunkContent = buffer.toString().trim();
|
||
int actualStart = paraPositions.get(chunkParaStart).start;
|
||
int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
|
||
DocumentChunk chunk = DocumentChunk.builder()
|
||
.content(chunkContent)
|
||
.startOffset(section.startIndex + actualStart)
|
||
.endOffset(section.startIndex + actualEnd)
|
||
.chunkIndex(chunkIndex)
|
||
.title(title)
|
||
.breadcrumb(breadcrumb)
|
||
.build();
|
||
chunks.add(chunk);
|
||
}
|
||
|
||
return chunks;
|
||
}
|
||
|
||
/**
|
||
* 保存当前分块,返回下一个分块的起始段落索引
|
||
* <p>
|
||
* 从 section.content 中提取原始文本(而非手工拼装),修复 index 漂移问题
|
||
*/
|
||
private int saveChunkAndGetNextStart(
|
||
List<DocumentChunk> chunks,
|
||
Section section,
|
||
List<ParagraphPos> paraPositions,
|
||
int fromPara,
|
||
int toPara,
|
||
String title,
|
||
String breadcrumb,
|
||
int chunkIndex) {
|
||
|
||
int actualStart = paraPositions.get(fromPara).start;
|
||
int actualEnd = paraPositions.get(toPara - 1).end;
|
||
String originalText = section.content.substring(actualStart, actualEnd);
|
||
|
||
DocumentChunk chunk = DocumentChunk.builder()
|
||
.content(originalText)
|
||
.startOffset(section.startIndex + actualStart)
|
||
.endOffset(section.startIndex + actualEnd)
|
||
.chunkIndex(chunkIndex)
|
||
.title(title)
|
||
.breadcrumb(breadcrumb)
|
||
.build();
|
||
chunks.add(chunk);
|
||
|
||
return toPara; // 下一个分块的起始段落索引
|
||
}
|
||
|
||
/**
|
||
* 按段落分割文本
|
||
*/
|
||
private List<String> splitByParagraphs(String content) {
|
||
List<String> paragraphs = new ArrayList<>();
|
||
|
||
// 按双换行符分割段落
|
||
String[] parts = content.split("\n\n+");
|
||
for (String part : parts) {
|
||
String trimmed = part.trim();
|
||
if (!trimmed.isEmpty()) {
|
||
paragraphs.add(trimmed);
|
||
}
|
||
}
|
||
|
||
return paragraphs;
|
||
}
|
||
|
||
/**
|
||
* 定位每个段落在原始文本中的字符偏移
|
||
*/
|
||
private List<ParagraphPos> locateParagraphPositions(List<String> paragraphs, String sectionContent) {
|
||
List<ParagraphPos> positions = new ArrayList<>();
|
||
int searchFrom = 0;
|
||
for (String p : paragraphs) {
|
||
int idx = sectionContent.indexOf(p, searchFrom);
|
||
if (idx >= 0) {
|
||
positions.add(new ParagraphPos(idx, idx + p.length()));
|
||
searchFrom = idx + p.length();
|
||
} else {
|
||
// fallback: 段落在原文中找不到(不应该发生)
|
||
positions.add(new ParagraphPos(searchFrom, searchFrom + p.length()));
|
||
searchFrom += p.length();
|
||
}
|
||
}
|
||
return positions;
|
||
}
|
||
|
||
/**
|
||
* 启发式 token 估算(无需外部依赖)
|
||
* <p>
|
||
* 中文(BMP): ~1 字符/token
|
||
* 英文/数字/标点: ~4 字符/token
|
||
* 空白字符忽略
|
||
*/
|
||
private int estimateTokens(String text) {
|
||
int nonCjkCount = 0;
|
||
int cjkCount = 0;
|
||
for (char c : text.toCharArray()) {
|
||
if (Character.isWhitespace(c)) {
|
||
continue;
|
||
}
|
||
Character.UnicodeBlock block = Character.UnicodeBlock.of(c);
|
||
if (block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS
|
||
|| block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A
|
||
|| block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B
|
||
|| block == Character.UnicodeBlock.CJK_COMPATIBILITY_IDEOGRAPHS) {
|
||
cjkCount++;
|
||
} else {
|
||
nonCjkCount++;
|
||
}
|
||
}
|
||
return cjkCount + (nonCjkCount + 3) / 4; // 非中文每 4 字符算 1 token,向上取整
|
||
}
|
||
|
||
/**
|
||
* 判断当前段落是否属于不可中断的结构
|
||
* <p>
|
||
* 不可中断结构包括:
|
||
* - 有序列表项("1. ", "2. " 格式)
|
||
* - 无序列表项("- " 或 "* " 格式)
|
||
* - 未闭合的代码块(``` 内)
|
||
*/
|
||
private boolean isInUnbreakableContext(String buffer, String nextParagraph) {
|
||
// 有序列表:判断 buffer 末尾和下一段是否都是列表项
|
||
if (nextParagraph.matches("^\\d{1,2}\\.\\s.*")) {
|
||
String lastLine = getLastNonEmptyLine(buffer);
|
||
if (lastLine != null && lastLine.matches("^\\d{1,2}\\.\\s.*")) {
|
||
return true;
|
||
}
|
||
}
|
||
// 无序列表:"- " 或 "* " 格式
|
||
if (nextParagraph.matches("^[-*]\\s.*")) {
|
||
String lastLine = getLastNonEmptyLine(buffer);
|
||
if (lastLine != null && lastLine.matches("^[-*]\\s.*")) {
|
||
return true;
|
||
}
|
||
}
|
||
// 代码块:``` 未闭合
|
||
if (buffer.contains("```")) {
|
||
int count = 0;
|
||
for (int i = 0; i <= buffer.length() - 3; i++) {
|
||
if (buffer.substring(i).startsWith("```")) {
|
||
count++;
|
||
i += 2;
|
||
}
|
||
}
|
||
if (count % 2 == 1) {
|
||
return true; // 奇数个 ``` → 在代码块内部
|
||
}
|
||
}
|
||
return false;
|
||
}
|
||
|
||
/**
|
||
* 获取 buffer 中最后一行非空白文本
|
||
*/
|
||
private String getLastNonEmptyLine(String buffer) {
|
||
String[] lines = buffer.split("\n");
|
||
for (int i = lines.length - 1; i >= 0; i--) {
|
||
String line = lines[i].trim();
|
||
if (!line.isEmpty()) {
|
||
return line;
|
||
}
|
||
}
|
||
return null;
|
||
}
|
||
|
||
/**
|
||
* 获取重叠文本
|
||
* 从文本末尾提取指定长度的内容作为下一个分片的开头
|
||
*/
|
||
private String getOverlapText(String text) {
|
||
int overlapSize = Math.min(chunkConfig.getOverlap(), text.length());
|
||
if (overlapSize <= 0) {
|
||
return "";
|
||
}
|
||
|
||
// 从末尾提取重叠内容
|
||
String overlap = text.substring(text.length() - overlapSize);
|
||
|
||
// 尝试在句子边界截断(查找最后一个句号、问号、感叹号)
|
||
int lastSentenceEnd = Math.max(
|
||
overlap.lastIndexOf('。'),
|
||
Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!'))
|
||
);
|
||
|
||
if (lastSentenceEnd > overlapSize / 2) {
|
||
return overlap.substring(lastSentenceEnd + 1).trim();
|
||
}
|
||
|
||
return overlap.trim();
|
||
}
|
||
|
||
/**
|
||
* 段落在原文中的位置
|
||
*/
|
||
private static class ParagraphPos {
|
||
final int start;
|
||
final int end;
|
||
|
||
ParagraphPos(int start, int end) {
|
||
this.start = start;
|
||
this.end = end;
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 章节数据类
|
||
*/
|
||
private static class Section {
|
||
String title; // 最近一级标题名称
|
||
int level; // 标题级别(1-6),0=无标题
|
||
String breadcrumb; // 完整面包屑路径
|
||
String content; // 章节内容
|
||
int startIndex; // 在原文中的起始偏移
|
||
|
||
Section(String title, int level, String breadcrumb, String content, int startIndex) {
|
||
this.title = title;
|
||
this.level = level;
|
||
this.breadcrumb = breadcrumb;
|
||
this.content = content;
|
||
this.startIndex = startIndex;
|
||
}
|
||
}
|
||
}
|