package com.superbiz.agent.service; import com.superbiz.agent.config.DocumentChunkConfig; import com.superbiz.agent.dto.DocumentChunk; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.springframework.beans.factory.annotation.Autowired; import org.springframework.stereotype.Service; import java.util.ArrayList; import java.util.List; import java.util.regex.Matcher; import java.util.regex.Pattern; /** * 文档切片服务(RAG 入库前处理)。 * *

把长 Markdown/文本切成带 title/breadcrumb 的 {@link com.superbiz.agent.dto.DocumentChunk}, * 供 {@link VectorIndexService} 向量化。

* *

策略摘要

*
    *
  1. 先按 Markdown 标题分 section,并维护 breadcrumb 层级
  2. *
  3. section 过长再按段落累积;用 token 估算做软边界 / 硬上限
  4. *
  5. 尽量不在有序/无序列表或未闭合代码块中间切断
  6. *
  7. 相邻 chunk 保留 overlap,减轻边界语义断裂
  8. *
* *

检索命中单个 chunk 后,当前主链路不会自动回补同章节相邻 chunk * (上下文重建仍是后续增强点)。

*/ @Service public class DocumentChunkService { private static final Logger logger = LoggerFactory.getLogger(DocumentChunkService.class); @Autowired private DocumentChunkConfig chunkConfig; /** * 智能分片文档 * 优先按照标题、段落边界进行分割,保持语义完整性 * * @param content 文档内容 * @param filePath 文件路径(用于日志) * @return 文档分片列表 */ public List chunkDocument(String content, String filePath) { List chunks = new ArrayList<>(); if (content == null || content.trim().isEmpty()) { logger.warn("文档内容为空: {}", filePath); return chunks; } // 1. 首先尝试按标题分割(Markdown格式) List
sections = splitByHeadings(content); // 2. 对每个章节进行进一步分片 int globalChunkIndex = 0; for (Section section : sections) { List sectionChunks = chunkSection(section, globalChunkIndex); chunks.addAll(sectionChunks); globalChunkIndex += sectionChunks.size(); } logger.info("文档分片完成: {} -> {} 个分片", filePath, chunks.size()); return chunks; } /** * 按照 Markdown 标题分割文档,同时构建面包屑层级路径 */ private List
splitByHeadings(String content) { List
sections = new ArrayList<>(); // 匹配 Markdown 标题:# 标题, ## 标题, ### 标题等 Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE); Matcher matcher = headingPattern.matcher(content); // 标题层级栈:维护当前标题的完整路径 List headingStack = new ArrayList<>(); int lastEnd = 0; String currentBreadcrumb = null; while (matcher.find()) { int level = matcher.group(1).length(); // #→1, ##→2, ###→3 ... String title = matcher.group(2).trim(); // 保存上一个章节 if (lastEnd < matcher.start()) { String sectionContent = content.substring(lastEnd, matcher.start()).trim(); if (!sectionContent.isEmpty()) { sections.add(new Section( headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1), level, currentBreadcrumb, sectionContent, lastEnd)); } } // 维护层级栈:同级别或更高级别 → 弹出,低级 → 追加 while (!headingStack.isEmpty() && headingStack.size() >= level) { headingStack.remove(headingStack.size() - 1); } headingStack.add(title); currentBreadcrumb = String.join(" > ", headingStack); lastEnd = matcher.start(); } // 添加最后一个章节 if (lastEnd < content.length()) { String sectionContent = content.substring(lastEnd).trim(); if (!sectionContent.isEmpty()) { sections.add(new Section( headingStack.isEmpty() ? null : headingStack.get(headingStack.size() - 1), headingStack.size(), currentBreadcrumb, sectionContent, lastEnd)); } } // 如果没有找到任何标题,将整个文档作为一个章节 if (sections.isEmpty()) { sections.add(new Section(null, 0, null, content, 0)); } return sections; } /** * 对单个章节进行分片 *

* 核心改造(Phase 1): * - Token 估算替代字符计数 * - 感知有序/无序列表结构,不在列表中间切断 * - 软边界(maxTokens)+ 硬上限(maxTokensHard)双重控制 * - 修复 currentStartIndex 漂移:用段落原始位置而非手工推算 */ private List chunkSection(Section section, int startChunkIndex) { List chunks = new ArrayList<>(); String content = section.content; String title = section.title; String breadcrumb = section.breadcrumb; // 短章节直接作为一个分片(用 token 估算替代字符数做短路判断) if (content.length() <= chunkConfig.getMaxSize() && estimateTokens(content) <= chunkConfig.getMaxTokens()) { DocumentChunk chunk = DocumentChunk.builder() .content(content) .startOffset(section.startIndex) .endOffset(section.startIndex + content.length()) .chunkIndex(startChunkIndex) .title(title) .breadcrumb(breadcrumb) .build(); chunks.add(chunk); return chunks; } // 章节内容较长,需要进一步分片 List paragraphs = splitByParagraphs(content); if (paragraphs.isEmpty()) { return chunks; } // 定位每个段落在 section.content 中的位置(修复 index 漂移) List paraPositions = locateParagraphPositions(paragraphs, content); // 当前分片的段落范围 int chunkParaStart = 0; // 当前分片第一个段落的索引(在 paragraphs 中) StringBuilder buffer = new StringBuilder(); int tokenCount = 0; int chunkIndex = startChunkIndex; for (int i = 0; i < paragraphs.size(); i++) { String paragraph = paragraphs.get(i); int paraTokens = estimateTokens(paragraph); // 判断是否需要切分 if (buffer.length() > 0 && tokenCount + paraTokens > chunkConfig.getMaxTokens()) { // 检查是否处于不可中断的上下文中 if (isInUnbreakableContext(buffer.toString(), paragraph)) { // 硬上限保护:即使不可中断也不能无限膨胀 if (tokenCount + paraTokens > chunkConfig.getMaxTokensHard()) { logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens); chunkParaStart = saveChunkAndGetNextStart( chunks, section, paraPositions, chunkParaStart, i, title, breadcrumb, chunkIndex); chunkIndex++; String prevChunkContent = chunks.get(chunks.size() - 1).getContent(); String overlap = getOverlapText(prevChunkContent); buffer = new StringBuilder(overlap); tokenCount = estimateTokens(overlap); } // 否则:容忍超出(软边界) } else { // 安全切点:段落边界 chunkParaStart = saveChunkAndGetNextStart( chunks, section, paraPositions, chunkParaStart, i, title, breadcrumb, chunkIndex); chunkIndex++; // 新分片以重叠文本开头 String prevChunkContent = chunks.get(chunks.size() - 1).getContent(); String overlap = getOverlapText(prevChunkContent); buffer = new StringBuilder(overlap); tokenCount = estimateTokens(overlap); } } buffer.append(paragraph).append("\n\n"); tokenCount += paraTokens; } // 保存最后一个分片 if (buffer.length() > 0 && chunkParaStart < paragraphs.size()) { String chunkContent = buffer.toString().trim(); int actualStart = paraPositions.get(chunkParaStart).start; int actualEnd = paraPositions.get(paragraphs.size() - 1).end; DocumentChunk chunk = DocumentChunk.builder() .content(chunkContent) .startOffset(section.startIndex + actualStart) .endOffset(section.startIndex + actualEnd) .chunkIndex(chunkIndex) .title(title) .breadcrumb(breadcrumb) .build(); chunks.add(chunk); } return chunks; } /** * 保存当前分块,返回下一个分块的起始段落索引 *

* 从 section.content 中提取原始文本(而非手工拼装),修复 index 漂移问题 */ private int saveChunkAndGetNextStart( List chunks, Section section, List paraPositions, int fromPara, int toPara, String title, String breadcrumb, int chunkIndex) { int actualStart = paraPositions.get(fromPara).start; int actualEnd = paraPositions.get(toPara - 1).end; String originalText = section.content.substring(actualStart, actualEnd); DocumentChunk chunk = DocumentChunk.builder() .content(originalText) .startOffset(section.startIndex + actualStart) .endOffset(section.startIndex + actualEnd) .chunkIndex(chunkIndex) .title(title) .breadcrumb(breadcrumb) .build(); chunks.add(chunk); return toPara; // 下一个分块的起始段落索引 } /** * 按段落分割文本 */ private List splitByParagraphs(String content) { List paragraphs = new ArrayList<>(); // 按双换行符分割段落 String[] parts = content.split("\n\n+"); for (String part : parts) { String trimmed = part.trim(); if (!trimmed.isEmpty()) { paragraphs.add(trimmed); } } return paragraphs; } /** * 定位每个段落在原始文本中的字符偏移 */ private List locateParagraphPositions(List paragraphs, String sectionContent) { List positions = new ArrayList<>(); int searchFrom = 0; for (String p : paragraphs) { int idx = sectionContent.indexOf(p, searchFrom); if (idx >= 0) { positions.add(new ParagraphPos(idx, idx + p.length())); searchFrom = idx + p.length(); } else { // fallback: 段落在原文中找不到(不应该发生) positions.add(new ParagraphPos(searchFrom, searchFrom + p.length())); searchFrom += p.length(); } } return positions; } /** * 启发式 token 估算(无需外部依赖) *

* 中文(BMP): ~1 字符/token * 英文/数字/标点: ~4 字符/token * 空白字符忽略 */ private int estimateTokens(String text) { int nonCjkCount = 0; int cjkCount = 0; for (char c : text.toCharArray()) { if (Character.isWhitespace(c)) { continue; } Character.UnicodeBlock block = Character.UnicodeBlock.of(c); if (block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS || block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A || block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B || block == Character.UnicodeBlock.CJK_COMPATIBILITY_IDEOGRAPHS) { cjkCount++; } else { nonCjkCount++; } } return cjkCount + (nonCjkCount + 3) / 4; // 非中文每 4 字符算 1 token,向上取整 } /** * 判断当前段落是否属于不可中断的结构 *

* 不可中断结构包括: * - 有序列表项("1. ", "2. " 格式) * - 无序列表项("- " 或 "* " 格式) * - 未闭合的代码块(``` 内) */ private boolean isInUnbreakableContext(String buffer, String nextParagraph) { // 有序列表:判断 buffer 末尾和下一段是否都是列表项 if (nextParagraph.matches("^\\d{1,2}\\.\\s.*")) { String lastLine = getLastNonEmptyLine(buffer); if (lastLine != null && lastLine.matches("^\\d{1,2}\\.\\s.*")) { return true; } } // 无序列表:"- " 或 "* " 格式 if (nextParagraph.matches("^[-*]\\s.*")) { String lastLine = getLastNonEmptyLine(buffer); if (lastLine != null && lastLine.matches("^[-*]\\s.*")) { return true; } } // 代码块:``` 未闭合 if (buffer.contains("```")) { int count = 0; for (int i = 0; i <= buffer.length() - 3; i++) { if (buffer.substring(i).startsWith("```")) { count++; i += 2; } } if (count % 2 == 1) { return true; // 奇数个 ``` → 在代码块内部 } } return false; } /** * 获取 buffer 中最后一行非空白文本 */ private String getLastNonEmptyLine(String buffer) { String[] lines = buffer.split("\n"); for (int i = lines.length - 1; i >= 0; i--) { String line = lines[i].trim(); if (!line.isEmpty()) { return line; } } return null; } /** * 获取重叠文本 * 从文本末尾提取指定长度的内容作为下一个分片的开头 */ private String getOverlapText(String text) { int overlapSize = Math.min(chunkConfig.getOverlap(), text.length()); if (overlapSize <= 0) { return ""; } // 从末尾提取重叠内容 String overlap = text.substring(text.length() - overlapSize); // 尝试在句子边界截断(查找最后一个句号、问号、感叹号) int lastSentenceEnd = Math.max( overlap.lastIndexOf('。'), Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!')) ); if (lastSentenceEnd > overlapSize / 2) { return overlap.substring(lastSentenceEnd + 1).trim(); } return overlap.trim(); } /** * 段落在原文中的位置 */ private static class ParagraphPos { final int start; final int end; ParagraphPos(int start, int end) { this.start = start; this.end = end; } } /** * 章节数据类 */ private static class Section { String title; // 最近一级标题名称 int level; // 标题级别(1-6),0=无标题 String breadcrumb; // 完整面包屑路径 String content; // 章节内容 int startIndex; // 在原文中的起始偏移 Section(String title, int level, String breadcrumb, String content, int startIndex) { this.title = title; this.level = level; this.breadcrumb = breadcrumb; this.content = content; this.startIndex = startIndex; } } }