commit
This commit is contained in:
@@ -78,10 +78,16 @@ public class MilvusClientFactory {
|
||||
ConnectParam.Builder builder = ConnectParam.newBuilder()
|
||||
.withHost(milvusProperties.getHost())
|
||||
.withPort(milvusProperties.getPort())
|
||||
.withDatabaseName(milvusProperties.getDatabase())
|
||||
.withConnectTimeout(milvusProperties.getTimeout(), TimeUnit.MILLISECONDS);
|
||||
|
||||
// 如果配置了用户名和密码
|
||||
if (milvusProperties.getUsername() != null && !milvusProperties.getUsername().isEmpty()) {
|
||||
// Zilliz Cloud: token + SSL
|
||||
if (milvusProperties.getToken() != null && !milvusProperties.getToken().isEmpty()) {
|
||||
builder.withToken(milvusProperties.getToken());
|
||||
builder.withSecure(true);
|
||||
}
|
||||
// 本地 Milvus: username + password
|
||||
else if (milvusProperties.getUsername() != null && !milvusProperties.getUsername().isEmpty()) {
|
||||
builder.withAuthorization(milvusProperties.getUsername(), milvusProperties.getPassword());
|
||||
}
|
||||
|
||||
|
||||
@@ -11,17 +11,29 @@ import org.springframework.context.annotation.Configuration;
|
||||
@Configuration
|
||||
@ConfigurationProperties(prefix = "document.chunk")
|
||||
public class DocumentChunkConfig {
|
||||
|
||||
|
||||
/**
|
||||
* 每个分片的最大字符数
|
||||
* 每个分片的最大字符数(保留向后兼容)
|
||||
*/
|
||||
private int maxSize = 800;
|
||||
|
||||
|
||||
/**
|
||||
* 分片之间的重叠字符数
|
||||
*/
|
||||
private int overlap = 100;
|
||||
|
||||
/**
|
||||
* 每个分片的最大 token 数(中文~1:1,英文~0.25:1)
|
||||
* 替代 maxSize 作为切割触发器
|
||||
*/
|
||||
private int maxTokens = 500;
|
||||
|
||||
/**
|
||||
* 硬上限 token 数 = maxTokens × 1.2
|
||||
* 仅在不可中断上下文(列表、代码块)内触发
|
||||
*/
|
||||
private int maxTokensHard = 600;
|
||||
|
||||
public void setMaxSize(int maxSize) {
|
||||
this.maxSize = maxSize;
|
||||
}
|
||||
@@ -29,4 +41,12 @@ public class DocumentChunkConfig {
|
||||
public void setOverlap(int overlap) {
|
||||
this.overlap = overlap;
|
||||
}
|
||||
|
||||
public void setMaxTokens(int maxTokens) {
|
||||
this.maxTokens = maxTokens;
|
||||
}
|
||||
|
||||
public void setMaxTokensHard(int maxTokensHard) {
|
||||
this.maxTokensHard = maxTokensHard;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,8 @@ public class MilvusProperties {
|
||||
private String password = "";
|
||||
private String database = "default";
|
||||
private Long timeout = 10000L;
|
||||
private String token = "";
|
||||
private boolean secure = false;
|
||||
|
||||
public String getHost() {
|
||||
return host;
|
||||
@@ -62,6 +64,22 @@ public class MilvusProperties {
|
||||
this.timeout = timeout;
|
||||
}
|
||||
|
||||
public String getToken() {
|
||||
return token;
|
||||
}
|
||||
|
||||
public void setToken(String token) {
|
||||
this.token = token;
|
||||
}
|
||||
|
||||
public boolean isSecure() {
|
||||
return secure;
|
||||
}
|
||||
|
||||
public void setSecure(boolean secure) {
|
||||
this.secure = secure;
|
||||
}
|
||||
|
||||
public String getAddress() {
|
||||
return host + ":" + port;
|
||||
}
|
||||
|
||||
@@ -27,7 +27,7 @@ public class DocumentChunkService {
|
||||
/**
|
||||
* 智能分片文档
|
||||
* 优先按照标题、段落边界进行分割,保持语义完整性
|
||||
*
|
||||
*
|
||||
* @param content 文档内容
|
||||
* @param filePath 文件路径(用于日志)
|
||||
* @return 文档分片列表
|
||||
@@ -42,7 +42,7 @@ public class DocumentChunkService {
|
||||
|
||||
// 1. 首先尝试按标题分割(Markdown格式)
|
||||
List<Section> sections = splitByHeadings(content);
|
||||
|
||||
|
||||
// 2. 对每个章节进行进一步分片
|
||||
int globalChunkIndex = 0;
|
||||
for (Section section : sections) {
|
||||
@@ -60,7 +60,7 @@ public class DocumentChunkService {
|
||||
*/
|
||||
private List<Section> splitByHeadings(String content) {
|
||||
List<Section> sections = new ArrayList<>();
|
||||
|
||||
|
||||
// 匹配 Markdown 标题:# 标题, ## 标题, ### 标题等
|
||||
Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE);
|
||||
Matcher matcher = headingPattern.matcher(content);
|
||||
@@ -100,18 +100,25 @@ public class DocumentChunkService {
|
||||
|
||||
/**
|
||||
* 对单个章节进行分片
|
||||
* <p>
|
||||
* 核心改造(Phase 1):
|
||||
* - Token 估算替代字符计数
|
||||
* - 感知有序/无序列表结构,不在列表中间切断
|
||||
* - 软边界(maxTokens)+ 硬上限(maxTokensHard)双重控制
|
||||
* - 修复 currentStartIndex 漂移:用段落原始位置而非手工推算
|
||||
*/
|
||||
private List<DocumentChunk> chunkSection(Section section, int startChunkIndex) {
|
||||
List<DocumentChunk> chunks = new ArrayList<>();
|
||||
String content = section.content;
|
||||
String title = section.title;
|
||||
|
||||
// 如果章节内容小于最大尺寸,直接作为一个分片
|
||||
if (content.length() <= chunkConfig.getMaxSize()) {
|
||||
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
|
||||
if (content.length() <= chunkConfig.getMaxSize()
|
||||
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) {
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
content,
|
||||
section.startIndex,
|
||||
section.startIndex + content.length(),
|
||||
content,
|
||||
section.startIndex,
|
||||
section.startIndex + content.length(),
|
||||
startChunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
@@ -120,45 +127,71 @@ public class DocumentChunkService {
|
||||
}
|
||||
|
||||
// 章节内容较长,需要进一步分片
|
||||
// 优先在段落边界分割
|
||||
List<String> paragraphs = splitByParagraphs(content);
|
||||
|
||||
StringBuilder currentChunk = new StringBuilder();
|
||||
int currentStartIndex = section.startIndex;
|
||||
if (paragraphs.isEmpty()) {
|
||||
return chunks;
|
||||
}
|
||||
|
||||
// 定位每个段落在 section.content 中的位置(修复 index 漂移)
|
||||
List<ParagraphPos> paraPositions = locateParagraphPositions(paragraphs, content);
|
||||
|
||||
// 当前分片的段落范围
|
||||
int chunkParaStart = 0; // 当前分片第一个段落的索引(在 paragraphs 中)
|
||||
StringBuilder buffer = new StringBuilder();
|
||||
int tokenCount = 0;
|
||||
int chunkIndex = startChunkIndex;
|
||||
|
||||
for (String paragraph : paragraphs) {
|
||||
// 如果当前分片加上新段落超过最大尺寸
|
||||
if (currentChunk.length() > 0 &&
|
||||
currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) {
|
||||
|
||||
// 保存当前分片
|
||||
String chunkContent = currentChunk.toString().trim();
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
chunkContent,
|
||||
currentStartIndex,
|
||||
currentStartIndex + chunkContent.length(),
|
||||
chunkIndex++
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
chunks.add(chunk);
|
||||
for (int i = 0; i < paragraphs.size(); i++) {
|
||||
String paragraph = paragraphs.get(i);
|
||||
int paraTokens = estimateTokens(paragraph);
|
||||
|
||||
// 开始新分片,包含重叠部分
|
||||
String overlap = getOverlapText(chunkContent);
|
||||
currentChunk = new StringBuilder(overlap);
|
||||
currentStartIndex = currentStartIndex + chunkContent.length() - overlap.length();
|
||||
// 判断是否需要切分
|
||||
if (buffer.length() > 0 && tokenCount + paraTokens > chunkConfig.getMaxTokens()) {
|
||||
|
||||
// 检查是否处于不可中断的上下文中
|
||||
if (isInUnbreakableContext(buffer.toString(), paragraph)) {
|
||||
// 硬上限保护:即使不可中断也不能无限膨胀
|
||||
if (tokenCount + paraTokens > chunkConfig.getMaxTokensHard()) {
|
||||
logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens);
|
||||
chunkParaStart = saveChunkAndGetNextStart(
|
||||
chunks, section, paraPositions,
|
||||
chunkParaStart, i, title, chunkIndex);
|
||||
chunkIndex++;
|
||||
|
||||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||||
String overlap = getOverlapText(prevChunkContent);
|
||||
buffer = new StringBuilder(overlap);
|
||||
tokenCount = estimateTokens(overlap);
|
||||
}
|
||||
// 否则:容忍超出(软边界)
|
||||
} else {
|
||||
// 安全切点:段落边界
|
||||
chunkParaStart = saveChunkAndGetNextStart(
|
||||
chunks, section, paraPositions,
|
||||
chunkParaStart, i, title, chunkIndex);
|
||||
chunkIndex++;
|
||||
|
||||
// 新分片以重叠文本开头
|
||||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||||
String overlap = getOverlapText(prevChunkContent);
|
||||
buffer = new StringBuilder(overlap);
|
||||
tokenCount = estimateTokens(overlap);
|
||||
}
|
||||
}
|
||||
|
||||
currentChunk.append(paragraph).append("\n\n");
|
||||
buffer.append(paragraph).append("\n\n");
|
||||
tokenCount += paraTokens;
|
||||
}
|
||||
|
||||
// 保存最后一个分片
|
||||
if (currentChunk.length() > 0) {
|
||||
String chunkContent = currentChunk.toString().trim();
|
||||
if (buffer.length() > 0 && chunkParaStart < paragraphs.size()) {
|
||||
String chunkContent = buffer.toString().trim();
|
||||
int actualStart = paraPositions.get(chunkParaStart).start;
|
||||
int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
chunkContent,
|
||||
currentStartIndex,
|
||||
currentStartIndex + chunkContent.length(),
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
@@ -168,12 +201,42 @@ public class DocumentChunkService {
|
||||
return chunks;
|
||||
}
|
||||
|
||||
/**
|
||||
* 保存当前分块,返回下一个分块的起始段落索引
|
||||
* <p>
|
||||
* 从 section.content 中提取原始文本(而非手工拼装),修复 index 漂移问题
|
||||
*/
|
||||
private int saveChunkAndGetNextStart(
|
||||
List<DocumentChunk> chunks,
|
||||
Section section,
|
||||
List<ParagraphPos> paraPositions,
|
||||
int fromPara,
|
||||
int toPara,
|
||||
String title,
|
||||
int chunkIndex) {
|
||||
|
||||
int actualStart = paraPositions.get(fromPara).start;
|
||||
int actualEnd = paraPositions.get(toPara - 1).end;
|
||||
String originalText = section.content.substring(actualStart, actualEnd);
|
||||
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
originalText,
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
chunks.add(chunk);
|
||||
|
||||
return toPara; // 下一个分块的起始段落索引
|
||||
}
|
||||
|
||||
/**
|
||||
* 按段落分割文本
|
||||
*/
|
||||
private List<String> splitByParagraphs(String content) {
|
||||
List<String> paragraphs = new ArrayList<>();
|
||||
|
||||
|
||||
// 按双换行符分割段落
|
||||
String[] parts = content.split("\n\n+");
|
||||
for (String part : parts) {
|
||||
@@ -186,6 +249,106 @@ public class DocumentChunkService {
|
||||
return paragraphs;
|
||||
}
|
||||
|
||||
/**
|
||||
* 定位每个段落在原始文本中的字符偏移
|
||||
*/
|
||||
private List<ParagraphPos> locateParagraphPositions(List<String> paragraphs, String sectionContent) {
|
||||
List<ParagraphPos> positions = new ArrayList<>();
|
||||
int searchFrom = 0;
|
||||
for (String p : paragraphs) {
|
||||
int idx = sectionContent.indexOf(p, searchFrom);
|
||||
if (idx >= 0) {
|
||||
positions.add(new ParagraphPos(idx, idx + p.length()));
|
||||
searchFrom = idx + p.length();
|
||||
} else {
|
||||
// fallback: 段落在原文中找不到(不应该发生)
|
||||
positions.add(new ParagraphPos(searchFrom, searchFrom + p.length()));
|
||||
searchFrom += p.length();
|
||||
}
|
||||
}
|
||||
return positions;
|
||||
}
|
||||
|
||||
/**
|
||||
* 启发式 token 估算(无需外部依赖)
|
||||
* <p>
|
||||
* 中文(BMP): ~1 字符/token
|
||||
* 英文/数字/标点: ~4 字符/token
|
||||
* 空白字符忽略
|
||||
*/
|
||||
private int estimateTokens(String text) {
|
||||
int nonCjkCount = 0;
|
||||
int cjkCount = 0;
|
||||
for (char c : text.toCharArray()) {
|
||||
if (Character.isWhitespace(c)) {
|
||||
continue;
|
||||
}
|
||||
Character.UnicodeBlock block = Character.UnicodeBlock.of(c);
|
||||
if (block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS
|
||||
|| block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A
|
||||
|| block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B
|
||||
|| block == Character.UnicodeBlock.CJK_COMPATIBILITY_IDEOGRAPHS) {
|
||||
cjkCount++;
|
||||
} else {
|
||||
nonCjkCount++;
|
||||
}
|
||||
}
|
||||
return cjkCount + (nonCjkCount + 3) / 4; // 非中文每 4 字符算 1 token,向上取整
|
||||
}
|
||||
|
||||
/**
|
||||
* 判断当前段落是否属于不可中断的结构
|
||||
* <p>
|
||||
* 不可中断结构包括:
|
||||
* - 有序列表项("1. ", "2. " 格式)
|
||||
* - 无序列表项("- " 或 "* " 格式)
|
||||
* - 未闭合的代码块(``` 内)
|
||||
*/
|
||||
private boolean isInUnbreakableContext(String buffer, String nextParagraph) {
|
||||
// 有序列表:判断 buffer 末尾和下一段是否都是列表项
|
||||
if (nextParagraph.matches("^\\d{1,2}\\.\\s.*")) {
|
||||
String lastLine = getLastNonEmptyLine(buffer);
|
||||
if (lastLine != null && lastLine.matches("^\\d{1,2}\\.\\s.*")) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// 无序列表:"- " 或 "* " 格式
|
||||
if (nextParagraph.matches("^[-*]\\s.*")) {
|
||||
String lastLine = getLastNonEmptyLine(buffer);
|
||||
if (lastLine != null && lastLine.matches("^[-*]\\s.*")) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// 代码块:``` 未闭合
|
||||
if (buffer.contains("```")) {
|
||||
int count = 0;
|
||||
for (int i = 0; i <= buffer.length() - 3; i++) {
|
||||
if (buffer.substring(i).startsWith("```")) {
|
||||
count++;
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
if (count % 2 == 1) {
|
||||
return true; // 奇数个 ``` → 在代码块内部
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取 buffer 中最后一行非空白文本
|
||||
*/
|
||||
private String getLastNonEmptyLine(String buffer) {
|
||||
String[] lines = buffer.split("\n");
|
||||
for (int i = lines.length - 1; i >= 0; i--) {
|
||||
String line = lines[i].trim();
|
||||
if (!line.isEmpty()) {
|
||||
return line;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取重叠文本
|
||||
* 从文本末尾提取指定长度的内容作为下一个分片的开头
|
||||
@@ -198,13 +361,13 @@ public class DocumentChunkService {
|
||||
|
||||
// 从末尾提取重叠内容
|
||||
String overlap = text.substring(text.length() - overlapSize);
|
||||
|
||||
|
||||
// 尝试在句子边界截断(查找最后一个句号、问号、感叹号)
|
||||
int lastSentenceEnd = Math.max(
|
||||
overlap.lastIndexOf('。'),
|
||||
Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!'))
|
||||
);
|
||||
|
||||
|
||||
if (lastSentenceEnd > overlapSize / 2) {
|
||||
return overlap.substring(lastSentenceEnd + 1).trim();
|
||||
}
|
||||
@@ -212,6 +375,19 @@ public class DocumentChunkService {
|
||||
return overlap.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* 段落在原文中的位置
|
||||
*/
|
||||
private static class ParagraphPos {
|
||||
final int start;
|
||||
final int end;
|
||||
|
||||
ParagraphPos(int start, int end) {
|
||||
this.start = start;
|
||||
this.end = end;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 章节数据类
|
||||
*/
|
||||
|
||||
@@ -12,12 +12,14 @@ file:
|
||||
allowed-extensions: txt,md
|
||||
|
||||
milvus:
|
||||
host: localhost
|
||||
port: 19530
|
||||
host: in03-4a578da0f27ce9d.serverless.aws-eu-central-1.cloud.zilliz.com
|
||||
port: 443
|
||||
username: ""
|
||||
password: ""
|
||||
database: default
|
||||
database: db_4a578da0f27ce9d
|
||||
timeout: 10000
|
||||
token: ${MILVUS_TOKEN:}
|
||||
secure: true
|
||||
|
||||
# Spring AI Alibaba DashScope 配置
|
||||
spring:
|
||||
|
||||
Reference in New Issue
Block a user