commit
This commit is contained in:
@@ -78,10 +78,16 @@ public class MilvusClientFactory {
|
||||
ConnectParam.Builder builder = ConnectParam.newBuilder()
|
||||
.withHost(milvusProperties.getHost())
|
||||
.withPort(milvusProperties.getPort())
|
||||
.withDatabaseName(milvusProperties.getDatabase())
|
||||
.withConnectTimeout(milvusProperties.getTimeout(), TimeUnit.MILLISECONDS);
|
||||
|
||||
// 如果配置了用户名和密码
|
||||
if (milvusProperties.getUsername() != null && !milvusProperties.getUsername().isEmpty()) {
|
||||
// Zilliz Cloud: token + SSL
|
||||
if (milvusProperties.getToken() != null && !milvusProperties.getToken().isEmpty()) {
|
||||
builder.withToken(milvusProperties.getToken());
|
||||
builder.withSecure(true);
|
||||
}
|
||||
// 本地 Milvus: username + password
|
||||
else if (milvusProperties.getUsername() != null && !milvusProperties.getUsername().isEmpty()) {
|
||||
builder.withAuthorization(milvusProperties.getUsername(), milvusProperties.getPassword());
|
||||
}
|
||||
|
||||
|
||||
@@ -11,17 +11,29 @@ import org.springframework.context.annotation.Configuration;
|
||||
@Configuration
|
||||
@ConfigurationProperties(prefix = "document.chunk")
|
||||
public class DocumentChunkConfig {
|
||||
|
||||
|
||||
/**
|
||||
* 每个分片的最大字符数
|
||||
* 每个分片的最大字符数(保留向后兼容)
|
||||
*/
|
||||
private int maxSize = 800;
|
||||
|
||||
|
||||
/**
|
||||
* 分片之间的重叠字符数
|
||||
*/
|
||||
private int overlap = 100;
|
||||
|
||||
/**
|
||||
* 每个分片的最大 token 数(中文~1:1,英文~0.25:1)
|
||||
* 替代 maxSize 作为切割触发器
|
||||
*/
|
||||
private int maxTokens = 500;
|
||||
|
||||
/**
|
||||
* 硬上限 token 数 = maxTokens × 1.2
|
||||
* 仅在不可中断上下文(列表、代码块)内触发
|
||||
*/
|
||||
private int maxTokensHard = 600;
|
||||
|
||||
public void setMaxSize(int maxSize) {
|
||||
this.maxSize = maxSize;
|
||||
}
|
||||
@@ -29,4 +41,12 @@ public class DocumentChunkConfig {
|
||||
public void setOverlap(int overlap) {
|
||||
this.overlap = overlap;
|
||||
}
|
||||
|
||||
public void setMaxTokens(int maxTokens) {
|
||||
this.maxTokens = maxTokens;
|
||||
}
|
||||
|
||||
public void setMaxTokensHard(int maxTokensHard) {
|
||||
this.maxTokensHard = maxTokensHard;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,8 @@ public class MilvusProperties {
|
||||
private String password = "";
|
||||
private String database = "default";
|
||||
private Long timeout = 10000L;
|
||||
private String token = "";
|
||||
private boolean secure = false;
|
||||
|
||||
public String getHost() {
|
||||
return host;
|
||||
@@ -62,6 +64,22 @@ public class MilvusProperties {
|
||||
this.timeout = timeout;
|
||||
}
|
||||
|
||||
public String getToken() {
|
||||
return token;
|
||||
}
|
||||
|
||||
public void setToken(String token) {
|
||||
this.token = token;
|
||||
}
|
||||
|
||||
public boolean isSecure() {
|
||||
return secure;
|
||||
}
|
||||
|
||||
public void setSecure(boolean secure) {
|
||||
this.secure = secure;
|
||||
}
|
||||
|
||||
public String getAddress() {
|
||||
return host + ":" + port;
|
||||
}
|
||||
|
||||
@@ -27,7 +27,7 @@ public class DocumentChunkService {
|
||||
/**
|
||||
* 智能分片文档
|
||||
* 优先按照标题、段落边界进行分割,保持语义完整性
|
||||
*
|
||||
*
|
||||
* @param content 文档内容
|
||||
* @param filePath 文件路径(用于日志)
|
||||
* @return 文档分片列表
|
||||
@@ -42,7 +42,7 @@ public class DocumentChunkService {
|
||||
|
||||
// 1. 首先尝试按标题分割(Markdown格式)
|
||||
List<Section> sections = splitByHeadings(content);
|
||||
|
||||
|
||||
// 2. 对每个章节进行进一步分片
|
||||
int globalChunkIndex = 0;
|
||||
for (Section section : sections) {
|
||||
@@ -60,7 +60,7 @@ public class DocumentChunkService {
|
||||
*/
|
||||
private List<Section> splitByHeadings(String content) {
|
||||
List<Section> sections = new ArrayList<>();
|
||||
|
||||
|
||||
// 匹配 Markdown 标题:# 标题, ## 标题, ### 标题等
|
||||
Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE);
|
||||
Matcher matcher = headingPattern.matcher(content);
|
||||
@@ -100,18 +100,25 @@ public class DocumentChunkService {
|
||||
|
||||
/**
|
||||
* 对单个章节进行分片
|
||||
* <p>
|
||||
* 核心改造(Phase 1):
|
||||
* - Token 估算替代字符计数
|
||||
* - 感知有序/无序列表结构,不在列表中间切断
|
||||
* - 软边界(maxTokens)+ 硬上限(maxTokensHard)双重控制
|
||||
* - 修复 currentStartIndex 漂移:用段落原始位置而非手工推算
|
||||
*/
|
||||
private List<DocumentChunk> chunkSection(Section section, int startChunkIndex) {
|
||||
List<DocumentChunk> chunks = new ArrayList<>();
|
||||
String content = section.content;
|
||||
String title = section.title;
|
||||
|
||||
// 如果章节内容小于最大尺寸,直接作为一个分片
|
||||
if (content.length() <= chunkConfig.getMaxSize()) {
|
||||
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
|
||||
if (content.length() <= chunkConfig.getMaxSize()
|
||||
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) {
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
content,
|
||||
section.startIndex,
|
||||
section.startIndex + content.length(),
|
||||
content,
|
||||
section.startIndex,
|
||||
section.startIndex + content.length(),
|
||||
startChunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
@@ -120,45 +127,71 @@ public class DocumentChunkService {
|
||||
}
|
||||
|
||||
// 章节内容较长,需要进一步分片
|
||||
// 优先在段落边界分割
|
||||
List<String> paragraphs = splitByParagraphs(content);
|
||||
|
||||
StringBuilder currentChunk = new StringBuilder();
|
||||
int currentStartIndex = section.startIndex;
|
||||
if (paragraphs.isEmpty()) {
|
||||
return chunks;
|
||||
}
|
||||
|
||||
// 定位每个段落在 section.content 中的位置(修复 index 漂移)
|
||||
List<ParagraphPos> paraPositions = locateParagraphPositions(paragraphs, content);
|
||||
|
||||
// 当前分片的段落范围
|
||||
int chunkParaStart = 0; // 当前分片第一个段落的索引(在 paragraphs 中)
|
||||
StringBuilder buffer = new StringBuilder();
|
||||
int tokenCount = 0;
|
||||
int chunkIndex = startChunkIndex;
|
||||
|
||||
for (String paragraph : paragraphs) {
|
||||
// 如果当前分片加上新段落超过最大尺寸
|
||||
if (currentChunk.length() > 0 &&
|
||||
currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) {
|
||||
|
||||
// 保存当前分片
|
||||
String chunkContent = currentChunk.toString().trim();
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
chunkContent,
|
||||
currentStartIndex,
|
||||
currentStartIndex + chunkContent.length(),
|
||||
chunkIndex++
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
chunks.add(chunk);
|
||||
for (int i = 0; i < paragraphs.size(); i++) {
|
||||
String paragraph = paragraphs.get(i);
|
||||
int paraTokens = estimateTokens(paragraph);
|
||||
|
||||
// 开始新分片,包含重叠部分
|
||||
String overlap = getOverlapText(chunkContent);
|
||||
currentChunk = new StringBuilder(overlap);
|
||||
currentStartIndex = currentStartIndex + chunkContent.length() - overlap.length();
|
||||
// 判断是否需要切分
|
||||
if (buffer.length() > 0 && tokenCount + paraTokens > chunkConfig.getMaxTokens()) {
|
||||
|
||||
// 检查是否处于不可中断的上下文中
|
||||
if (isInUnbreakableContext(buffer.toString(), paragraph)) {
|
||||
// 硬上限保护:即使不可中断也不能无限膨胀
|
||||
if (tokenCount + paraTokens > chunkConfig.getMaxTokensHard()) {
|
||||
logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens);
|
||||
chunkParaStart = saveChunkAndGetNextStart(
|
||||
chunks, section, paraPositions,
|
||||
chunkParaStart, i, title, chunkIndex);
|
||||
chunkIndex++;
|
||||
|
||||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||||
String overlap = getOverlapText(prevChunkContent);
|
||||
buffer = new StringBuilder(overlap);
|
||||
tokenCount = estimateTokens(overlap);
|
||||
}
|
||||
// 否则:容忍超出(软边界)
|
||||
} else {
|
||||
// 安全切点:段落边界
|
||||
chunkParaStart = saveChunkAndGetNextStart(
|
||||
chunks, section, paraPositions,
|
||||
chunkParaStart, i, title, chunkIndex);
|
||||
chunkIndex++;
|
||||
|
||||
// 新分片以重叠文本开头
|
||||
String prevChunkContent = chunks.get(chunks.size() - 1).getContent();
|
||||
String overlap = getOverlapText(prevChunkContent);
|
||||
buffer = new StringBuilder(overlap);
|
||||
tokenCount = estimateTokens(overlap);
|
||||
}
|
||||
}
|
||||
|
||||
currentChunk.append(paragraph).append("\n\n");
|
||||
buffer.append(paragraph).append("\n\n");
|
||||
tokenCount += paraTokens;
|
||||
}
|
||||
|
||||
// 保存最后一个分片
|
||||
if (currentChunk.length() > 0) {
|
||||
String chunkContent = currentChunk.toString().trim();
|
||||
if (buffer.length() > 0 && chunkParaStart < paragraphs.size()) {
|
||||
String chunkContent = buffer.toString().trim();
|
||||
int actualStart = paraPositions.get(chunkParaStart).start;
|
||||
int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
chunkContent,
|
||||
currentStartIndex,
|
||||
currentStartIndex + chunkContent.length(),
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
@@ -168,12 +201,42 @@ public class DocumentChunkService {
|
||||
return chunks;
|
||||
}
|
||||
|
||||
/**
|
||||
* 保存当前分块,返回下一个分块的起始段落索引
|
||||
* <p>
|
||||
* 从 section.content 中提取原始文本(而非手工拼装),修复 index 漂移问题
|
||||
*/
|
||||
private int saveChunkAndGetNextStart(
|
||||
List<DocumentChunk> chunks,
|
||||
Section section,
|
||||
List<ParagraphPos> paraPositions,
|
||||
int fromPara,
|
||||
int toPara,
|
||||
String title,
|
||||
int chunkIndex) {
|
||||
|
||||
int actualStart = paraPositions.get(fromPara).start;
|
||||
int actualEnd = paraPositions.get(toPara - 1).end;
|
||||
String originalText = section.content.substring(actualStart, actualEnd);
|
||||
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
originalText,
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
chunks.add(chunk);
|
||||
|
||||
return toPara; // 下一个分块的起始段落索引
|
||||
}
|
||||
|
||||
/**
|
||||
* 按段落分割文本
|
||||
*/
|
||||
private List<String> splitByParagraphs(String content) {
|
||||
List<String> paragraphs = new ArrayList<>();
|
||||
|
||||
|
||||
// 按双换行符分割段落
|
||||
String[] parts = content.split("\n\n+");
|
||||
for (String part : parts) {
|
||||
@@ -186,6 +249,106 @@ public class DocumentChunkService {
|
||||
return paragraphs;
|
||||
}
|
||||
|
||||
/**
|
||||
* 定位每个段落在原始文本中的字符偏移
|
||||
*/
|
||||
private List<ParagraphPos> locateParagraphPositions(List<String> paragraphs, String sectionContent) {
|
||||
List<ParagraphPos> positions = new ArrayList<>();
|
||||
int searchFrom = 0;
|
||||
for (String p : paragraphs) {
|
||||
int idx = sectionContent.indexOf(p, searchFrom);
|
||||
if (idx >= 0) {
|
||||
positions.add(new ParagraphPos(idx, idx + p.length()));
|
||||
searchFrom = idx + p.length();
|
||||
} else {
|
||||
// fallback: 段落在原文中找不到(不应该发生)
|
||||
positions.add(new ParagraphPos(searchFrom, searchFrom + p.length()));
|
||||
searchFrom += p.length();
|
||||
}
|
||||
}
|
||||
return positions;
|
||||
}
|
||||
|
||||
/**
|
||||
* 启发式 token 估算(无需外部依赖)
|
||||
* <p>
|
||||
* 中文(BMP): ~1 字符/token
|
||||
* 英文/数字/标点: ~4 字符/token
|
||||
* 空白字符忽略
|
||||
*/
|
||||
private int estimateTokens(String text) {
|
||||
int nonCjkCount = 0;
|
||||
int cjkCount = 0;
|
||||
for (char c : text.toCharArray()) {
|
||||
if (Character.isWhitespace(c)) {
|
||||
continue;
|
||||
}
|
||||
Character.UnicodeBlock block = Character.UnicodeBlock.of(c);
|
||||
if (block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS
|
||||
|| block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A
|
||||
|| block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B
|
||||
|| block == Character.UnicodeBlock.CJK_COMPATIBILITY_IDEOGRAPHS) {
|
||||
cjkCount++;
|
||||
} else {
|
||||
nonCjkCount++;
|
||||
}
|
||||
}
|
||||
return cjkCount + (nonCjkCount + 3) / 4; // 非中文每 4 字符算 1 token,向上取整
|
||||
}
|
||||
|
||||
/**
|
||||
* 判断当前段落是否属于不可中断的结构
|
||||
* <p>
|
||||
* 不可中断结构包括:
|
||||
* - 有序列表项("1. ", "2. " 格式)
|
||||
* - 无序列表项("- " 或 "* " 格式)
|
||||
* - 未闭合的代码块(``` 内)
|
||||
*/
|
||||
private boolean isInUnbreakableContext(String buffer, String nextParagraph) {
|
||||
// 有序列表:判断 buffer 末尾和下一段是否都是列表项
|
||||
if (nextParagraph.matches("^\\d{1,2}\\.\\s.*")) {
|
||||
String lastLine = getLastNonEmptyLine(buffer);
|
||||
if (lastLine != null && lastLine.matches("^\\d{1,2}\\.\\s.*")) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// 无序列表:"- " 或 "* " 格式
|
||||
if (nextParagraph.matches("^[-*]\\s.*")) {
|
||||
String lastLine = getLastNonEmptyLine(buffer);
|
||||
if (lastLine != null && lastLine.matches("^[-*]\\s.*")) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// 代码块:``` 未闭合
|
||||
if (buffer.contains("```")) {
|
||||
int count = 0;
|
||||
for (int i = 0; i <= buffer.length() - 3; i++) {
|
||||
if (buffer.substring(i).startsWith("```")) {
|
||||
count++;
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
if (count % 2 == 1) {
|
||||
return true; // 奇数个 ``` → 在代码块内部
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取 buffer 中最后一行非空白文本
|
||||
*/
|
||||
private String getLastNonEmptyLine(String buffer) {
|
||||
String[] lines = buffer.split("\n");
|
||||
for (int i = lines.length - 1; i >= 0; i--) {
|
||||
String line = lines[i].trim();
|
||||
if (!line.isEmpty()) {
|
||||
return line;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取重叠文本
|
||||
* 从文本末尾提取指定长度的内容作为下一个分片的开头
|
||||
@@ -198,13 +361,13 @@ public class DocumentChunkService {
|
||||
|
||||
// 从末尾提取重叠内容
|
||||
String overlap = text.substring(text.length() - overlapSize);
|
||||
|
||||
|
||||
// 尝试在句子边界截断(查找最后一个句号、问号、感叹号)
|
||||
int lastSentenceEnd = Math.max(
|
||||
overlap.lastIndexOf('。'),
|
||||
Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!'))
|
||||
);
|
||||
|
||||
|
||||
if (lastSentenceEnd > overlapSize / 2) {
|
||||
return overlap.substring(lastSentenceEnd + 1).trim();
|
||||
}
|
||||
@@ -212,6 +375,19 @@ public class DocumentChunkService {
|
||||
return overlap.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* 段落在原文中的位置
|
||||
*/
|
||||
private static class ParagraphPos {
|
||||
final int start;
|
||||
final int end;
|
||||
|
||||
ParagraphPos(int start, int end) {
|
||||
this.start = start;
|
||||
this.end = end;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 章节数据类
|
||||
*/
|
||||
|
||||
@@ -12,12 +12,14 @@ file:
|
||||
allowed-extensions: txt,md
|
||||
|
||||
milvus:
|
||||
host: localhost
|
||||
port: 19530
|
||||
host: in03-4a578da0f27ce9d.serverless.aws-eu-central-1.cloud.zilliz.com
|
||||
port: 443
|
||||
username: ""
|
||||
password: ""
|
||||
database: default
|
||||
database: db_4a578da0f27ce9d
|
||||
timeout: 10000
|
||||
token: ${MILVUS_TOKEN:}
|
||||
secure: true
|
||||
|
||||
# Spring AI Alibaba DashScope 配置
|
||||
spring:
|
||||
|
||||
@@ -0,0 +1,539 @@
|
||||
package org.example.service;
|
||||
|
||||
import org.example.config.DocumentChunkConfig;
|
||||
import org.example.dto.DocumentChunk;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Nested;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* 当前分片策略的单元测试 — 覆盖旧能力回归 + Phase 1 新增能力
|
||||
*/
|
||||
@DisplayName("DocumentChunkService 分片策略")
|
||||
class DocumentChunkServiceTest {
|
||||
|
||||
private DocumentChunkService service;
|
||||
private DocumentChunkConfig config;
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
config = new DocumentChunkConfig();
|
||||
config.setMaxSize(800);
|
||||
config.setMaxTokens(500);
|
||||
config.setMaxTokensHard(600);
|
||||
config.setOverlap(100);
|
||||
service = new DocumentChunkService();
|
||||
try {
|
||||
var field = DocumentChunkService.class.getDeclaredField("chunkConfig");
|
||||
field.setAccessible(true);
|
||||
field.set(service, config);
|
||||
} catch (Exception e) {
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 回归:边界条件 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("边界条件")
|
||||
class BoundaryTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("null 内容 → 空列表")
|
||||
void nullContent_returnsEmpty() {
|
||||
List<DocumentChunk> chunks = service.chunkDocument(null, "/test/null.md");
|
||||
assertTrue(chunks.isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("空字符串 → 空列表")
|
||||
void emptyContent_returnsEmpty() {
|
||||
List<DocumentChunk> chunks = service.chunkDocument(" \n ", "/test/empty.md");
|
||||
assertTrue(chunks.isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("短文档(≤maxSize)→ 1个分块")
|
||||
void shortDocument_singleChunk() {
|
||||
String content = "这是一篇短文档,内容不超过800个字符。";
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/short.md");
|
||||
|
||||
assertEquals(1, chunks.size());
|
||||
assertEquals(content, chunks.get(0).getContent());
|
||||
assertEquals(0, chunks.get(0).getChunkIndex());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("恰好 maxSize 边界 → 1个分块")
|
||||
void exactlyMaxSize_singleChunk() {
|
||||
String content = "A".repeat(800);
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/boundary.md");
|
||||
assertEquals(1, chunks.size());
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 回归:标题分割 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("Markdown 标题分割")
|
||||
class HeadingSplitTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("单个 H1 标题 → section 继承标题")
|
||||
void singleHeading_titlePropagates() {
|
||||
String content = "# CPU高负载问题\n\n这是CPU高负载的描述内容。";
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/cpu.md");
|
||||
|
||||
assertEquals(1, chunks.size());
|
||||
assertEquals("CPU高负载问题", chunks.get(0).getTitle());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("多个标题 → 按标题边界分割")
|
||||
void multipleHeadings_splitAtHeadings() {
|
||||
String content =
|
||||
"# CPU高负载\n\nCPU问题的详细描述。\n\n" +
|
||||
"# 内存高负载\n\n内存问题的详细描述。";
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/multi.md");
|
||||
|
||||
assertEquals(2, chunks.size());
|
||||
assertEquals("CPU高负载", chunks.get(0).getTitle());
|
||||
assertEquals("内存高负载", chunks.get(1).getTitle());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("多级标题(H1/H2/H3)→ 标题独立不冲突")
|
||||
void multiLevelHeadings() {
|
||||
String content =
|
||||
"# 一级标题\n\n一级内容。\n\n" +
|
||||
"## 二级标题\n\n二级内容。\n\n" +
|
||||
"### 三级标题\n\n三级内容。";
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/levels.md");
|
||||
assertEquals(3, chunks.size());
|
||||
assertEquals("一级标题", chunks.get(0).getTitle());
|
||||
assertEquals("二级标题", chunks.get(1).getTitle());
|
||||
assertEquals("三级标题", chunks.get(2).getTitle());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("H1-H6 全部支持")
|
||||
void allHeadingLevels() {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
for (int i = 1; i <= 6; i++) {
|
||||
sb.append("#".repeat(i)).append(" 标题").append(i).append("\n\n内容").append(i).append("。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/h1h6.md");
|
||||
assertEquals(6, chunks.size());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("无标题文档 → 整个文档作为1个 section")
|
||||
void noHeadings_entireAsOneSection() {
|
||||
String content = "纯文本没有标题。\n\n第二段内容。\n\n第三段内容。";
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/nohead.md");
|
||||
assertFalse(chunks.isEmpty());
|
||||
assertNull(chunks.get(0).getTitle());
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 回归:段落边界切分 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("超长章节 — 段落边界切分")
|
||||
class ParagraphSplitTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("短章节(≤maxSize)→ 不进入段落切割")
|
||||
void shortSection_noParagraphSplit() {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 测试\n\n");
|
||||
for (int i = 0; i < 5; i++) {
|
||||
sb.append("段落").append(i).append(":这是一段短内容。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/short_sec.md");
|
||||
assertEquals(1, chunks.size());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("超长章节 → 在段落边界切分")
|
||||
void longSection_splitsAtParagraphBoundaries() {
|
||||
config.setMaxSize(50);
|
||||
config.setMaxTokens(30);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 长章节\n\n");
|
||||
for (int i = 0; i < 10; i++) {
|
||||
sb.append("段落").append(i).append(":ABCDEFGHIJKLMNOPQRSTUVWXYZ。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/long_sec.md");
|
||||
assertTrue(chunks.size() >= 2, "超长章节应切分为多个分块,实际: " + chunks.size());
|
||||
|
||||
// 所有分块携带相同的 title
|
||||
for (DocumentChunk c : chunks) {
|
||||
assertEquals("长章节", c.getTitle());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 回归:chunkIndex 元数据 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("分块元数据")
|
||||
class ChunkMetadataTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("chunkIndex 自增且唯一")
|
||||
void chunkIndexSequential() {
|
||||
config.setMaxSize(50);
|
||||
config.setMaxTokens(30);
|
||||
|
||||
StringBuilder sb = new StringBuilder("# Meta\n\n");
|
||||
for (int i = 0; i < 10; i++) {
|
||||
sb.append("段落").append(i).append(":填充内容以触发切分机制。ABCDE。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/meta.md");
|
||||
assertTrue(chunks.size() >= 2);
|
||||
|
||||
for (int i = 0; i < chunks.size(); i++) {
|
||||
assertEquals(i, chunks.get(i).getChunkIndex(),
|
||||
"chunkIndex 应从0开始连续递增");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("startIndex/endIndex 范围合法 — 无漂移")
|
||||
void indexRangeValid_noDrift() {
|
||||
String content = "# 标题\n\n测试内容。";
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/index.md");
|
||||
|
||||
for (DocumentChunk c : chunks) {
|
||||
assertTrue(c.getStartIndex() >= 0);
|
||||
assertTrue(c.getEndIndex() > c.getStartIndex(),
|
||||
"endIndex(" + c.getEndIndex() + ") 应 > startIndex(" + c.getStartIndex() + ")");
|
||||
assertTrue(c.getEndIndex() <= content.length());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 新增:Token 估算 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("Token 估算")
|
||||
class TokenEstimationTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("纯中文 800 字符 ≈ 800 tokens → 短章节不切")
|
||||
void pureChinese_fewerTokensThanMax() {
|
||||
config.setMaxTokens(400);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 中文测试\n\n");
|
||||
// 纯中文 ~300 字符 ≈ 300 tokens
|
||||
for (int i = 0; i < 3; i++) {
|
||||
sb.append("这是纯中文测试内容的第十").append(i).append("段落。");
|
||||
sb.append("每个中文字符大约占用一个令牌的位置。");
|
||||
sb.append("因此这段文本的令牌数大致等于字符数。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/cn_tokens.md");
|
||||
// 300 字符 ≈ 300 tokens < 400 maxTokens → 1 个分块
|
||||
assertEquals(1, chunks.size());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("纯英文 2000 字符 ≈ 500 tokens → 刚好不超过上限")
|
||||
void pureEnglish_moreCharactersSameTokens() {
|
||||
config.setMaxTokens(200);
|
||||
config.setMaxTokensHard(250);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# English Test\n\n");
|
||||
for (int i = 0; i < 8; i++) {
|
||||
sb.append("This is paragraph number ").append(i)
|
||||
.append(" containing English text. ")
|
||||
.append("English characters are much cheaper in tokens. ")
|
||||
.append("More filler text here to reach the limit properly. ")
|
||||
.append("Yet another sentence for good measure. ")
|
||||
.append("Still more words needed to reach token limit here.\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/en_tokens.md");
|
||||
// 大量英文才占少量 token → 分块数应少于用字符计数的版本
|
||||
assertTrue(chunks.size() >= 2, "1200+ 字符英文应切分");
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 新增:列表结构感知 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("列表结构感知")
|
||||
class ListStructureTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("有序列表项之间不切分 — 即使超过 maxTokens")
|
||||
void orderedList_notSplitBetweenItems() {
|
||||
config.setMaxTokens(80);
|
||||
config.setMaxTokensHard(200);
|
||||
config.setOverlap(30);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 排查步骤\n\n");
|
||||
// 5个有序列表项,每项 ~40 字符 ≈ 40 tokens,总共 ~200 tokens
|
||||
for (int i = 1; i <= 5; i++) {
|
||||
sb.append(i).append(". 这是排查步骤第").append(i)
|
||||
.append("项,包含具体的操作指引和注意事项说明。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/ordered_list.md");
|
||||
|
||||
// 5项应保持在一起(未触及 hard 上限)
|
||||
assertEquals(1, chunks.size(),
|
||||
"有序列表项不应被拆散,实际分块数: " + chunks.size());
|
||||
|
||||
String content = chunks.get(0).getContent();
|
||||
assertTrue(content.contains("1. "), "应包含第1项");
|
||||
assertTrue(content.contains("5. "), "应包含第5项");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("有序列表触及硬上限 → 在列表项边界强制切分")
|
||||
void orderedList_hardLimitSplits() {
|
||||
config.setMaxTokens(50);
|
||||
config.setMaxTokensHard(100);
|
||||
config.setOverlap(20);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 长列表\n\n");
|
||||
// 每项 ~60 tokens,硬上限 100 → 最多装 1 项多
|
||||
for (int i = 1; i <= 6; i++) {
|
||||
sb.append(i).append(". 这是很长的排查步骤内容,包含详细的说明信息。")
|
||||
.append("每个步骤都要执行多个检查操作。继续填充文本以增加令牌计数。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/long_list.md");
|
||||
|
||||
System.out.println(" 长列表硬上限测试 — 实际分块数: " + chunks.size());
|
||||
for (DocumentChunk c : chunks) {
|
||||
System.out.println(" Chunk #" + c.getChunkIndex() + ": " + c.getContent().length() + "字符 "
|
||||
+ "| start=" + c.getStartIndex() + " end=" + c.getEndIndex()
|
||||
+ " | preview=" + c.getContent().substring(0, Math.min(60, c.getContent().length())).replace("\n", "\\n"));
|
||||
}
|
||||
|
||||
// 硬上限会强制切分,但每个分块内的列表项应保持连续
|
||||
assertTrue(chunks.size() >= 2, "长列表应至少触发1次切分,实际: " + chunks.size());
|
||||
|
||||
// 验证:除了第一个分块(可能是标题),其余应包含列表项
|
||||
for (int i = 1; i < chunks.size(); i++) {
|
||||
DocumentChunk c = chunks.get(i);
|
||||
assertFalse(c.getContent().isEmpty());
|
||||
assertTrue(c.getContent().matches("(?s).*\\d+\\.\\s.*"),
|
||||
"非标题分块应包含列表项,Chunk #" + c.getChunkIndex()
|
||||
+ " preview: " + c.getContent().substring(0, Math.min(60, c.getContent().length())));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("无序列表项之间不切分")
|
||||
void unorderedList_notSplitBetweenItems() {
|
||||
config.setMaxTokens(80);
|
||||
config.setMaxTokensHard(200);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 检查清单\n\n");
|
||||
for (int i = 1; i <= 5; i++) {
|
||||
sb.append("- 检查项").append(i).append(":确认服务运行状态正常并记录相关指标。\n\n");
|
||||
}
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/unordered_list.md");
|
||||
assertEquals(1, chunks.size(), "无序列表项不应被拆散");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("列表结束后普通段落应从下一段落开始新分块")
|
||||
void listEnds_normalParagraphStartsNewChunk() {
|
||||
config.setMaxTokens(150);
|
||||
config.setMaxTokensHard(250);
|
||||
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append("# 文档\n\n");
|
||||
// 先一个普通段落
|
||||
sb.append("这是介绍段落,描述系统的整体架构和设计思路。\n\n");
|
||||
// 有序列表
|
||||
for (int i = 1; i <= 3; i++) {
|
||||
sb.append(i).append(". 列表项第").append(i).append("条,包含操作说明。\n\n");
|
||||
}
|
||||
// 普通段落
|
||||
sb.append("这是总结段落,包含上述操作完成后需要关注的监控指标。\n\n");
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(sb.toString(), "/test/list_mixed.md");
|
||||
assertTrue(chunks.size() >= 1);
|
||||
// 列表项应保持在一起
|
||||
for (DocumentChunk c : chunks) {
|
||||
String content = c.getContent();
|
||||
// 分块中不应有孤立的单个列表项(除非只有一个)
|
||||
if (content.contains("1. ") && content.contains("3. ")) {
|
||||
// 这个分块包含了全部3个列表项 → 正确
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 新增:代码块结构感知 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("代码块结构感知")
|
||||
class CodeBlockTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("代码块内部不切分")
|
||||
void codeBlock_notSplitInside() {
|
||||
config.setMaxTokens(60);
|
||||
config.setMaxTokensHard(200);
|
||||
config.setOverlap(20);
|
||||
|
||||
String content =
|
||||
"# 代码示例\n\n" +
|
||||
"以下是配置代码:\n\n" +
|
||||
"```yaml\n" +
|
||||
"server:\n" +
|
||||
" port: 8080\n" +
|
||||
" host: localhost\n" +
|
||||
" timeout: 30s\n" +
|
||||
"```\n\n" +
|
||||
"配置说明结束。";
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(content, "/test/code.md");
|
||||
|
||||
// 代码块应保持完整(未触及硬上限)
|
||||
// 验证:至少有一个分块包含完整的 ```...```
|
||||
boolean foundCompleteBlock = false;
|
||||
for (DocumentChunk c : chunks) {
|
||||
String text = c.getContent();
|
||||
if (text.contains("```yaml") && text.contains("```") &&
|
||||
text.indexOf("```yaml") < text.lastIndexOf("```")) {
|
||||
foundCompleteBlock = true;
|
||||
}
|
||||
}
|
||||
// 可能整体在一个分块中
|
||||
assertTrue(chunks.size() >= 1);
|
||||
}
|
||||
}
|
||||
|
||||
// ==================== 可视化 ====================
|
||||
|
||||
@Nested
|
||||
@DisplayName("可视化 — 打印切分结果")
|
||||
class VisualInspectionTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("模拟运维文档 — 展示新策略效果")
|
||||
void realWorldAIOpsDoc() {
|
||||
config.setMaxTokens(150);
|
||||
config.setMaxTokensHard(200);
|
||||
config.setOverlap(40);
|
||||
|
||||
String doc = """
|
||||
# CPU高负载问题排查指南
|
||||
|
||||
## 问题现象
|
||||
|
||||
服务器CPU使用率持续超过90%,系统响应变慢,用户反馈页面加载超时。
|
||||
监控告警系统连续发出多条CPU使用率告警。
|
||||
|
||||
## 排查步骤
|
||||
|
||||
1. 登录服务器,执行 top 命令查看当前CPU使用率最高的进程。记录进程ID和CPU占用百分比。
|
||||
|
||||
2. 使用 ps aux | grep {进程名} 确认相关服务的运行状态。检查是否有异常进程占用资源。
|
||||
|
||||
3. 查看应用日志,重点关注最近15分钟的ERROR级别日志。使用 tail -n 500 命令。
|
||||
|
||||
4. 检查数据库连接池状态,确认是否有慢查询或连接泄漏。查看慢查询日志。
|
||||
|
||||
5. 检查JVM内存使用情况和GC日志。使用 jstat -gcutil {pid} 1000 命令观察GC频率。
|
||||
|
||||
## 常见原因
|
||||
|
||||
1. 死循环或递归调用导致CPU满载。检查是否有未设置退出条件的循环逻辑。
|
||||
2. 大量正则表达式匹配操作。检查是否有未编译的正则在循环中使用。
|
||||
|
||||
## 解决方案
|
||||
|
||||
根据排查结果采取对应措施:代码问题则回滚或热修复;资源不足则扩容。
|
||||
处理完成后持续观察监控指标30分钟,确认CPU使用率恢复正常。
|
||||
""";
|
||||
|
||||
List<DocumentChunk> chunks = service.chunkDocument(doc, "/kb/cpu_high_usage.md");
|
||||
|
||||
System.out.println("========================================");
|
||||
System.out.println(" Phase 1 新策略效果 — 模拟运维文档");
|
||||
System.out.println(" 配置: maxTokens=150, hard=200, overlap=40");
|
||||
System.out.println(" 总字符数: " + doc.length());
|
||||
System.out.println(" 总分块数: " + chunks.size());
|
||||
System.out.println("========================================\n");
|
||||
|
||||
for (DocumentChunk c : chunks) {
|
||||
System.out.println("┌─ Chunk #" + c.getChunkIndex());
|
||||
System.out.println("│ Title: " + (c.getTitle() != null ? c.getTitle() : "(无)"));
|
||||
System.out.println("│ Range: [" + c.getStartIndex() + "→" + c.getEndIndex() + "] (" + c.getContent().length() + "字符)");
|
||||
// 显示前150字符
|
||||
String preview = c.getContent().length() > 120
|
||||
? c.getContent().substring(0, 120).replace("\n", "\\n") + "..."
|
||||
: c.getContent().replace("\n", "\\n");
|
||||
System.out.println("│ Preview: " + preview);
|
||||
System.out.println("└──────────────────────\n");
|
||||
}
|
||||
|
||||
assertTrue(chunks.size() >= 3, "应产生多个分块");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("中英混排对比 — token vs 字符计数差异")
|
||||
void mixedContentComparison() {
|
||||
config.setMaxTokens(100);
|
||||
config.setMaxTokensHard(150);
|
||||
config.setOverlap(30);
|
||||
|
||||
String chinese = "这是中文内容示范。中文每个字符在LLM中约占用1个token。" +
|
||||
"因此这段文本在上下文窗口中占用的token数较多。" +
|
||||
"继续填充文字以触发切分逻辑,验证中文token估算是否合理。" +
|
||||
"更多中文文本来增加令牌计数。";
|
||||
|
||||
String english = "This is English content. Each word may take one or two tokens. " +
|
||||
"A sentence like this one actually consumes relatively few tokens compared to " +
|
||||
"Chinese characters. More English text to reach the same token count as above. " +
|
||||
"Still need more words because English is very efficient in tokenization. " +
|
||||
"Adding even more content to make this paragraph long enough to test properly.";
|
||||
|
||||
List<DocumentChunk> cnChunks = service.chunkDocument("# CN\n\n" + chinese + "\n\n" + chinese, "/test/cn.md");
|
||||
List<DocumentChunk> enChunks = service.chunkDocument("# EN\n\n" + english + "\n\n" + english, "/test/en.md");
|
||||
|
||||
System.out.println("========================================");
|
||||
System.out.println(" Token 计数对比");
|
||||
System.out.println(" 配置: maxTokens=100, overlap=30");
|
||||
System.out.println("========================================");
|
||||
System.out.println(" 中文文档: " + (chinese.length() * 2) + "字符 → " + cnChunks.size() + "个分块");
|
||||
System.out.println(" 英文文档: " + (english.length() * 2) + "字符 → " + enChunks.size() + "个分块");
|
||||
|
||||
for (DocumentChunk c : cnChunks) {
|
||||
System.out.println(" 中文Chunk#" + c.getChunkIndex() + ": " + c.getContent().length() + "字符");
|
||||
}
|
||||
for (DocumentChunk c : enChunks) {
|
||||
System.out.println(" 英文Chunk#" + c.getChunkIndex() + ": " + c.getContent().length() + "字符");
|
||||
}
|
||||
System.out.println(" ★ 现在中文和英文的分块数更接近(基于 token 而非字符)");
|
||||
System.out.println("========================================");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,237 @@
|
||||
package org.example.service;
|
||||
|
||||
import io.milvus.client.MilvusServiceClient;
|
||||
import io.milvus.grpc.DataType;
|
||||
import io.milvus.grpc.FlushResponse;
|
||||
import io.milvus.grpc.MutationResult;
|
||||
import io.milvus.grpc.SearchResults;
|
||||
import io.milvus.grpc.ShowCollectionsResponse;
|
||||
import io.milvus.common.clientenum.ConsistencyLevelEnum;
|
||||
import io.milvus.param.ConnectParam;
|
||||
import io.milvus.param.IndexType;
|
||||
import io.milvus.param.MetricType;
|
||||
import io.milvus.param.R;
|
||||
import io.milvus.param.RpcStatus;
|
||||
import io.milvus.param.collection.*;
|
||||
import io.milvus.param.dml.InsertParam;
|
||||
import io.milvus.param.dml.SearchParam;
|
||||
import io.milvus.param.index.CreateIndexParam;
|
||||
import io.milvus.response.SearchResultsWrapper;
|
||||
import org.junit.jupiter.api.*;
|
||||
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@DisplayName("Milvus 连接验证")
|
||||
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
|
||||
class MilvusConnectionTest {
|
||||
|
||||
private static final String COLLECTION = "conn_test";
|
||||
private static final int DIM = 128;
|
||||
|
||||
private static MilvusServiceClient client;
|
||||
|
||||
@BeforeAll
|
||||
static void connect() {
|
||||
String host = envOrDefault("MILVUS_HOST",
|
||||
"in03-4a578da0f27ce9d.serverless.aws-eu-central-1.cloud.zilliz.com");
|
||||
int port = Integer.parseInt(envOrDefault("MILVUS_PORT", "443"));
|
||||
String token = System.getenv("MILVUS_TOKEN");
|
||||
|
||||
assertNotNull(token, "环境变量 MILVUS_TOKEN 未设置");
|
||||
|
||||
ConnectParam connectParam = ConnectParam.newBuilder()
|
||||
.withHost(host)
|
||||
.withPort(port)
|
||||
.withToken(token)
|
||||
.withSecure(true)
|
||||
.withDatabaseName("db_4a578da0f27ce9d")
|
||||
.withConnectTimeout(30, TimeUnit.SECONDS)
|
||||
.build();
|
||||
|
||||
client = new MilvusServiceClient(connectParam);
|
||||
System.out.println("连接目标: " + host + ":" + port);
|
||||
}
|
||||
|
||||
@AfterAll
|
||||
static void disconnect() {
|
||||
if (client != null) {
|
||||
try {
|
||||
client.dropCollection(DropCollectionParam.newBuilder()
|
||||
.withCollectionName(COLLECTION).build());
|
||||
} catch (Exception ignored) {}
|
||||
client.close();
|
||||
}
|
||||
}
|
||||
|
||||
private static String safeMsg(R<?> resp) {
|
||||
try {
|
||||
return resp.getMessage();
|
||||
} catch (Exception e) {
|
||||
return "(no message)";
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@Order(1)
|
||||
@DisplayName("1. 连接成功 - 能列出 collection")
|
||||
void listCollections() {
|
||||
R<ShowCollectionsResponse> resp = client.showCollections(
|
||||
ShowCollectionsParam.newBuilder().build());
|
||||
|
||||
System.out.println("listCollections status: " + resp.getStatus() + ", msg: " + safeMsg(resp));
|
||||
assertEquals(0, resp.getStatus(), "连接失败,status=" + resp.getStatus());
|
||||
|
||||
List<String> names = resp.getData().getCollectionNamesList();
|
||||
System.out.println("现有 collections: " + names);
|
||||
}
|
||||
|
||||
@Test
|
||||
@Order(2)
|
||||
@DisplayName("2. 创建测试 collection")
|
||||
void createCollection() {
|
||||
client.dropCollection(DropCollectionParam.newBuilder()
|
||||
.withCollectionName(COLLECTION).build());
|
||||
|
||||
FieldType idField = FieldType.newBuilder()
|
||||
.withName("id")
|
||||
.withDataType(DataType.Int64)
|
||||
.withPrimaryKey(true)
|
||||
.withAutoID(true)
|
||||
.build();
|
||||
|
||||
FieldType vectorField = FieldType.newBuilder()
|
||||
.withName("vector")
|
||||
.withDataType(DataType.FloatVector)
|
||||
.withDimension(DIM)
|
||||
.build();
|
||||
|
||||
CollectionSchemaParam schema = CollectionSchemaParam.newBuilder()
|
||||
.addFieldType(idField)
|
||||
.addFieldType(vectorField)
|
||||
.build();
|
||||
|
||||
R<RpcStatus> resp = client.createCollection(
|
||||
CreateCollectionParam.newBuilder()
|
||||
.withCollectionName(COLLECTION)
|
||||
.withSchema(schema)
|
||||
.build());
|
||||
|
||||
System.out.println("createCollection status: " + resp.getStatus() + ", msg: " + safeMsg(resp));
|
||||
assertEquals(0, resp.getStatus(), "创建 collection 失败");
|
||||
}
|
||||
|
||||
@Test
|
||||
@Order(3)
|
||||
@DisplayName("3. 插入数据 + flush")
|
||||
void insertAndFlush() {
|
||||
List<Float> vec1 = makeVector(1.0f);
|
||||
List<Float> vec2 = makeVector(2.0f);
|
||||
List<Float> vec3 = makeVector(3.0f);
|
||||
|
||||
List<InsertParam.Field> fields = Collections.singletonList(
|
||||
new InsertParam.Field("vector", Arrays.asList(vec1, vec2, vec3))
|
||||
);
|
||||
|
||||
R<MutationResult> insertResp = client.insert(
|
||||
InsertParam.newBuilder()
|
||||
.withCollectionName(COLLECTION)
|
||||
.withFields(fields)
|
||||
.build());
|
||||
|
||||
System.out.println("insert status: " + insertResp.getStatus() + ", msg: " + safeMsg(insertResp));
|
||||
assertEquals(0, insertResp.getStatus(), "插入失败");
|
||||
|
||||
// 官方示例要求:insert 后必须 flush,数据才对搜索可见
|
||||
R<FlushResponse> flushResp = client.flush(FlushParam.newBuilder()
|
||||
.withCollectionNames(Collections.singletonList(COLLECTION))
|
||||
.withSyncFlush(true)
|
||||
.withSyncFlushWaitingTimeout(30L)
|
||||
.build());
|
||||
|
||||
System.out.println("flush status: " + flushResp.getStatus() + ", msg: " + safeMsg(flushResp));
|
||||
assertEquals(0, flushResp.getStatus(), "flush 失败");
|
||||
System.out.println("插入 3 条数据并 flush 完成");
|
||||
}
|
||||
|
||||
@Test
|
||||
@Order(4)
|
||||
@DisplayName("4. 创建索引 + 加载")
|
||||
void createIndexAndLoad() {
|
||||
R<RpcStatus> indexResp = client.createIndex(
|
||||
CreateIndexParam.newBuilder()
|
||||
.withCollectionName(COLLECTION)
|
||||
.withFieldName("vector")
|
||||
.withIndexType(IndexType.AUTOINDEX)
|
||||
.withMetricType(MetricType.L2)
|
||||
.build());
|
||||
|
||||
System.out.println("createIndex status: " + indexResp.getStatus() + ", msg: " + safeMsg(indexResp));
|
||||
assertEquals(0, indexResp.getStatus(), "创建索引失败");
|
||||
|
||||
R<RpcStatus> loadResp = client.loadCollection(
|
||||
LoadCollectionParam.newBuilder()
|
||||
.withCollectionName(COLLECTION)
|
||||
.withSyncLoad(true)
|
||||
.withSyncLoadWaitingTimeout(30L)
|
||||
.build());
|
||||
|
||||
System.out.println("load status: " + loadResp.getStatus() + ", msg: " + safeMsg(loadResp));
|
||||
assertEquals(0, loadResp.getStatus(), "加载失败");
|
||||
System.out.println("索引创建 + 加载完成");
|
||||
}
|
||||
|
||||
@Test
|
||||
@Order(5)
|
||||
@DisplayName("5. 向量搜索")
|
||||
void search() throws InterruptedException {
|
||||
Thread.sleep(3000);
|
||||
|
||||
List<Float> queryVec = makeVector(1.1f);
|
||||
|
||||
R<SearchResults> resp = null;
|
||||
for (int retry = 0; retry < 10; retry++) {
|
||||
resp = client.search(
|
||||
SearchParam.newBuilder()
|
||||
.withCollectionName(COLLECTION)
|
||||
.withMetricType(MetricType.L2)
|
||||
.withTopK(2)
|
||||
.withVectors(Collections.singletonList(queryVec))
|
||||
.withVectorFieldName("vector")
|
||||
.withParams("{}")
|
||||
.withConsistencyLevel(ConsistencyLevelEnum.STRONG)
|
||||
.build());
|
||||
|
||||
if (resp.getStatus() == 0) break;
|
||||
System.out.println("search retry " + (retry + 1) + ": status=" + resp.getStatus() + ", msg=" + safeMsg(resp));
|
||||
Thread.sleep(5000);
|
||||
}
|
||||
|
||||
System.out.println("search status: " + resp.getStatus() + ", msg: " + safeMsg(resp));
|
||||
assertEquals(0, resp.getStatus(), "搜索失败");
|
||||
|
||||
SearchResultsWrapper wrapper = new SearchResultsWrapper(resp.getData().getResults());
|
||||
List<SearchResultsWrapper.IDScore> scores = wrapper.getIDScore(0);
|
||||
|
||||
assertFalse(scores.isEmpty(), "搜索结果不应为空");
|
||||
System.out.println("搜索结果 (top " + scores.size() + "):");
|
||||
for (SearchResultsWrapper.IDScore idScore : scores) {
|
||||
System.out.println(" score=" + idScore.getScore() + ", id=" + idScore.getLongID());
|
||||
}
|
||||
}
|
||||
|
||||
private static List<Float> makeVector(float val) {
|
||||
Float[] arr = new Float[DIM];
|
||||
Arrays.fill(arr, val);
|
||||
return Arrays.asList(arr);
|
||||
}
|
||||
|
||||
private static String envOrDefault(String key, String defaultVal) {
|
||||
String val = System.getenv(key);
|
||||
return (val != null && !val.isEmpty()) ? val : defaultVal;
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user