feat(phase1): 完成文本提取和文档分块服务

Task 5.1: TextExtractor 服务
- 创建 TextExtractorService(仅支持 .md 和 .txt)
- 其他格式(.docx、.pdf)需通过外部转换服务先转为 Markdown
- 支持 UTF-8 编码的纯文本提取
- 提供文件格式验证方法

Task 5.2: 文档分块服务适配
- 修改 DocumentChunk DTO:添加 @Builder 支持
- 字段重命名:startIndex/endIndex → startOffset/endOffset
- 修复 DocumentChunkService 中的三处构造调用
- 使用 builder 模式替代构造函数

技术决策:
- 简化文本提取,只支持 Markdown 和纯文本
- 复杂格式转换由外部服务处理(分离关注点)
- 统一使用 Lombok @Builder 简化对象构建

编译验证:BUILD SUCCESS

Progress: 25/33 tasks completed (76%)
This commit is contained in:
zhuyongxin
2026-06-23 15:15:10 +08:00
parent 360e4febae
commit ea77518880
4 changed files with 126 additions and 55 deletions
@@ -115,13 +115,13 @@ public class DocumentChunkService {
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
if (content.length() <= chunkConfig.getMaxSize()
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) {
DocumentChunk chunk = new DocumentChunk(
content,
section.startIndex,
section.startIndex + content.length(),
startChunkIndex
);
chunk.setTitle(title);
DocumentChunk chunk = DocumentChunk.builder()
.content(content)
.startOffset(section.startIndex)
.endOffset(section.startIndex + content.length())
.chunkIndex(startChunkIndex)
.title(title)
.build();
chunks.add(chunk);
return chunks;
}
@@ -188,13 +188,13 @@ public class DocumentChunkService {
String chunkContent = buffer.toString().trim();
int actualStart = paraPositions.get(chunkParaStart).start;
int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
DocumentChunk chunk = new DocumentChunk(
chunkContent,
section.startIndex + actualStart,
section.startIndex + actualEnd,
chunkIndex
);
chunk.setTitle(title);
DocumentChunk chunk = DocumentChunk.builder()
.content(chunkContent)
.startOffset(section.startIndex + actualStart)
.endOffset(section.startIndex + actualEnd)
.chunkIndex(chunkIndex)
.title(title)
.build();
chunks.add(chunk);
}
@@ -219,13 +219,13 @@ public class DocumentChunkService {
int actualEnd = paraPositions.get(toPara - 1).end;
String originalText = section.content.substring(actualStart, actualEnd);
DocumentChunk chunk = new DocumentChunk(
originalText,
section.startIndex + actualStart,
section.startIndex + actualEnd,
chunkIndex
);
chunk.setTitle(title);
DocumentChunk chunk = DocumentChunk.builder()
.content(originalText)
.startOffset(section.startIndex + actualStart)
.endOffset(section.startIndex + actualEnd)
.chunkIndex(chunkIndex)
.title(title)
.build();
chunks.add(chunk);
return toPara; // 下一个分块的起始段落索引