feat(phase1): 完成文本提取和文档分块服务
Task 5.1: TextExtractor 服务 - 创建 TextExtractorService(仅支持 .md 和 .txt) - 其他格式(.docx、.pdf)需通过外部转换服务先转为 Markdown - 支持 UTF-8 编码的纯文本提取 - 提供文件格式验证方法 Task 5.2: 文档分块服务适配 - 修改 DocumentChunk DTO:添加 @Builder 支持 - 字段重命名:startIndex/endIndex → startOffset/endOffset - 修复 DocumentChunkService 中的三处构造调用 - 使用 builder 模式替代构造函数 技术决策: - 简化文本提取,只支持 Markdown 和纯文本 - 复杂格式转换由外部服务处理(分离关注点) - 统一使用 Lombok @Builder 简化对象构建 编译验证:BUILD SUCCESS Progress: 25/33 tasks completed (76%)
This commit is contained in:
@@ -115,13 +115,13 @@ public class DocumentChunkService {
|
||||
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
|
||||
if (content.length() <= chunkConfig.getMaxSize()
|
||||
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) {
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
content,
|
||||
section.startIndex,
|
||||
section.startIndex + content.length(),
|
||||
startChunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.content(content)
|
||||
.startOffset(section.startIndex)
|
||||
.endOffset(section.startIndex + content.length())
|
||||
.chunkIndex(startChunkIndex)
|
||||
.title(title)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
return chunks;
|
||||
}
|
||||
@@ -188,13 +188,13 @@ public class DocumentChunkService {
|
||||
String chunkContent = buffer.toString().trim();
|
||||
int actualStart = paraPositions.get(chunkParaStart).start;
|
||||
int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
chunkContent,
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.content(chunkContent)
|
||||
.startOffset(section.startIndex + actualStart)
|
||||
.endOffset(section.startIndex + actualEnd)
|
||||
.chunkIndex(chunkIndex)
|
||||
.title(title)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
}
|
||||
|
||||
@@ -219,13 +219,13 @@ public class DocumentChunkService {
|
||||
int actualEnd = paraPositions.get(toPara - 1).end;
|
||||
String originalText = section.content.substring(actualStart, actualEnd);
|
||||
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
originalText,
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.content(originalText)
|
||||
.startOffset(section.startIndex + actualStart)
|
||||
.endOffset(section.startIndex + actualEnd)
|
||||
.chunkIndex(chunkIndex)
|
||||
.title(title)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
|
||||
return toPara; // 下一个分块的起始段落索引
|
||||
|
||||
Reference in New Issue
Block a user