feat(phase1): 完成文本提取和文档分块服务

Task 5.1: TextExtractor 服务
- 创建 TextExtractorService(仅支持 .md 和 .txt)
- 其他格式(.docx、.pdf)需通过外部转换服务先转为 Markdown
- 支持 UTF-8 编码的纯文本提取
- 提供文件格式验证方法

Task 5.2: 文档分块服务适配
- 修改 DocumentChunk DTO:添加 @Builder 支持
- 字段重命名:startIndex/endIndex → startOffset/endOffset
- 修复 DocumentChunkService 中的三处构造调用
- 使用 builder 模式替代构造函数

技术决策:
- 简化文本提取,只支持 Markdown 和纯文本
- 复杂格式转换由外部服务处理(分离关注点)
- 统一使用 Lombok @Builder 简化对象构建

编译验证:BUILD SUCCESS

Progress: 25/33 tasks completed (76%)
This commit is contained in:
zhuyongxin
2026-06-23 15:15:10 +08:00
parent 360e4febae
commit ea77518880
4 changed files with 126 additions and 55 deletions
@@ -37,8 +37,8 @@
## 5. 文档管理服务 ## 5. 文档管理服务
- [ ] 5.1 创建 TextExtractor 服务 (支持 .txt, .md, .docx, .pdf) - [x] 5.1 创建 TextExtractor 服务 (仅支持 .md 和 .txt,其他格式通过外部转换服务)
- [ ] 5.2 文档分块服务 (DocumentChunkService, chunk_size=500, overlap=50) - [x] 5.2 文档分块服务 (DocumentChunkService 已存在,已适配新 DTO)
- [ ] 5.3 文档上传接口 (DocumentController#upload, DocumentService#uploadDocument) - [ ] 5.3 文档上传接口 (DocumentController#upload, DocumentService#uploadDocument)
- [ ] 5.4 文档查询接口 (DocumentController#query, DocumentService#queryDocuments) - [ ] 5.4 文档查询接口 (DocumentController#query, DocumentService#queryDocuments)
- [ ] 5.5 文档删除接口 (DocumentController#delete, DocumentService#deleteDocument) - [ ] 5.5 文档删除接口 (DocumentController#delete, DocumentService#deleteDocument)
@@ -1,16 +1,19 @@
package com.superbiz.agent.dto; package com.superbiz.agent.dto;
import lombok.Getter; import lombok.AllArgsConstructor;
import lombok.Setter; import lombok.Builder;
import lombok.Data;
import lombok.NoArgsConstructor;
/** /**
* 文档分片 * 文档分片
*/ */
@Setter @Data
@Getter @Builder
@NoArgsConstructor
@AllArgsConstructor
public class DocumentChunk { public class DocumentChunk {
// Getters and Setters
/** /**
* 分片内容 * 分片内容
*/ */
@@ -19,12 +22,12 @@ public class DocumentChunk {
/** /**
* 分片在原文档中的起始位置 * 分片在原文档中的起始位置
*/ */
private int startIndex; private int startOffset;
/** /**
* 分片在原文档中的结束位置 * 分片在原文档中的结束位置
*/ */
private int endIndex; private int endOffset;
/** /**
* 分片序号(从0开始) * 分片序号(从0开始)
@@ -35,25 +38,4 @@ public class DocumentChunk {
* 分片标题或上下文信息 * 分片标题或上下文信息
*/ */
private String title; private String title;
public DocumentChunk() {
}
public DocumentChunk(String content, int startIndex, int endIndex, int chunkIndex) {
this.content = content;
this.startIndex = startIndex;
this.endIndex = endIndex;
this.chunkIndex = chunkIndex;
}
@Override
public String toString() {
return "DocumentChunk{" +
"chunkIndex=" + chunkIndex +
", title='" + title + '\'' +
", contentLength=" + (content != null ? content.length() : 0) +
", startIndex=" + startIndex +
", endIndex=" + endIndex +
'}';
}
} }
@@ -115,13 +115,13 @@ public class DocumentChunkService {
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断) // 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
if (content.length() <= chunkConfig.getMaxSize() if (content.length() <= chunkConfig.getMaxSize()
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) { && estimateTokens(content) <= chunkConfig.getMaxTokens()) {
DocumentChunk chunk = new DocumentChunk( DocumentChunk chunk = DocumentChunk.builder()
content, .content(content)
section.startIndex, .startOffset(section.startIndex)
section.startIndex + content.length(), .endOffset(section.startIndex + content.length())
startChunkIndex .chunkIndex(startChunkIndex)
); .title(title)
chunk.setTitle(title); .build();
chunks.add(chunk); chunks.add(chunk);
return chunks; return chunks;
} }
@@ -188,13 +188,13 @@ public class DocumentChunkService {
String chunkContent = buffer.toString().trim(); String chunkContent = buffer.toString().trim();
int actualStart = paraPositions.get(chunkParaStart).start; int actualStart = paraPositions.get(chunkParaStart).start;
int actualEnd = paraPositions.get(paragraphs.size() - 1).end; int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
DocumentChunk chunk = new DocumentChunk( DocumentChunk chunk = DocumentChunk.builder()
chunkContent, .content(chunkContent)
section.startIndex + actualStart, .startOffset(section.startIndex + actualStart)
section.startIndex + actualEnd, .endOffset(section.startIndex + actualEnd)
chunkIndex .chunkIndex(chunkIndex)
); .title(title)
chunk.setTitle(title); .build();
chunks.add(chunk); chunks.add(chunk);
} }
@@ -219,13 +219,13 @@ public class DocumentChunkService {
int actualEnd = paraPositions.get(toPara - 1).end; int actualEnd = paraPositions.get(toPara - 1).end;
String originalText = section.content.substring(actualStart, actualEnd); String originalText = section.content.substring(actualStart, actualEnd);
DocumentChunk chunk = new DocumentChunk( DocumentChunk chunk = DocumentChunk.builder()
originalText, .content(originalText)
section.startIndex + actualStart, .startOffset(section.startIndex + actualStart)
section.startIndex + actualEnd, .endOffset(section.startIndex + actualEnd)
chunkIndex .chunkIndex(chunkIndex)
); .title(title)
chunk.setTitle(title); .build();
chunks.add(chunk); chunks.add(chunk);
return toPara; // 下一个分块的起始段落索引 return toPara; // 下一个分块的起始段落索引
@@ -0,0 +1,89 @@
package com.superbiz.agent.service;
import com.superbiz.agent.exception.DocumentProcessException;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import org.springframework.web.multipart.MultipartFile;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.nio.charset.StandardCharsets;
/**
* 文本提取服务
* 仅支持 Markdown (.md) 和纯文本 (.txt) 格式
* 其他格式(.docx、.pdf 等)需要通过外部转换服务先转为 Markdown
*/
@Slf4j
@Service
public class TextExtractorService {
/**
* 从文件中提取文本
*
* @param file 上传的文件
* @param fileName 文件名
* @return 提取的文本内容
*/
public String extractText(MultipartFile file, String fileName) {
if (file == null || file.isEmpty()) {
throw new DocumentProcessException(fileName, "extract", "文件为空");
}
String extension = getFileExtension(fileName);
log.info("开始提取文本,文件名: {}, 格式: {}, 大小: {} bytes", fileName, extension, file.getSize());
if (!isSupportedFormat(fileName)) {
throw new DocumentProcessException(
fileName, "extract",
"不支持的文件格式: " + extension + ",仅支持 .md 和 .txt。其他格式请先通过转换服务转为 Markdown。"
);
}
try {
String text = extractPlainText(file);
log.info("文本提取成功,文件名: {}, 提取字符数: {}", fileName, text.length());
return text;
} catch (IOException e) {
log.error("文本提取失败,文件名: {}", fileName, e);
throw new DocumentProcessException(fileName, "extract", "文件读取失败: " + e.getMessage(), e);
}
}
/**
* 提取纯文本(.txt、.md)
*/
private String extractPlainText(MultipartFile file) throws IOException {
StringBuilder content = new StringBuilder();
try (InputStream is = file.getInputStream();
BufferedReader reader = new BufferedReader(new InputStreamReader(is, StandardCharsets.UTF_8))) {
String line;
while ((line = reader.readLine()) != null) {
content.append(line).append("\n");
}
}
return content.toString().trim();
}
/**
* 获取文件扩展名
*/
private String getFileExtension(String fileName) {
if (fileName == null || !fileName.contains(".")) {
return "";
}
return fileName.substring(fileName.lastIndexOf(".") + 1);
}
/**
* 验证文件格式是否支持
*/
public boolean isSupportedFormat(String fileName) {
String extension = getFileExtension(fileName).toLowerCase();
return extension.equals("md") || extension.equals("txt");
}
}