Files
SuperBizAgent-java/src/main/java/com/superbiz/agent/service/TextExtractorService.java
T
zhuyongxin ea77518880 feat(phase1): 完成文本提取和文档分块服务
Task 5.1: TextExtractor 服务
- 创建 TextExtractorService(仅支持 .md 和 .txt)
- 其他格式(.docx、.pdf)需通过外部转换服务先转为 Markdown
- 支持 UTF-8 编码的纯文本提取
- 提供文件格式验证方法

Task 5.2: 文档分块服务适配
- 修改 DocumentChunk DTO:添加 @Builder 支持
- 字段重命名:startIndex/endIndex → startOffset/endOffset
- 修复 DocumentChunkService 中的三处构造调用
- 使用 builder 模式替代构造函数

技术决策:
- 简化文本提取,只支持 Markdown 和纯文本
- 复杂格式转换由外部服务处理(分离关注点)
- 统一使用 Lombok @Builder 简化对象构建

编译验证:BUILD SUCCESS

Progress: 25/33 tasks completed (76%)
2026-06-23 15:15:10 +08:00

90 lines
2.9 KiB
Java
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package com.superbiz.agent.service;
import com.superbiz.agent.exception.DocumentProcessException;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import org.springframework.web.multipart.MultipartFile;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.nio.charset.StandardCharsets;
/**
* 文本提取服务
* 仅支持 Markdown (.md) 和纯文本 (.txt) 格式
* 其他格式(.docx、.pdf 等)需要通过外部转换服务先转为 Markdown
*/
@Slf4j
@Service
public class TextExtractorService {
/**
* 从文件中提取文本
*
* @param file 上传的文件
* @param fileName 文件名
* @return 提取的文本内容
*/
public String extractText(MultipartFile file, String fileName) {
if (file == null || file.isEmpty()) {
throw new DocumentProcessException(fileName, "extract", "文件为空");
}
String extension = getFileExtension(fileName);
log.info("开始提取文本,文件名: {}, 格式: {}, 大小: {} bytes", fileName, extension, file.getSize());
if (!isSupportedFormat(fileName)) {
throw new DocumentProcessException(
fileName, "extract",
"不支持的文件格式: " + extension + ",仅支持 .md 和 .txt。其他格式请先通过转换服务转为 Markdown。"
);
}
try {
String text = extractPlainText(file);
log.info("文本提取成功,文件名: {}, 提取字符数: {}", fileName, text.length());
return text;
} catch (IOException e) {
log.error("文本提取失败,文件名: {}", fileName, e);
throw new DocumentProcessException(fileName, "extract", "文件读取失败: " + e.getMessage(), e);
}
}
/**
* 提取纯文本(.txt、.md)
*/
private String extractPlainText(MultipartFile file) throws IOException {
StringBuilder content = new StringBuilder();
try (InputStream is = file.getInputStream();
BufferedReader reader = new BufferedReader(new InputStreamReader(is, StandardCharsets.UTF_8))) {
String line;
while ((line = reader.readLine()) != null) {
content.append(line).append("\n");
}
}
return content.toString().trim();
}
/**
* 获取文件扩展名
*/
private String getFileExtension(String fileName) {
if (fileName == null || !fileName.contains(".")) {
return "";
}
return fileName.substring(fileName.lastIndexOf(".") + 1);
}
/**
* 验证文件格式是否支持
*/
public boolean isSupportedFormat(String fileName) {
String extension = getFileExtension(fileName).toLowerCase();
return extension.equals("md") || extension.equals("txt");
}
}