Task 5.1: TextExtractor 服务 - 创建 TextExtractorService(仅支持 .md 和 .txt) - 其他格式(.docx、.pdf)需通过外部转换服务先转为 Markdown - 支持 UTF-8 编码的纯文本提取 - 提供文件格式验证方法 Task 5.2: 文档分块服务适配 - 修改 DocumentChunk DTO:添加 @Builder 支持 - 字段重命名:startIndex/endIndex → startOffset/endOffset - 修复 DocumentChunkService 中的三处构造调用 - 使用 builder 模式替代构造函数 技术决策: - 简化文本提取,只支持 Markdown 和纯文本 - 复杂格式转换由外部服务处理(分离关注点) - 统一使用 Lombok @Builder 简化对象构建 编译验证:BUILD SUCCESS Progress: 25/33 tasks completed (76%)
90 lines
2.9 KiB
Java
90 lines
2.9 KiB
Java
package com.superbiz.agent.service;
|
||
|
||
import com.superbiz.agent.exception.DocumentProcessException;
|
||
import lombok.extern.slf4j.Slf4j;
|
||
import org.springframework.stereotype.Service;
|
||
import org.springframework.web.multipart.MultipartFile;
|
||
|
||
import java.io.BufferedReader;
|
||
import java.io.IOException;
|
||
import java.io.InputStream;
|
||
import java.io.InputStreamReader;
|
||
import java.nio.charset.StandardCharsets;
|
||
|
||
/**
|
||
* 文本提取服务
|
||
* 仅支持 Markdown (.md) 和纯文本 (.txt) 格式
|
||
* 其他格式(.docx、.pdf 等)需要通过外部转换服务先转为 Markdown
|
||
*/
|
||
@Slf4j
|
||
@Service
|
||
public class TextExtractorService {
|
||
|
||
/**
|
||
* 从文件中提取文本
|
||
*
|
||
* @param file 上传的文件
|
||
* @param fileName 文件名
|
||
* @return 提取的文本内容
|
||
*/
|
||
public String extractText(MultipartFile file, String fileName) {
|
||
if (file == null || file.isEmpty()) {
|
||
throw new DocumentProcessException(fileName, "extract", "文件为空");
|
||
}
|
||
|
||
String extension = getFileExtension(fileName);
|
||
log.info("开始提取文本,文件名: {}, 格式: {}, 大小: {} bytes", fileName, extension, file.getSize());
|
||
|
||
if (!isSupportedFormat(fileName)) {
|
||
throw new DocumentProcessException(
|
||
fileName, "extract",
|
||
"不支持的文件格式: " + extension + ",仅支持 .md 和 .txt。其他格式请先通过转换服务转为 Markdown。"
|
||
);
|
||
}
|
||
|
||
try {
|
||
String text = extractPlainText(file);
|
||
log.info("文本提取成功,文件名: {}, 提取字符数: {}", fileName, text.length());
|
||
return text;
|
||
|
||
} catch (IOException e) {
|
||
log.error("文本提取失败,文件名: {}", fileName, e);
|
||
throw new DocumentProcessException(fileName, "extract", "文件读取失败: " + e.getMessage(), e);
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 提取纯文本(.txt、.md)
|
||
*/
|
||
private String extractPlainText(MultipartFile file) throws IOException {
|
||
StringBuilder content = new StringBuilder();
|
||
try (InputStream is = file.getInputStream();
|
||
BufferedReader reader = new BufferedReader(new InputStreamReader(is, StandardCharsets.UTF_8))) {
|
||
|
||
String line;
|
||
while ((line = reader.readLine()) != null) {
|
||
content.append(line).append("\n");
|
||
}
|
||
}
|
||
return content.toString().trim();
|
||
}
|
||
|
||
/**
|
||
* 获取文件扩展名
|
||
*/
|
||
private String getFileExtension(String fileName) {
|
||
if (fileName == null || !fileName.contains(".")) {
|
||
return "";
|
||
}
|
||
return fileName.substring(fileName.lastIndexOf(".") + 1);
|
||
}
|
||
|
||
/**
|
||
* 验证文件格式是否支持
|
||
*/
|
||
public boolean isSupportedFormat(String fileName) {
|
||
String extension = getFileExtension(fileName).toLowerCase();
|
||
return extension.equals("md") || extension.equals("txt");
|
||
}
|
||
}
|