feat(phase1): 完成文本提取和文档分块服务
Task 5.1: TextExtractor 服务 - 创建 TextExtractorService(仅支持 .md 和 .txt) - 其他格式(.docx、.pdf)需通过外部转换服务先转为 Markdown - 支持 UTF-8 编码的纯文本提取 - 提供文件格式验证方法 Task 5.2: 文档分块服务适配 - 修改 DocumentChunk DTO:添加 @Builder 支持 - 字段重命名:startIndex/endIndex → startOffset/endOffset - 修复 DocumentChunkService 中的三处构造调用 - 使用 builder 模式替代构造函数 技术决策: - 简化文本提取,只支持 Markdown 和纯文本 - 复杂格式转换由外部服务处理(分离关注点) - 统一使用 Lombok @Builder 简化对象构建 编译验证:BUILD SUCCESS Progress: 25/33 tasks completed (76%)
This commit is contained in:
@@ -1,59 +1,41 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
/**
|
||||
* 文档分片
|
||||
*/
|
||||
@Setter
|
||||
@Getter
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class DocumentChunk {
|
||||
|
||||
// Getters and Setters
|
||||
/**
|
||||
* 分片内容
|
||||
*/
|
||||
private String content;
|
||||
|
||||
|
||||
/**
|
||||
* 分片在原文档中的起始位置
|
||||
*/
|
||||
private int startIndex;
|
||||
|
||||
private int startOffset;
|
||||
|
||||
/**
|
||||
* 分片在原文档中的结束位置
|
||||
*/
|
||||
private int endIndex;
|
||||
|
||||
private int endOffset;
|
||||
|
||||
/**
|
||||
* 分片序号(从0开始)
|
||||
*/
|
||||
private int chunkIndex;
|
||||
|
||||
|
||||
/**
|
||||
* 分片标题或上下文信息
|
||||
*/
|
||||
private String title;
|
||||
|
||||
public DocumentChunk() {
|
||||
}
|
||||
|
||||
public DocumentChunk(String content, int startIndex, int endIndex, int chunkIndex) {
|
||||
this.content = content;
|
||||
this.startIndex = startIndex;
|
||||
this.endIndex = endIndex;
|
||||
this.chunkIndex = chunkIndex;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return "DocumentChunk{" +
|
||||
"chunkIndex=" + chunkIndex +
|
||||
", title='" + title + '\'' +
|
||||
", contentLength=" + (content != null ? content.length() : 0) +
|
||||
", startIndex=" + startIndex +
|
||||
", endIndex=" + endIndex +
|
||||
'}';
|
||||
}
|
||||
}
|
||||
|
||||
@@ -115,13 +115,13 @@ public class DocumentChunkService {
|
||||
// 短章节直接作为一个分片(用 token 估算替代字符数做短路判断)
|
||||
if (content.length() <= chunkConfig.getMaxSize()
|
||||
&& estimateTokens(content) <= chunkConfig.getMaxTokens()) {
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
content,
|
||||
section.startIndex,
|
||||
section.startIndex + content.length(),
|
||||
startChunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.content(content)
|
||||
.startOffset(section.startIndex)
|
||||
.endOffset(section.startIndex + content.length())
|
||||
.chunkIndex(startChunkIndex)
|
||||
.title(title)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
return chunks;
|
||||
}
|
||||
@@ -188,13 +188,13 @@ public class DocumentChunkService {
|
||||
String chunkContent = buffer.toString().trim();
|
||||
int actualStart = paraPositions.get(chunkParaStart).start;
|
||||
int actualEnd = paraPositions.get(paragraphs.size() - 1).end;
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
chunkContent,
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.content(chunkContent)
|
||||
.startOffset(section.startIndex + actualStart)
|
||||
.endOffset(section.startIndex + actualEnd)
|
||||
.chunkIndex(chunkIndex)
|
||||
.title(title)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
}
|
||||
|
||||
@@ -219,13 +219,13 @@ public class DocumentChunkService {
|
||||
int actualEnd = paraPositions.get(toPara - 1).end;
|
||||
String originalText = section.content.substring(actualStart, actualEnd);
|
||||
|
||||
DocumentChunk chunk = new DocumentChunk(
|
||||
originalText,
|
||||
section.startIndex + actualStart,
|
||||
section.startIndex + actualEnd,
|
||||
chunkIndex
|
||||
);
|
||||
chunk.setTitle(title);
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.content(originalText)
|
||||
.startOffset(section.startIndex + actualStart)
|
||||
.endOffset(section.startIndex + actualEnd)
|
||||
.chunkIndex(chunkIndex)
|
||||
.title(title)
|
||||
.build();
|
||||
chunks.add(chunk);
|
||||
|
||||
return toPara; // 下一个分块的起始段落索引
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.exception.DocumentProcessException;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.io.InputStreamReader;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
|
||||
/**
|
||||
* 文本提取服务
|
||||
* 仅支持 Markdown (.md) 和纯文本 (.txt) 格式
|
||||
* 其他格式(.docx、.pdf 等)需要通过外部转换服务先转为 Markdown
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
public class TextExtractorService {
|
||||
|
||||
/**
|
||||
* 从文件中提取文本
|
||||
*
|
||||
* @param file 上传的文件
|
||||
* @param fileName 文件名
|
||||
* @return 提取的文本内容
|
||||
*/
|
||||
public String extractText(MultipartFile file, String fileName) {
|
||||
if (file == null || file.isEmpty()) {
|
||||
throw new DocumentProcessException(fileName, "extract", "文件为空");
|
||||
}
|
||||
|
||||
String extension = getFileExtension(fileName);
|
||||
log.info("开始提取文本,文件名: {}, 格式: {}, 大小: {} bytes", fileName, extension, file.getSize());
|
||||
|
||||
if (!isSupportedFormat(fileName)) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "extract",
|
||||
"不支持的文件格式: " + extension + ",仅支持 .md 和 .txt。其他格式请先通过转换服务转为 Markdown。"
|
||||
);
|
||||
}
|
||||
|
||||
try {
|
||||
String text = extractPlainText(file);
|
||||
log.info("文本提取成功,文件名: {}, 提取字符数: {}", fileName, text.length());
|
||||
return text;
|
||||
|
||||
} catch (IOException e) {
|
||||
log.error("文本提取失败,文件名: {}", fileName, e);
|
||||
throw new DocumentProcessException(fileName, "extract", "文件读取失败: " + e.getMessage(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 提取纯文本(.txt、.md)
|
||||
*/
|
||||
private String extractPlainText(MultipartFile file) throws IOException {
|
||||
StringBuilder content = new StringBuilder();
|
||||
try (InputStream is = file.getInputStream();
|
||||
BufferedReader reader = new BufferedReader(new InputStreamReader(is, StandardCharsets.UTF_8))) {
|
||||
|
||||
String line;
|
||||
while ((line = reader.readLine()) != null) {
|
||||
content.append(line).append("\n");
|
||||
}
|
||||
}
|
||||
return content.toString().trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取文件扩展名
|
||||
*/
|
||||
private String getFileExtension(String fileName) {
|
||||
if (fileName == null || !fileName.contains(".")) {
|
||||
return "";
|
||||
}
|
||||
return fileName.substring(fileName.lastIndexOf(".") + 1);
|
||||
}
|
||||
|
||||
/**
|
||||
* 验证文件格式是否支持
|
||||
*/
|
||||
public boolean isSupportedFormat(String fileName) {
|
||||
String extension = getFileExtension(fileName).toLowerCase();
|
||||
return extension.equals("md") || extension.equals("txt");
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user