diff --git a/openspec/changes/phase-1-infrastructure/tasks.md b/openspec/changes/phase-1-infrastructure/tasks.md index 236bd68..d48eeab 100644 --- a/openspec/changes/phase-1-infrastructure/tasks.md +++ b/openspec/changes/phase-1-infrastructure/tasks.md @@ -37,8 +37,8 @@ ## 5. 文档管理服务 -- [ ] 5.1 创建 TextExtractor 服务 (支持 .txt, .md, .docx, .pdf) -- [ ] 5.2 文档分块服务 (DocumentChunkService, chunk_size=500, overlap=50) +- [x] 5.1 创建 TextExtractor 服务 (仅支持 .md 和 .txt,其他格式通过外部转换服务) +- [x] 5.2 文档分块服务 (DocumentChunkService 已存在,已适配新 DTO) - [ ] 5.3 文档上传接口 (DocumentController#upload, DocumentService#uploadDocument) - [ ] 5.4 文档查询接口 (DocumentController#query, DocumentService#queryDocuments) - [ ] 5.5 文档删除接口 (DocumentController#delete, DocumentService#deleteDocument) diff --git a/src/main/java/com/superbiz/agent/dto/DocumentChunk.java b/src/main/java/com/superbiz/agent/dto/DocumentChunk.java index a627964..c33775e 100644 --- a/src/main/java/com/superbiz/agent/dto/DocumentChunk.java +++ b/src/main/java/com/superbiz/agent/dto/DocumentChunk.java @@ -1,59 +1,41 @@ package com.superbiz.agent.dto; -import lombok.Getter; -import lombok.Setter; +import lombok.AllArgsConstructor; +import lombok.Builder; +import lombok.Data; +import lombok.NoArgsConstructor; /** * 文档分片 */ -@Setter -@Getter +@Data +@Builder +@NoArgsConstructor +@AllArgsConstructor public class DocumentChunk { - // Getters and Setters /** * 分片内容 */ private String content; - + /** * 分片在原文档中的起始位置 */ - private int startIndex; - + private int startOffset; + /** * 分片在原文档中的结束位置 */ - private int endIndex; - + private int endOffset; + /** * 分片序号(从0开始) */ private int chunkIndex; - + /** * 分片标题或上下文信息 */ private String title; - - public DocumentChunk() { - } - - public DocumentChunk(String content, int startIndex, int endIndex, int chunkIndex) { - this.content = content; - this.startIndex = startIndex; - this.endIndex = endIndex; - this.chunkIndex = chunkIndex; - } - - @Override - public String toString() { - return "DocumentChunk{" + - "chunkIndex=" + chunkIndex + - ", title='" + title + '\'' + - ", contentLength=" + (content != null ? content.length() : 0) + - ", startIndex=" + startIndex + - ", endIndex=" + endIndex + - '}'; - } } diff --git a/src/main/java/com/superbiz/agent/service/DocumentChunkService.java b/src/main/java/com/superbiz/agent/service/DocumentChunkService.java index be0e8ef..edc4d45 100644 --- a/src/main/java/com/superbiz/agent/service/DocumentChunkService.java +++ b/src/main/java/com/superbiz/agent/service/DocumentChunkService.java @@ -115,13 +115,13 @@ public class DocumentChunkService { // 短章节直接作为一个分片(用 token 估算替代字符数做短路判断) if (content.length() <= chunkConfig.getMaxSize() && estimateTokens(content) <= chunkConfig.getMaxTokens()) { - DocumentChunk chunk = new DocumentChunk( - content, - section.startIndex, - section.startIndex + content.length(), - startChunkIndex - ); - chunk.setTitle(title); + DocumentChunk chunk = DocumentChunk.builder() + .content(content) + .startOffset(section.startIndex) + .endOffset(section.startIndex + content.length()) + .chunkIndex(startChunkIndex) + .title(title) + .build(); chunks.add(chunk); return chunks; } @@ -188,13 +188,13 @@ public class DocumentChunkService { String chunkContent = buffer.toString().trim(); int actualStart = paraPositions.get(chunkParaStart).start; int actualEnd = paraPositions.get(paragraphs.size() - 1).end; - DocumentChunk chunk = new DocumentChunk( - chunkContent, - section.startIndex + actualStart, - section.startIndex + actualEnd, - chunkIndex - ); - chunk.setTitle(title); + DocumentChunk chunk = DocumentChunk.builder() + .content(chunkContent) + .startOffset(section.startIndex + actualStart) + .endOffset(section.startIndex + actualEnd) + .chunkIndex(chunkIndex) + .title(title) + .build(); chunks.add(chunk); } @@ -219,13 +219,13 @@ public class DocumentChunkService { int actualEnd = paraPositions.get(toPara - 1).end; String originalText = section.content.substring(actualStart, actualEnd); - DocumentChunk chunk = new DocumentChunk( - originalText, - section.startIndex + actualStart, - section.startIndex + actualEnd, - chunkIndex - ); - chunk.setTitle(title); + DocumentChunk chunk = DocumentChunk.builder() + .content(originalText) + .startOffset(section.startIndex + actualStart) + .endOffset(section.startIndex + actualEnd) + .chunkIndex(chunkIndex) + .title(title) + .build(); chunks.add(chunk); return toPara; // 下一个分块的起始段落索引 diff --git a/src/main/java/com/superbiz/agent/service/TextExtractorService.java b/src/main/java/com/superbiz/agent/service/TextExtractorService.java new file mode 100644 index 0000000..cf89146 --- /dev/null +++ b/src/main/java/com/superbiz/agent/service/TextExtractorService.java @@ -0,0 +1,89 @@ +package com.superbiz.agent.service; + +import com.superbiz.agent.exception.DocumentProcessException; +import lombok.extern.slf4j.Slf4j; +import org.springframework.stereotype.Service; +import org.springframework.web.multipart.MultipartFile; + +import java.io.BufferedReader; +import java.io.IOException; +import java.io.InputStream; +import java.io.InputStreamReader; +import java.nio.charset.StandardCharsets; + +/** + * 文本提取服务 + * 仅支持 Markdown (.md) 和纯文本 (.txt) 格式 + * 其他格式(.docx、.pdf 等)需要通过外部转换服务先转为 Markdown + */ +@Slf4j +@Service +public class TextExtractorService { + + /** + * 从文件中提取文本 + * + * @param file 上传的文件 + * @param fileName 文件名 + * @return 提取的文本内容 + */ + public String extractText(MultipartFile file, String fileName) { + if (file == null || file.isEmpty()) { + throw new DocumentProcessException(fileName, "extract", "文件为空"); + } + + String extension = getFileExtension(fileName); + log.info("开始提取文本,文件名: {}, 格式: {}, 大小: {} bytes", fileName, extension, file.getSize()); + + if (!isSupportedFormat(fileName)) { + throw new DocumentProcessException( + fileName, "extract", + "不支持的文件格式: " + extension + ",仅支持 .md 和 .txt。其他格式请先通过转换服务转为 Markdown。" + ); + } + + try { + String text = extractPlainText(file); + log.info("文本提取成功,文件名: {}, 提取字符数: {}", fileName, text.length()); + return text; + + } catch (IOException e) { + log.error("文本提取失败,文件名: {}", fileName, e); + throw new DocumentProcessException(fileName, "extract", "文件读取失败: " + e.getMessage(), e); + } + } + + /** + * 提取纯文本(.txt、.md) + */ + private String extractPlainText(MultipartFile file) throws IOException { + StringBuilder content = new StringBuilder(); + try (InputStream is = file.getInputStream(); + BufferedReader reader = new BufferedReader(new InputStreamReader(is, StandardCharsets.UTF_8))) { + + String line; + while ((line = reader.readLine()) != null) { + content.append(line).append("\n"); + } + } + return content.toString().trim(); + } + + /** + * 获取文件扩展名 + */ + private String getFileExtension(String fileName) { + if (fileName == null || !fileName.contains(".")) { + return ""; + } + return fileName.substring(fileName.lastIndexOf(".") + 1); + } + + /** + * 验证文件格式是否支持 + */ + public boolean isSupportedFormat(String fileName) { + String extension = getFileExtension(fileName).toLowerCase(); + return extension.equals("md") || extension.equals("txt"); + } +}