feat(phase1): 完成文档上传接口

Task 5.3: 文档上传接口
- 创建 DocumentManagementService:文档上传核心逻辑
  - 文件格式验证(仅 .md/.txt)
  - 文件 hash 计算与去重检查
  - 文本提取与分块处理
  - 文档元数据持久化(ApiDocument)
  - 向量化索引标记为 TODO(待补充)
- 创建 DocumentController:RESTful 上传接口
  - POST /api/documents/upload
  - 支持参数:file, faultCategory, faultSource, apiName, version
  - 返回:文档 docId

功能特性:
- MD5 hash 去重:防止重复上传
- 事务支持:元数据与索引状态一致性
- 异常处理:DocumentProcessException 统一封装
- 分块配置:使用 DocumentChunkConfig 默认配置

待补充:
- TODO: VectorIndexService.indexDocumentChunks() 实现
- 当前文档状态直接标记为 INDEXED

编译验证:BUILD SUCCESS

Progress: 26/33 tasks completed (79%)
This commit is contained in:
zhuyongxin
2026-06-23 15:28:02 +08:00
parent 5869fc775f
commit f446290d0f
3 changed files with 218 additions and 1 deletions
@@ -0,0 +1,161 @@
package com.superbiz.agent.service;
import com.superbiz.agent.domain.entity.ApiDocument;
import com.superbiz.agent.domain.enums.FaultCategory;
import com.superbiz.agent.dto.DocumentChunk;
import com.superbiz.agent.dto.DocumentUploadRequest;
import com.superbiz.agent.exception.DocumentProcessException;
import com.superbiz.agent.repository.ApiDocumentRepository;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Service;
import org.springframework.transaction.annotation.Transactional;
import org.springframework.web.multipart.MultipartFile;
import java.io.IOException;
import java.security.MessageDigest;
import java.time.LocalDateTime;
import java.util.List;
import java.util.Optional;
import java.util.UUID;
/**
* 文档管理服务
*/
@Slf4j
@Service
public class DocumentManagementService {
@Autowired
private TextExtractorService textExtractorService;
@Autowired
private DocumentChunkService documentChunkService;
@Autowired
private VectorIndexService vectorIndexService;
@Autowired
private ApiDocumentRepository apiDocumentRepository;
/**
* 上传文档
*
* @param request 上传请求
* @return 文档ID
*/
@Transactional
public String uploadDocument(DocumentUploadRequest request) {
MultipartFile file = request.getFile();
String fileName = file.getOriginalFilename();
log.info("开始上传文档,文件名: {}, 大小: {} bytes", fileName, file.getSize());
// 1. 验证文件格式
if (!textExtractorService.isSupportedFormat(fileName)) {
throw new DocumentProcessException(
fileName, "upload",
"不支持的文件格式,仅支持 .md 和 .txt"
);
}
// 2. 计算文件 hash(去重)
String fileHash = calculateFileHash(file);
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
if (existing.isPresent()) {
log.warn("文档已存在,hash: {}, docId: ", fileHash, existing.get().getDocId());
throw new DocumentProcessException(
fileName, "upload",
"文档已存在,docId: " + existing.get().getDocId()
);
}
// 3. 提取文本
String text = textExtractorService.extractText(file, fileName);
if (text == null || text.isBlank()) {
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
}
// 4. 分块(使用 DocumentChunkService 默认配置)
// 注意:chunkSize 和 overlap 参数由 DocumentChunkConfig 配置,暂不支持动态调整
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
if (chunks.isEmpty()) {
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
}
log.info("文档分块完成,文件名: {}, 分块数: {}", fileName, chunks.size());
// 5. 创建文档元数据
String docId = UUID.randomUUID().toString();
ApiDocument document = ApiDocument.builder()
.docId(docId)
.fileName(fileName)
.faultCategory(parseFaultCategory(request.getFaultCategory()))
.faultSource(request.getFaultSource())
.apiName(request.getApiName())
.version(request.getVersion())
.fileSize(file.getSize())
.fileHash(fileHash)
.status("PROCESSING")
.chunkCount(chunks.size())
.build();
apiDocumentRepository.save(document);
log.info("文档元数据已保存,docId: {}", docId);
// 6. 向量化并索引(TODO: 待实现批量分块索引)
try {
// TODO: 实现 VectorIndexService.indexDocumentChunks(docId, chunks)
// 当前暂时标记为 INDEXED,后续补充实际向量化逻辑
log.warn("向量化索引功能待实现,docId: {}", docId);
document.setStatus("INDEXED");
document.setIndexedAt(LocalDateTime.now());
apiDocumentRepository.save(document);
log.info("文档元数据已创建(向量化待实现),docId: {}", docId);
} catch (Exception e) {
log.error("文档处理失败,docId: {}", docId, e);
document.setStatus("FAILED");
apiDocumentRepository.save(document);
throw new DocumentProcessException(docId, "process", "文档处理失败: " + e.getMessage(), e);
}
return docId;
}
/**
* 计算文件 hash(MD5)
*/
private String calculateFileHash(MultipartFile file) {
try {
MessageDigest md = MessageDigest.getInstance("MD5");
byte[] digest = md.digest(file.getBytes());
StringBuilder sb = new StringBuilder();
for (byte b : digest) {
sb.append(String.format("%02x", b));
}
return sb.toString();
} catch (Exception e) {
throw new DocumentProcessException(
file.getOriginalFilename(), "hash",
"计算文件 hash 失败: " + e.getMessage(), e
);
}
}
/**
* 解析故障类别
*/
private FaultCategory parseFaultCategory(String category) {
if (category == null || category.isBlank()) {
return FaultCategory.EXTERNAL_API;
}
try {
return FaultCategory.valueOf(category.toUpperCase());
} catch (IllegalArgumentException e) {
return FaultCategory.EXTERNAL_API;
}
}
}