Files
SuperBizAgent-java/src/main/java/com/superbiz/agent/service/DocumentManagementService.java
T

464 lines
17 KiB
Java
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package com.superbiz.agent.service;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.superbiz.agent.domain.entity.ApiDocument;
import com.superbiz.agent.domain.enums.FaultCategory;
import com.superbiz.agent.dto.DocumentChunk;
import com.superbiz.agent.dto.DocumentQueryResponse;
import com.superbiz.agent.dto.DocumentUploadRequest;
import com.superbiz.agent.dto.Frontmatter;
import com.superbiz.agent.dto.KnowledgeEntry;
import com.superbiz.agent.exception.DocumentProcessException;
import com.superbiz.agent.repository.ApiDocumentRepository;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.PageRequest;
import org.springframework.stereotype.Service;
import org.springframework.transaction.annotation.Transactional;
import org.springframework.web.multipart.MultipartFile;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.security.MessageDigest;
import java.time.LocalDateTime;
import java.util.List;
import java.util.Optional;
import java.util.UUID;
import java.util.stream.Collectors;
/**
* 文档管理服务
*/
@Slf4j
@Service
public class DocumentManagementService {
@Value("${knowledge.base-path}")
private String knowledgeBasePath;
@Autowired
private TextExtractorService textExtractorService;
@Autowired
private DocumentChunkService documentChunkService;
@Autowired
private VectorIndexService vectorIndexService;
@Autowired
private ApiDocumentRepository apiDocumentRepository;
@Autowired
private FrontmatterParser frontmatterParser;
@Autowired
private KnowledgeIndexService knowledgeIndexService;
@Autowired
private DocumentFieldEnricher documentFieldEnricher;
@Autowired
private KnowledgeDomainService knowledgeDomainService;
@Autowired
private ObjectMapper objectMapper;
/**
* 上传文档
*
* @param request 上传请求
* @return 文档ID
*/
@Transactional
public String uploadDocument(DocumentUploadRequest request) {
MultipartFile file = request.getFile();
String fileName = file.getOriginalFilename();
String localPath = null;
long startTime = System.currentTimeMillis();
log.info("开始上传文档,文件名: {}, 大小: {} bytes", fileName, file.getSize());
try {
// 1. 验证文件格式
if (!textExtractorService.isSupportedFormat(fileName)) {
throw new DocumentProcessException(
fileName, "upload",
"不支持的文件格式,仅支持 .md 和 .txt"
);
}
// 2. 计算文件 hash(去重)
long hashStart = System.currentTimeMillis();
String fileHash = calculateFileHash(file);
log.debug("文件hash计算完成: hash={}, time={}ms", fileHash, System.currentTimeMillis() - hashStart);
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
if (existing.isPresent()) {
log.warn("文档已存在,hash: {}, docId: {}", fileHash, existing.get().getDocId());
throw new DocumentProcessException(
fileName, "upload",
"文档已存在,docId: " + existing.get().getDocId()
);
}
// 3. 提取文本
long extractStart = System.currentTimeMillis();
String text = textExtractorService.extractText(file, fileName);
log.debug("文本提取完成: length={}, time={}ms", text != null ? text.length() : 0, System.currentTimeMillis() - extractStart);
if (text == null || text.isBlank()) {
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
}
// 4. 保存原始文件到本地
String category = request.getCategory();
if (category == null || category.isBlank()) {
category = "default";
}
long saveStart = System.currentTimeMillis();
localPath = saveToLocal(file, fileName, category);
log.debug("文件保存到本地完成: path={}, time={}ms", localPath, System.currentTimeMillis() - saveStart);
// 5. 解析 frontmatter
long frontmatterStart = System.currentTimeMillis();
Frontmatter frontmatter = null;
String bodyText = text;
if (frontmatterParser.hasFrontmatter(text)) {
frontmatter = frontmatterParser.parse(text);
if (frontmatter != null) {
// LLM 补全 covers / whenToRetrieve(已有值则跳过)
bodyText = frontmatterParser.stripFrontmatter(text);
documentFieldEnricher.enrich(frontmatter, bodyText, category);
log.info("解析到frontmatter: title={}, keywords={}, time={}ms",
frontmatter.getTitle(), frontmatter.getKeywords(), System.currentTimeMillis() - frontmatterStart);
} else {
log.warn("frontmatter解析失败,文件名: {}", fileName);
}
} else {
log.debug("文件不包含frontmatter: {}", fileName);
}
// 6. 分块
long chunkStart = System.currentTimeMillis();
List<DocumentChunk> chunks = documentChunkService.chunkDocument(bodyText, fileName);
if (chunks.isEmpty()) {
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
}
log.info("文档分块完成: fileName={}, chunks={}, time={}ms",
fileName, chunks.size(), System.currentTimeMillis() - chunkStart);
// 7. 创建文档元数据
String docId = resolveDocumentId(frontmatter);
String metadataJson = null;
if (frontmatter != null) {
try {
metadataJson = objectMapper.writeValueAsString(frontmatter);
} catch (Exception e) {
log.warn("Frontmatter序列化失败", e);
}
}
ApiDocument document = ApiDocument.builder()
.docId(docId)
.fileName(fileName)
.filePath(localPath)
.metadata(metadataJson)
.faultCategory(parseFaultCategory(request.getFaultCategory()))
.faultSource(request.getFaultSource())
.apiName(request.getApiName())
.version(request.getVersion())
.fileSize(file.getSize())
.fileHash(fileHash)
.status("PROCESSING")
.chunkCount(chunks.size())
.build();
apiDocumentRepository.save(document);
log.info("文档元数据已保存: docId={}", docId);
// 8. 向量化并索引
try {
long vectorStart = System.currentTimeMillis();
vectorIndexService.indexDocumentChunks(docId, chunks, category, frontmatter);
document.setStatus("INDEXED");
document.setIndexedAt(LocalDateTime.now());
apiDocumentRepository.save(document);
log.info("文档向量索引完成: docId={}, category={}, time={}ms",
docId, category, System.currentTimeMillis() - vectorStart);
} catch (Exception e) {
log.error("文档索引失败: docId={}", docId, e);
document.setStatus("FAILED");
apiDocumentRepository.save(document);
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
}
// 9. 更新 L0 索引
if (frontmatter != null) {
KnowledgeEntry entry = KnowledgeEntry.builder()
.filePath(localPath)
.title(frontmatter.getTitle())
.keywords(frontmatter.getKeywords())
.summary(frontmatter.getSummary())
.category(category)
.kbScope(frontmatter.getKbScope())
.sections(frontmatter.getSections())
.covers(frontmatter.getCovers())
.whenToRetrieve(frontmatter.getWhenToRetrieve())
.build();
knowledgeIndexService.addToIndex(entry);
log.info("文档已加入L0索引: docId={}, title={}", docId, frontmatter.getTitle());
}
// 触发域级聚合重算
knowledgeDomainService.onDocumentChange(category);
long totalTime = System.currentTimeMillis() - startTime;
log.info("文档上传完成: docId={}, fileName={}, hasFrontmatter={}, totalTime={}ms",
docId, fileName, frontmatter != null, totalTime);
return docId;
} catch (Exception e) {
// 失败时清理本地文件
cleanupLocalFile(localPath);
log.error("文档上传失败: fileName={}", fileName, e);
throw e;
}
}
/**
* 计算文件 hash(MD5)
*/
private String calculateFileHash(MultipartFile file) {
try {
MessageDigest md = MessageDigest.getInstance("MD5");
byte[] digest = md.digest(file.getBytes());
StringBuilder sb = new StringBuilder();
for (byte b : digest) {
sb.append(String.format("%02x", b));
}
return sb.toString();
} catch (Exception e) {
throw new DocumentProcessException(
file.getOriginalFilename(), "hash",
"计算文件 hash 失败: " + e.getMessage(), e
);
}
}
/**
* 保存文件到本地
*
* @param file 上传的文件
* @param fileName 文件名
* @param category 类别
* @return 本地文件路径
*/
private String saveToLocal(MultipartFile file, String fileName, String category) {
try {
// 1. 构建目标路径
Path baseDir = Paths.get(knowledgeBasePath).normalize();
Path categoryDir = baseDir.resolve(category).normalize();
Files.createDirectories(categoryDir);
Path targetPath = categoryDir.resolve(fileName);
// 2. 保存文件
file.transferTo(targetPath.toFile());
String relativePath = baseDir.relativize(targetPath.normalize()).toString().replace("\\", "/");
log.info("文件已保存到本地: {}, storedPath={}", targetPath, relativePath);
return relativePath;
} catch (IOException e) {
throw new DocumentProcessException(
fileName, "save-local",
"保存文件到本地失败: " + e.getMessage(), e
);
}
}
/**
* 清理本地文件(事务回滚时调用)
*
* @param localPath 本地文件路径
*/
private void cleanupLocalFile(String localPath) {
if (localPath != null) {
try {
Files.deleteIfExists(resolveLocalPath(localPath));
log.info("已清理本地文件: {}", localPath);
} catch (IOException e) {
log.warn("清理本地文件失败: {}", localPath, e);
}
}
}
/**
* 解析故障类别
*/
private FaultCategory parseFaultCategory(String category) {
if (category == null || category.isBlank()) {
return FaultCategory.GENERAL;
}
try {
return FaultCategory.valueOf(category.toUpperCase());
} catch (IllegalArgumentException e) {
return FaultCategory.GENERAL;
}
}
private String resolveDocumentId(Frontmatter frontmatter) {
if (frontmatter != null && frontmatter.getSource() != null) {
String source = frontmatter.getSource().trim();
if (!source.isEmpty() && source.length() <= 64) {
return source;
}
}
return UUID.randomUUID().toString();
}
/**
* 根据 docId 查询文档
*/
public DocumentQueryResponse queryDocumentById(String docId) {
Optional<ApiDocument> optional = apiDocumentRepository.findByDocId(docId);
if (optional.isEmpty()) {
throw new DocumentProcessException(docId, "query", "文档不存在");
}
ApiDocument doc = optional.get();
return convertToResponse(doc);
}
/**
* 根据状态查询文档列表(分页)
*/
public List<DocumentQueryResponse> queryDocumentsByStatus(String status, int page, int size) {
Page<ApiDocument> documents = apiDocumentRepository.findByStatus(status, PageRequest.of(page, size));
return documents.stream()
.map(this::convertToResponse)
.collect(Collectors.toList());
}
/**
* 根据故障源查询文档列表
*/
public List<DocumentQueryResponse> queryDocumentsByFaultSource(String faultSource) {
List<ApiDocument> documents = apiDocumentRepository.findByFaultSource(faultSource);
return documents.stream()
.map(this::convertToResponse)
.collect(Collectors.toList());
}
/**
* 删除文档
*/
@Transactional
public void deleteDocument(String docId) {
Optional<ApiDocument> optional = apiDocumentRepository.findByDocId(docId);
if (optional.isEmpty()) {
throw new DocumentProcessException(docId, "delete", "文档不存在");
}
ApiDocument doc = optional.get();
// 删除本地文件
if (doc.getFilePath() != null) {
try {
Files.deleteIfExists(resolveLocalPath(doc.getFilePath()));
log.info("本地文件已删除: {}", doc.getFilePath());
} catch (IOException e) {
log.warn("删除本地文件失败: {}", doc.getFilePath(), e);
}
}
// 删除 L0 索引
if (doc.getFilePath() != null) {
knowledgeIndexService.removeFromIndex(doc.getFilePath());
}
// 删除向量索引
try {
vectorIndexService.deleteDocumentChunks(docId);
log.info("文档向量索引已删除,docId: {}", docId);
} catch (Exception e) {
log.warn("删除向量索引失败,docId: {}", docId, e);
}
// 删除元数据
apiDocumentRepository.delete(doc);
log.info("文档已删除,docId: {}", docId);
// 触发域级聚合重算
String category = doc.getFilePath() != null
? resolveCategory(doc.getFilePath()) : null;
if (category != null) {
knowledgeDomainService.onDocumentChange(category);
}
}
/**
* 转换为响应 DTO
*/
/**
* 从 filePath 解析 category(取 knowledge_base/{category}/... 中的 category 段)
*/
private String resolveCategory(String filePath) {
try {
java.nio.file.Path p = java.nio.file.Paths.get(filePath);
// filePath 形如 knowledge_base/payment/xxx.md,取倒数第二段
int nameCount = p.getNameCount();
if (nameCount >= 2) {
return p.getName(nameCount - 2).toString();
}
} catch (Exception ignored) {}
return null;
}
private Path resolveLocalPath(String filePath) {
Path path = Paths.get(filePath).normalize();
if (path.isAbsolute()) {
return path;
}
Path basePath = Paths.get(knowledgeBasePath).toAbsolutePath().normalize();
Path baseName = basePath.getFileName();
if (baseName != null && path.startsWith(baseName) && basePath.getParent() != null) {
return basePath.getParent().resolve(path).normalize();
}
Path pathFromWorkingDir = path.toAbsolutePath().normalize();
if (pathFromWorkingDir.startsWith(basePath)) {
return pathFromWorkingDir;
}
return basePath.resolve(path).normalize();
}
/**
* 转换为响应 DTO
*/
private DocumentQueryResponse convertToResponse(ApiDocument doc) {
return DocumentQueryResponse.builder()
.docId(doc.getDocId())
.fileName(doc.getFileName())
.faultCategory(doc.getFaultCategory().name())
.faultSource(doc.getFaultSource())
.apiName(doc.getApiName())
.version(doc.getVersion())
.fileSize(doc.getFileSize())
.status(doc.getStatus())
.chunkCount(doc.getChunkCount())
.indexedAt(doc.getIndexedAt())
.createdAt(doc.getCreatedAt())
.build();
}
}