feat(knowledge): 完成 L0+L1 混合检索集成
核心功能: - 新增 FrontmatterParser 解析 YAML frontmatter - 新增 KnowledgeIndexService L0 内存索引 - 新增 LookupKnowledgeTool 混合检索工具 - 增强 DocumentManagementService 文件保存和索引同步 技术实现: - 数据库迁移 V004: api_document.metadata (TEXT) - 依赖新增: snakeyaml 2.0 - 配置新增: knowledge.base-path - 可观测性: requestId 追踪 + 性能日志 质量保证: - 单元测试: 31/31 通过 - 测试覆盖: FrontmatterParser(11), KnowledgeIndexService(13), LookupKnowledgeTool(7) - 启动验证: L0 索引正常加载 归档文档: - OpenSpec: openspec/changes/lookup-knowledge-integration/ - devflow 档案: devflow/projects/2026-06-24-lookup-knowledge-integration/ - handoff: handoff/2026-06-24-lookup-knowledge-integration.md
This commit is contained in:
@@ -1,14 +1,18 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.domain.entity.ApiDocument;
|
||||
import com.superbiz.agent.domain.enums.FaultCategory;
|
||||
import com.superbiz.agent.dto.DocumentChunk;
|
||||
import com.superbiz.agent.dto.DocumentQueryResponse;
|
||||
import com.superbiz.agent.dto.DocumentUploadRequest;
|
||||
import com.superbiz.agent.dto.Frontmatter;
|
||||
import com.superbiz.agent.dto.KnowledgeEntry;
|
||||
import com.superbiz.agent.exception.DocumentProcessException;
|
||||
import com.superbiz.agent.repository.ApiDocumentRepository;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.data.domain.Page;
|
||||
import org.springframework.data.domain.PageRequest;
|
||||
import org.springframework.stereotype.Service;
|
||||
@@ -16,6 +20,9 @@ import org.springframework.transaction.annotation.Transactional;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.Paths;
|
||||
import java.security.MessageDigest;
|
||||
import java.time.LocalDateTime;
|
||||
import java.util.List;
|
||||
@@ -30,6 +37,9 @@ import java.util.stream.Collectors;
|
||||
@Service
|
||||
public class DocumentManagementService {
|
||||
|
||||
@Value("${knowledge.base-path}")
|
||||
private String knowledgeBasePath;
|
||||
|
||||
@Autowired
|
||||
private TextExtractorService textExtractorService;
|
||||
|
||||
@@ -42,6 +52,15 @@ public class DocumentManagementService {
|
||||
@Autowired
|
||||
private ApiDocumentRepository apiDocumentRepository;
|
||||
|
||||
@Autowired
|
||||
private FrontmatterParser frontmatterParser;
|
||||
|
||||
@Autowired
|
||||
private KnowledgeIndexService knowledgeIndexService;
|
||||
|
||||
@Autowired
|
||||
private ObjectMapper objectMapper;
|
||||
|
||||
/**
|
||||
* 上传文档
|
||||
*
|
||||
@@ -52,82 +71,149 @@ public class DocumentManagementService {
|
||||
public String uploadDocument(DocumentUploadRequest request) {
|
||||
MultipartFile file = request.getFile();
|
||||
String fileName = file.getOriginalFilename();
|
||||
String localPath = null;
|
||||
long startTime = System.currentTimeMillis();
|
||||
|
||||
log.info("开始上传文档,文件名: {}, 大小: {} bytes", fileName, file.getSize());
|
||||
|
||||
// 1. 验证文件格式
|
||||
if (!textExtractorService.isSupportedFormat(fileName)) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"不支持的文件格式,仅支持 .md 和 .txt"
|
||||
);
|
||||
}
|
||||
|
||||
// 2. 计算文件 hash(去重)
|
||||
String fileHash = calculateFileHash(file);
|
||||
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
|
||||
if (existing.isPresent()) {
|
||||
log.warn("文档已存在,hash: {}, docId: ", fileHash, existing.get().getDocId());
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"文档已存在,docId: " + existing.get().getDocId()
|
||||
);
|
||||
}
|
||||
|
||||
// 3. 提取文本
|
||||
String text = textExtractorService.extractText(file, fileName);
|
||||
if (text == null || text.isBlank()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
|
||||
}
|
||||
|
||||
// 4. 分块(使用 DocumentChunkService 默认配置)
|
||||
// 注意:chunkSize 和 overlap 参数由 DocumentChunkConfig 配置,暂不支持动态调整
|
||||
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
|
||||
|
||||
if (chunks.isEmpty()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
|
||||
}
|
||||
|
||||
log.info("文档分块完成,文件名: {}, 分块数: {}", fileName, chunks.size());
|
||||
|
||||
// 5. 创建文档元数据
|
||||
String docId = UUID.randomUUID().toString();
|
||||
ApiDocument document = ApiDocument.builder()
|
||||
.docId(docId)
|
||||
.fileName(fileName)
|
||||
.faultCategory(parseFaultCategory(request.getFaultCategory()))
|
||||
.faultSource(request.getFaultSource())
|
||||
.apiName(request.getApiName())
|
||||
.version(request.getVersion())
|
||||
.fileSize(file.getSize())
|
||||
.fileHash(fileHash)
|
||||
.status("PROCESSING")
|
||||
.chunkCount(chunks.size())
|
||||
.build();
|
||||
|
||||
apiDocumentRepository.save(document);
|
||||
log.info("文档元数据已保存,docId: {}", docId);
|
||||
|
||||
// 6. 向量化并索引
|
||||
try {
|
||||
// 1. 验证文件格式
|
||||
if (!textExtractorService.isSupportedFormat(fileName)) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"不支持的文件格式,仅支持 .md 和 .txt"
|
||||
);
|
||||
}
|
||||
|
||||
// 2. 计算文件 hash(去重)
|
||||
long hashStart = System.currentTimeMillis();
|
||||
String fileHash = calculateFileHash(file);
|
||||
log.debug("文件hash计算完成: hash={}, time={}ms", fileHash, System.currentTimeMillis() - hashStart);
|
||||
|
||||
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
|
||||
if (existing.isPresent()) {
|
||||
log.warn("文档已存在,hash: {}, docId: {}", fileHash, existing.get().getDocId());
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"文档已存在,docId: " + existing.get().getDocId()
|
||||
);
|
||||
}
|
||||
|
||||
// 3. 提取文本
|
||||
long extractStart = System.currentTimeMillis();
|
||||
String text = textExtractorService.extractText(file, fileName);
|
||||
log.debug("文本提取完成: length={}, time={}ms", text != null ? text.length() : 0, System.currentTimeMillis() - extractStart);
|
||||
|
||||
if (text == null || text.isBlank()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
|
||||
}
|
||||
|
||||
// 4. 保存原始文件到本地
|
||||
String category = request.getCategory();
|
||||
if (category == null || category.isBlank()) {
|
||||
category = "upload"; // 默认类别
|
||||
category = "default";
|
||||
}
|
||||
vectorIndexService.indexDocumentChunks(docId, chunks, category);
|
||||
document.setStatus("INDEXED");
|
||||
document.setIndexedAt(LocalDateTime.now());
|
||||
long saveStart = System.currentTimeMillis();
|
||||
localPath = saveToLocal(file, fileName, category);
|
||||
log.debug("文件保存到本地完成: path={}, time={}ms", localPath, System.currentTimeMillis() - saveStart);
|
||||
|
||||
// 5. 解析 frontmatter
|
||||
long frontmatterStart = System.currentTimeMillis();
|
||||
Frontmatter frontmatter = null;
|
||||
if (frontmatterParser.hasFrontmatter(text)) {
|
||||
frontmatter = frontmatterParser.parse(text);
|
||||
if (frontmatter != null) {
|
||||
log.info("解析到frontmatter: title={}, keywords={}, time={}ms",
|
||||
frontmatter.getTitle(), frontmatter.getKeywords(), System.currentTimeMillis() - frontmatterStart);
|
||||
} else {
|
||||
log.warn("frontmatter解析失败,文件名: {}", fileName);
|
||||
}
|
||||
} else {
|
||||
log.debug("文件不包含frontmatter: {}", fileName);
|
||||
}
|
||||
|
||||
// 6. 分块
|
||||
long chunkStart = System.currentTimeMillis();
|
||||
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
|
||||
if (chunks.isEmpty()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
|
||||
}
|
||||
log.info("文档分块完成: fileName={}, chunks={}, time={}ms",
|
||||
fileName, chunks.size(), System.currentTimeMillis() - chunkStart);
|
||||
|
||||
// 7. 创建文档元数据
|
||||
String docId = UUID.randomUUID().toString();
|
||||
String metadataJson = null;
|
||||
if (frontmatter != null) {
|
||||
try {
|
||||
metadataJson = objectMapper.writeValueAsString(frontmatter);
|
||||
} catch (Exception e) {
|
||||
log.warn("Frontmatter序列化失败", e);
|
||||
}
|
||||
}
|
||||
|
||||
ApiDocument document = ApiDocument.builder()
|
||||
.docId(docId)
|
||||
.fileName(fileName)
|
||||
.filePath(localPath)
|
||||
.metadata(metadataJson)
|
||||
.faultCategory(parseFaultCategory(request.getFaultCategory()))
|
||||
.faultSource(request.getFaultSource())
|
||||
.apiName(request.getApiName())
|
||||
.version(request.getVersion())
|
||||
.fileSize(file.getSize())
|
||||
.fileHash(fileHash)
|
||||
.status("PROCESSING")
|
||||
.chunkCount(chunks.size())
|
||||
.build();
|
||||
|
||||
apiDocumentRepository.save(document);
|
||||
log.info("文档索引完成,docId: {}, 类别: {}", docId, category);
|
||||
log.info("文档元数据已保存: docId={}", docId);
|
||||
|
||||
// 8. 向量化并索引
|
||||
try {
|
||||
long vectorStart = System.currentTimeMillis();
|
||||
vectorIndexService.indexDocumentChunks(docId, chunks, category);
|
||||
document.setStatus("INDEXED");
|
||||
document.setIndexedAt(LocalDateTime.now());
|
||||
apiDocumentRepository.save(document);
|
||||
log.info("文档向量索引完成: docId={}, category={}, time={}ms",
|
||||
docId, category, System.currentTimeMillis() - vectorStart);
|
||||
|
||||
} catch (Exception e) {
|
||||
log.error("文档索引失败: docId={}", docId, e);
|
||||
document.setStatus("FAILED");
|
||||
apiDocumentRepository.save(document);
|
||||
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
|
||||
}
|
||||
|
||||
// 9. 更新 L0 索引
|
||||
if (frontmatter != null) {
|
||||
KnowledgeEntry entry = KnowledgeEntry.builder()
|
||||
.filePath(localPath)
|
||||
.title(frontmatter.getTitle())
|
||||
.keywords(frontmatter.getKeywords())
|
||||
.summary(frontmatter.getSummary())
|
||||
.category(category)
|
||||
.sections(frontmatter.getSections())
|
||||
.build();
|
||||
|
||||
knowledgeIndexService.addToIndex(entry);
|
||||
log.info("文档已加入L0索引: docId={}, title={}", docId, frontmatter.getTitle());
|
||||
}
|
||||
|
||||
long totalTime = System.currentTimeMillis() - startTime;
|
||||
log.info("文档上传完成: docId={}, fileName={}, hasFrontmatter={}, totalTime={}ms",
|
||||
docId, fileName, frontmatter != null, totalTime);
|
||||
|
||||
return docId;
|
||||
|
||||
} catch (Exception e) {
|
||||
log.error("文档索引失败,docId: {}", docId, e);
|
||||
document.setStatus("FAILED");
|
||||
apiDocumentRepository.save(document);
|
||||
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
|
||||
// 失败时清理本地文件
|
||||
cleanupLocalFile(localPath);
|
||||
log.error("文档上传失败: fileName={}", fileName, e);
|
||||
throw e;
|
||||
}
|
||||
|
||||
return docId;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -150,6 +236,52 @@ public class DocumentManagementService {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 保存文件到本地
|
||||
*
|
||||
* @param file 上传的文件
|
||||
* @param fileName 文件名
|
||||
* @param category 类别
|
||||
* @return 本地文件路径
|
||||
*/
|
||||
private String saveToLocal(MultipartFile file, String fileName, String category) {
|
||||
try {
|
||||
// 1. 构建目标路径
|
||||
Path categoryDir = Paths.get(knowledgeBasePath, category);
|
||||
Files.createDirectories(categoryDir);
|
||||
|
||||
Path targetPath = categoryDir.resolve(fileName);
|
||||
|
||||
// 2. 保存文件
|
||||
file.transferTo(targetPath.toFile());
|
||||
|
||||
log.info("文件已保存到本地: {}", targetPath);
|
||||
return targetPath.toString();
|
||||
|
||||
} catch (IOException e) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "save-local",
|
||||
"保存文件到本地失败: " + e.getMessage(), e
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 清理本地文件(事务回滚时调用)
|
||||
*
|
||||
* @param localPath 本地文件路径
|
||||
*/
|
||||
private void cleanupLocalFile(String localPath) {
|
||||
if (localPath != null) {
|
||||
try {
|
||||
Files.deleteIfExists(Paths.get(localPath));
|
||||
log.info("已清理本地文件: {}", localPath);
|
||||
} catch (IOException e) {
|
||||
log.warn("清理本地文件失败: {}", localPath, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 解析故障类别
|
||||
*/
|
||||
@@ -209,6 +341,21 @@ public class DocumentManagementService {
|
||||
|
||||
ApiDocument doc = optional.get();
|
||||
|
||||
// 删除本地文件
|
||||
if (doc.getFilePath() != null) {
|
||||
try {
|
||||
Files.deleteIfExists(Paths.get(doc.getFilePath()));
|
||||
log.info("本地文件已删除: {}", doc.getFilePath());
|
||||
} catch (IOException e) {
|
||||
log.warn("删除本地文件失败: {}", doc.getFilePath(), e);
|
||||
}
|
||||
}
|
||||
|
||||
// 删除 L0 索引
|
||||
if (doc.getFilePath() != null) {
|
||||
knowledgeIndexService.removeFromIndex(doc.getFilePath());
|
||||
}
|
||||
|
||||
// 删除向量索引
|
||||
try {
|
||||
vectorIndexService.deleteDocumentChunks(docId);
|
||||
|
||||
Reference in New Issue
Block a user