feat(knowledge): 完成 L0+L1 混合检索集成

核心功能:
- 新增 FrontmatterParser 解析 YAML frontmatter
- 新增 KnowledgeIndexService L0 内存索引
- 新增 LookupKnowledgeTool 混合检索工具
- 增强 DocumentManagementService 文件保存和索引同步

技术实现:
- 数据库迁移 V004: api_document.metadata (TEXT)
- 依赖新增: snakeyaml 2.0
- 配置新增: knowledge.base-path
- 可观测性: requestId 追踪 + 性能日志

质量保证:
- 单元测试: 31/31 通过
- 测试覆盖: FrontmatterParser(11), KnowledgeIndexService(13), LookupKnowledgeTool(7)
- 启动验证: L0 索引正常加载

归档文档:
- OpenSpec: openspec/changes/lookup-knowledge-integration/
- devflow 档案: devflow/projects/2026-06-24-lookup-knowledge-integration/
- handoff: handoff/2026-06-24-lookup-knowledge-integration.md
This commit is contained in:
zhuyongxin
2026-06-24 16:07:10 +08:00
parent c86045b33f
commit d6229f3385
32 changed files with 5396 additions and 67 deletions
@@ -1,14 +1,18 @@
package com.superbiz.agent.service;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.superbiz.agent.domain.entity.ApiDocument;
import com.superbiz.agent.domain.enums.FaultCategory;
import com.superbiz.agent.dto.DocumentChunk;
import com.superbiz.agent.dto.DocumentQueryResponse;
import com.superbiz.agent.dto.DocumentUploadRequest;
import com.superbiz.agent.dto.Frontmatter;
import com.superbiz.agent.dto.KnowledgeEntry;
import com.superbiz.agent.exception.DocumentProcessException;
import com.superbiz.agent.repository.ApiDocumentRepository;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.PageRequest;
import org.springframework.stereotype.Service;
@@ -16,6 +20,9 @@ import org.springframework.transaction.annotation.Transactional;
import org.springframework.web.multipart.MultipartFile;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.security.MessageDigest;
import java.time.LocalDateTime;
import java.util.List;
@@ -30,6 +37,9 @@ import java.util.stream.Collectors;
@Service
public class DocumentManagementService {
@Value("${knowledge.base-path}")
private String knowledgeBasePath;
@Autowired
private TextExtractorService textExtractorService;
@@ -42,6 +52,15 @@ public class DocumentManagementService {
@Autowired
private ApiDocumentRepository apiDocumentRepository;
@Autowired
private FrontmatterParser frontmatterParser;
@Autowired
private KnowledgeIndexService knowledgeIndexService;
@Autowired
private ObjectMapper objectMapper;
/**
* 上传文档
*
@@ -52,82 +71,149 @@ public class DocumentManagementService {
public String uploadDocument(DocumentUploadRequest request) {
MultipartFile file = request.getFile();
String fileName = file.getOriginalFilename();
String localPath = null;
long startTime = System.currentTimeMillis();
log.info("开始上传文档,文件名: {}, 大小: {} bytes", fileName, file.getSize());
// 1. 验证文件格式
if (!textExtractorService.isSupportedFormat(fileName)) {
throw new DocumentProcessException(
fileName, "upload",
"不支持的文件格式,仅支持 .md 和 .txt"
);
}
// 2. 计算文件 hash(去重)
String fileHash = calculateFileHash(file);
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
if (existing.isPresent()) {
log.warn("文档已存在,hash: {}, docId: ", fileHash, existing.get().getDocId());
throw new DocumentProcessException(
fileName, "upload",
"文档已存在,docId: " + existing.get().getDocId()
);
}
// 3. 提取文本
String text = textExtractorService.extractText(file, fileName);
if (text == null || text.isBlank()) {
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
}
// 4. 分块(使用 DocumentChunkService 默认配置)
// 注意:chunkSize 和 overlap 参数由 DocumentChunkConfig 配置,暂不支持动态调整
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
if (chunks.isEmpty()) {
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
}
log.info("文档分块完成,文件名: {}, 分块数: {}", fileName, chunks.size());
// 5. 创建文档元数据
String docId = UUID.randomUUID().toString();
ApiDocument document = ApiDocument.builder()
.docId(docId)
.fileName(fileName)
.faultCategory(parseFaultCategory(request.getFaultCategory()))
.faultSource(request.getFaultSource())
.apiName(request.getApiName())
.version(request.getVersion())
.fileSize(file.getSize())
.fileHash(fileHash)
.status("PROCESSING")
.chunkCount(chunks.size())
.build();
apiDocumentRepository.save(document);
log.info("文档元数据已保存,docId: {}", docId);
// 6. 向量化并索引
try {
// 1. 验证文件格式
if (!textExtractorService.isSupportedFormat(fileName)) {
throw new DocumentProcessException(
fileName, "upload",
"不支持的文件格式,仅支持 .md 和 .txt"
);
}
// 2. 计算文件 hash(去重)
long hashStart = System.currentTimeMillis();
String fileHash = calculateFileHash(file);
log.debug("文件hash计算完成: hash={}, time={}ms", fileHash, System.currentTimeMillis() - hashStart);
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
if (existing.isPresent()) {
log.warn("文档已存在,hash: {}, docId: {}", fileHash, existing.get().getDocId());
throw new DocumentProcessException(
fileName, "upload",
"文档已存在,docId: " + existing.get().getDocId()
);
}
// 3. 提取文本
long extractStart = System.currentTimeMillis();
String text = textExtractorService.extractText(file, fileName);
log.debug("文本提取完成: length={}, time={}ms", text != null ? text.length() : 0, System.currentTimeMillis() - extractStart);
if (text == null || text.isBlank()) {
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
}
// 4. 保存原始文件到本地
String category = request.getCategory();
if (category == null || category.isBlank()) {
category = "upload"; // 默认类别
category = "default";
}
vectorIndexService.indexDocumentChunks(docId, chunks, category);
document.setStatus("INDEXED");
document.setIndexedAt(LocalDateTime.now());
long saveStart = System.currentTimeMillis();
localPath = saveToLocal(file, fileName, category);
log.debug("文件保存到本地完成: path={}, time={}ms", localPath, System.currentTimeMillis() - saveStart);
// 5. 解析 frontmatter
long frontmatterStart = System.currentTimeMillis();
Frontmatter frontmatter = null;
if (frontmatterParser.hasFrontmatter(text)) {
frontmatter = frontmatterParser.parse(text);
if (frontmatter != null) {
log.info("解析到frontmatter: title={}, keywords={}, time={}ms",
frontmatter.getTitle(), frontmatter.getKeywords(), System.currentTimeMillis() - frontmatterStart);
} else {
log.warn("frontmatter解析失败,文件名: {}", fileName);
}
} else {
log.debug("文件不包含frontmatter: {}", fileName);
}
// 6. 分块
long chunkStart = System.currentTimeMillis();
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
if (chunks.isEmpty()) {
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
}
log.info("文档分块完成: fileName={}, chunks={}, time={}ms",
fileName, chunks.size(), System.currentTimeMillis() - chunkStart);
// 7. 创建文档元数据
String docId = UUID.randomUUID().toString();
String metadataJson = null;
if (frontmatter != null) {
try {
metadataJson = objectMapper.writeValueAsString(frontmatter);
} catch (Exception e) {
log.warn("Frontmatter序列化失败", e);
}
}
ApiDocument document = ApiDocument.builder()
.docId(docId)
.fileName(fileName)
.filePath(localPath)
.metadata(metadataJson)
.faultCategory(parseFaultCategory(request.getFaultCategory()))
.faultSource(request.getFaultSource())
.apiName(request.getApiName())
.version(request.getVersion())
.fileSize(file.getSize())
.fileHash(fileHash)
.status("PROCESSING")
.chunkCount(chunks.size())
.build();
apiDocumentRepository.save(document);
log.info("文档索引完成,docId: {}, 类别: {}", docId, category);
log.info("文档元数据已保存: docId={}", docId);
// 8. 向量化并索引
try {
long vectorStart = System.currentTimeMillis();
vectorIndexService.indexDocumentChunks(docId, chunks, category);
document.setStatus("INDEXED");
document.setIndexedAt(LocalDateTime.now());
apiDocumentRepository.save(document);
log.info("文档向量索引完成: docId={}, category={}, time={}ms",
docId, category, System.currentTimeMillis() - vectorStart);
} catch (Exception e) {
log.error("文档索引失败: docId={}", docId, e);
document.setStatus("FAILED");
apiDocumentRepository.save(document);
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
}
// 9. 更新 L0 索引
if (frontmatter != null) {
KnowledgeEntry entry = KnowledgeEntry.builder()
.filePath(localPath)
.title(frontmatter.getTitle())
.keywords(frontmatter.getKeywords())
.summary(frontmatter.getSummary())
.category(category)
.sections(frontmatter.getSections())
.build();
knowledgeIndexService.addToIndex(entry);
log.info("文档已加入L0索引: docId={}, title={}", docId, frontmatter.getTitle());
}
long totalTime = System.currentTimeMillis() - startTime;
log.info("文档上传完成: docId={}, fileName={}, hasFrontmatter={}, totalTime={}ms",
docId, fileName, frontmatter != null, totalTime);
return docId;
} catch (Exception e) {
log.error("文档索引失败,docId: {}", docId, e);
document.setStatus("FAILED");
apiDocumentRepository.save(document);
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
// 失败时清理本地文件
cleanupLocalFile(localPath);
log.error("文档上传失败: fileName={}", fileName, e);
throw e;
}
return docId;
}
/**
@@ -150,6 +236,52 @@ public class DocumentManagementService {
}
}
/**
* 保存文件到本地
*
* @param file 上传的文件
* @param fileName 文件名
* @param category 类别
* @return 本地文件路径
*/
private String saveToLocal(MultipartFile file, String fileName, String category) {
try {
// 1. 构建目标路径
Path categoryDir = Paths.get(knowledgeBasePath, category);
Files.createDirectories(categoryDir);
Path targetPath = categoryDir.resolve(fileName);
// 2. 保存文件
file.transferTo(targetPath.toFile());
log.info("文件已保存到本地: {}", targetPath);
return targetPath.toString();
} catch (IOException e) {
throw new DocumentProcessException(
fileName, "save-local",
"保存文件到本地失败: " + e.getMessage(), e
);
}
}
/**
* 清理本地文件(事务回滚时调用)
*
* @param localPath 本地文件路径
*/
private void cleanupLocalFile(String localPath) {
if (localPath != null) {
try {
Files.deleteIfExists(Paths.get(localPath));
log.info("已清理本地文件: {}", localPath);
} catch (IOException e) {
log.warn("清理本地文件失败: {}", localPath, e);
}
}
}
/**
* 解析故障类别
*/
@@ -209,6 +341,21 @@ public class DocumentManagementService {
ApiDocument doc = optional.get();
// 删除本地文件
if (doc.getFilePath() != null) {
try {
Files.deleteIfExists(Paths.get(doc.getFilePath()));
log.info("本地文件已删除: {}", doc.getFilePath());
} catch (IOException e) {
log.warn("删除本地文件失败: {}", doc.getFilePath(), e);
}
}
// 删除 L0 索引
if (doc.getFilePath() != null) {
knowledgeIndexService.removeFromIndex(doc.getFilePath());
}
// 删除向量索引
try {
vectorIndexService.deleteDocumentChunks(docId);