feat(knowledge): 完成 L0+L1 混合检索集成

核心功能:
- 新增 FrontmatterParser 解析 YAML frontmatter
- 新增 KnowledgeIndexService L0 内存索引
- 新增 LookupKnowledgeTool 混合检索工具
- 增强 DocumentManagementService 文件保存和索引同步

技术实现:
- 数据库迁移 V004: api_document.metadata (TEXT)
- 依赖新增: snakeyaml 2.0
- 配置新增: knowledge.base-path
- 可观测性: requestId 追踪 + 性能日志

质量保证:
- 单元测试: 31/31 通过
- 测试覆盖: FrontmatterParser(11), KnowledgeIndexService(13), LookupKnowledgeTool(7)
- 启动验证: L0 索引正常加载

归档文档:
- OpenSpec: openspec/changes/lookup-knowledge-integration/
- devflow 档案: devflow/projects/2026-06-24-lookup-knowledge-integration/
- handoff: handoff/2026-06-24-lookup-knowledge-integration.md
This commit is contained in:
zhuyongxin
2026-06-24 16:07:10 +08:00
parent c86045b33f
commit d6229f3385
32 changed files with 5396 additions and 67 deletions
@@ -0,0 +1,225 @@
package com.superbiz.agent.service;
import com.superbiz.agent.dto.Frontmatter;
import com.superbiz.agent.dto.KnowledgeEntry;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.stereotype.Service;
import jakarta.annotation.PostConstruct;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.List;
import java.util.concurrent.CopyOnWriteArrayList;
import java.util.stream.Collectors;
import java.util.stream.Stream;
/**
* 知识库索引服务
* 负责 L0 精确匹配索引的管理
*/
@Slf4j
@Service
public class KnowledgeIndexService {
@Value("${knowledge.base-path}")
private String knowledgeBasePath;
@Autowired
private FrontmatterParser frontmatterParser;
/**
* 内存索引(线程安全)
*/
private final List<KnowledgeEntry> knowledgeIndex = new CopyOnWriteArrayList<>();
/**
* 启动时扫描知识库目录,构建索引
*/
@PostConstruct
public void loadIndex() {
log.info("开始扫描知识库目录: {}", knowledgeBasePath);
try {
Path basePath = Paths.get(knowledgeBasePath);
// 目录不存在时自动创建
if (!Files.exists(basePath)) {
Files.createDirectories(basePath);
log.info("知识库目录已创建: {}", basePath.toAbsolutePath());
}
// 递归扫描 .md 文件
try (Stream<Path> paths = Files.walk(basePath)) {
paths.filter(p -> p.toString().endsWith(".md"))
.forEach(this::indexFile);
}
log.info("知识库索引加载完成,共 {} 个文档", knowledgeIndex.size());
} catch (IOException e) {
log.error("知识库索引加载失败", e);
}
}
/**
* 索引单个文件
*
* @param filePath 文件路径
*/
private void indexFile(Path filePath) {
try {
// 读取文件内容
String content = Files.readString(filePath);
// 解析 frontmatter
Frontmatter frontmatter = frontmatterParser.parse(content);
if (frontmatter == null) {
log.debug("跳过文件(无有效 frontmatter): {}", filePath);
return;
}
// 提取 category(从路径中获取)
String category = extractCategoryFromPath(filePath.toString());
// 构建索引条目
KnowledgeEntry entry = KnowledgeEntry.builder()
.filePath(filePath.toString())
.title(frontmatter.getTitle())
.keywords(frontmatter.getKeywords())
.summary(frontmatter.getSummary())
.category(category)
.sections(frontmatter.getSections())
.build();
knowledgeIndex.add(entry);
log.debug("文档已加入索引: title={}, filePath={}", entry.getTitle(), filePath);
} catch (IOException e) {
log.warn("读取文件失败: {}", filePath, e);
}
}
/**
* 从文件路径中提取 category
* 例如:knowledge_base/api/test.md -> api
*/
private String extractCategoryFromPath(String filePath) {
String normalized = filePath.replace("\\", "/");
String[] parts = normalized.split("/");
// 查找 knowledge_base 后的第一个目录
for (int i = 0; i < parts.length - 1; i++) {
if (parts[i].equals("knowledge_base") && i + 1 < parts.length) {
return parts[i + 1];
}
}
return "default";
}
/**
* L0 精确匹配
*
* @param query 查询关键词
* @return 匹配的文档列表
*/
public List<KnowledgeEntry> exactMatch(String query) {
long startTime = System.currentTimeMillis();
if (query == null || query.trim().isEmpty()) {
log.debug("查询关键词为空,返回空结果");
return List.of();
}
String queryLower = query.toLowerCase();
List<KnowledgeEntry> results = knowledgeIndex.stream()
.filter(entry -> matchesKeywords(entry, queryLower))
.collect(Collectors.toList());
long elapsedTime = System.currentTimeMillis() - startTime;
log.debug("L0精确匹配: query={}, matches={}, indexSize={}, time={}ms",
query, results.size(), knowledgeIndex.size(), elapsedTime);
return results;
}
/**
* 关键词匹配逻辑(不区分大小写)
*
* @param entry 索引条目
* @param query 查询关键词(小写)
* @return true 如果匹配
*/
private boolean matchesKeywords(KnowledgeEntry entry, String query) {
if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) {
return false;
}
for (String keyword : entry.getKeywords()) {
String keywordLower = keyword.toLowerCase();
// query 包含 keyword 或 keyword 包含 query
if (query.contains(keywordLower) || keywordLower.contains(query)) {
return true;
}
}
return false;
}
/**
* 读取文档内容
*
* @param filePath 文件路径
* @param maxChars 最大字符数
* @return 文档内容(前 maxChars 字符),失败返回 null
*/
public String readDocument(String filePath, int maxChars) {
try {
String content = Files.readString(Paths.get(filePath));
if (content.length() > maxChars) {
return content.substring(0, maxChars) + "...";
}
return content;
} catch (IOException e) {
log.error("读取文档失败: {}", filePath, e);
return null;
}
}
/**
* 添加文档到索引(上传时调用)
*
* @param entry 知识库条目
*/
public void addToIndex(KnowledgeEntry entry) {
knowledgeIndex.add(entry);
log.debug("文档已添加到 L0 索引: title={}", entry.getTitle());
}
/**
* 从索引中移除文档(删除时调用)
*
* @param filePath 文件路径
*/
public void removeFromIndex(String filePath) {
knowledgeIndex.removeIf(e -> e.getFilePath().equals(filePath));
log.debug("文档已从 L0 索引移除: {}", filePath);
}
/**
* 获取索引大小
*
* @return 索引中的文档数量
*/
public int getIndexSize() {
return knowledgeIndex.size();
}
}