Files
SuperBizAgent-java/src/main/java/com/superbiz/agent/service/KnowledgeIndexService.java
T
zhuyongxin d6229f3385 feat(knowledge): 完成 L0+L1 混合检索集成
核心功能:
- 新增 FrontmatterParser 解析 YAML frontmatter
- 新增 KnowledgeIndexService L0 内存索引
- 新增 LookupKnowledgeTool 混合检索工具
- 增强 DocumentManagementService 文件保存和索引同步

技术实现:
- 数据库迁移 V004: api_document.metadata (TEXT)
- 依赖新增: snakeyaml 2.0
- 配置新增: knowledge.base-path
- 可观测性: requestId 追踪 + 性能日志

质量保证:
- 单元测试: 31/31 通过
- 测试覆盖: FrontmatterParser(11), KnowledgeIndexService(13), LookupKnowledgeTool(7)
- 启动验证: L0 索引正常加载

归档文档:
- OpenSpec: openspec/changes/lookup-knowledge-integration/
- devflow 档案: devflow/projects/2026-06-24-lookup-knowledge-integration/
- handoff: handoff/2026-06-24-lookup-knowledge-integration.md
2026-06-24 16:07:10 +08:00

226 lines
6.7 KiB
Java
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package com.superbiz.agent.service;
import com.superbiz.agent.dto.Frontmatter;
import com.superbiz.agent.dto.KnowledgeEntry;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.stereotype.Service;
import jakarta.annotation.PostConstruct;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.List;
import java.util.concurrent.CopyOnWriteArrayList;
import java.util.stream.Collectors;
import java.util.stream.Stream;
/**
* 知识库索引服务
* 负责 L0 精确匹配索引的管理
*/
@Slf4j
@Service
public class KnowledgeIndexService {
@Value("${knowledge.base-path}")
private String knowledgeBasePath;
@Autowired
private FrontmatterParser frontmatterParser;
/**
* 内存索引(线程安全)
*/
private final List<KnowledgeEntry> knowledgeIndex = new CopyOnWriteArrayList<>();
/**
* 启动时扫描知识库目录,构建索引
*/
@PostConstruct
public void loadIndex() {
log.info("开始扫描知识库目录: {}", knowledgeBasePath);
try {
Path basePath = Paths.get(knowledgeBasePath);
// 目录不存在时自动创建
if (!Files.exists(basePath)) {
Files.createDirectories(basePath);
log.info("知识库目录已创建: {}", basePath.toAbsolutePath());
}
// 递归扫描 .md 文件
try (Stream<Path> paths = Files.walk(basePath)) {
paths.filter(p -> p.toString().endsWith(".md"))
.forEach(this::indexFile);
}
log.info("知识库索引加载完成,共 {} 个文档", knowledgeIndex.size());
} catch (IOException e) {
log.error("知识库索引加载失败", e);
}
}
/**
* 索引单个文件
*
* @param filePath 文件路径
*/
private void indexFile(Path filePath) {
try {
// 读取文件内容
String content = Files.readString(filePath);
// 解析 frontmatter
Frontmatter frontmatter = frontmatterParser.parse(content);
if (frontmatter == null) {
log.debug("跳过文件(无有效 frontmatter): {}", filePath);
return;
}
// 提取 category(从路径中获取)
String category = extractCategoryFromPath(filePath.toString());
// 构建索引条目
KnowledgeEntry entry = KnowledgeEntry.builder()
.filePath(filePath.toString())
.title(frontmatter.getTitle())
.keywords(frontmatter.getKeywords())
.summary(frontmatter.getSummary())
.category(category)
.sections(frontmatter.getSections())
.build();
knowledgeIndex.add(entry);
log.debug("文档已加入索引: title={}, filePath={}", entry.getTitle(), filePath);
} catch (IOException e) {
log.warn("读取文件失败: {}", filePath, e);
}
}
/**
* 从文件路径中提取 category
* 例如:knowledge_base/api/test.md -> api
*/
private String extractCategoryFromPath(String filePath) {
String normalized = filePath.replace("\\", "/");
String[] parts = normalized.split("/");
// 查找 knowledge_base 后的第一个目录
for (int i = 0; i < parts.length - 1; i++) {
if (parts[i].equals("knowledge_base") && i + 1 < parts.length) {
return parts[i + 1];
}
}
return "default";
}
/**
* L0 精确匹配
*
* @param query 查询关键词
* @return 匹配的文档列表
*/
public List<KnowledgeEntry> exactMatch(String query) {
long startTime = System.currentTimeMillis();
if (query == null || query.trim().isEmpty()) {
log.debug("查询关键词为空,返回空结果");
return List.of();
}
String queryLower = query.toLowerCase();
List<KnowledgeEntry> results = knowledgeIndex.stream()
.filter(entry -> matchesKeywords(entry, queryLower))
.collect(Collectors.toList());
long elapsedTime = System.currentTimeMillis() - startTime;
log.debug("L0精确匹配: query={}, matches={}, indexSize={}, time={}ms",
query, results.size(), knowledgeIndex.size(), elapsedTime);
return results;
}
/**
* 关键词匹配逻辑(不区分大小写)
*
* @param entry 索引条目
* @param query 查询关键词(小写)
* @return true 如果匹配
*/
private boolean matchesKeywords(KnowledgeEntry entry, String query) {
if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) {
return false;
}
for (String keyword : entry.getKeywords()) {
String keywordLower = keyword.toLowerCase();
// query 包含 keyword 或 keyword 包含 query
if (query.contains(keywordLower) || keywordLower.contains(query)) {
return true;
}
}
return false;
}
/**
* 读取文档内容
*
* @param filePath 文件路径
* @param maxChars 最大字符数
* @return 文档内容(前 maxChars 字符),失败返回 null
*/
public String readDocument(String filePath, int maxChars) {
try {
String content = Files.readString(Paths.get(filePath));
if (content.length() > maxChars) {
return content.substring(0, maxChars) + "...";
}
return content;
} catch (IOException e) {
log.error("读取文档失败: {}", filePath, e);
return null;
}
}
/**
* 添加文档到索引(上传时调用)
*
* @param entry 知识库条目
*/
public void addToIndex(KnowledgeEntry entry) {
knowledgeIndex.add(entry);
log.debug("文档已添加到 L0 索引: title={}", entry.getTitle());
}
/**
* 从索引中移除文档(删除时调用)
*
* @param filePath 文件路径
*/
public void removeFromIndex(String filePath) {
knowledgeIndex.removeIf(e -> e.getFilePath().equals(filePath));
log.debug("文档已从 L0 索引移除: {}", filePath);
}
/**
* 获取索引大小
*
* @return 索引中的文档数量
*/
public int getIndexSize() {
return knowledgeIndex.size();
}
}