核心功能: - 新增 FrontmatterParser 解析 YAML frontmatter - 新增 KnowledgeIndexService L0 内存索引 - 新增 LookupKnowledgeTool 混合检索工具 - 增强 DocumentManagementService 文件保存和索引同步 技术实现: - 数据库迁移 V004: api_document.metadata (TEXT) - 依赖新增: snakeyaml 2.0 - 配置新增: knowledge.base-path - 可观测性: requestId 追踪 + 性能日志 质量保证: - 单元测试: 31/31 通过 - 测试覆盖: FrontmatterParser(11), KnowledgeIndexService(13), LookupKnowledgeTool(7) - 启动验证: L0 索引正常加载 归档文档: - OpenSpec: openspec/changes/lookup-knowledge-integration/ - devflow 档案: devflow/projects/2026-06-24-lookup-knowledge-integration/ - handoff: handoff/2026-06-24-lookup-knowledge-integration.md
226 lines
6.7 KiB
Java
226 lines
6.7 KiB
Java
package com.superbiz.agent.service;
|
||
|
||
import com.superbiz.agent.dto.Frontmatter;
|
||
import com.superbiz.agent.dto.KnowledgeEntry;
|
||
import lombok.extern.slf4j.Slf4j;
|
||
import org.springframework.beans.factory.annotation.Autowired;
|
||
import org.springframework.beans.factory.annotation.Value;
|
||
import org.springframework.stereotype.Service;
|
||
|
||
import jakarta.annotation.PostConstruct;
|
||
import java.io.IOException;
|
||
import java.nio.file.Files;
|
||
import java.nio.file.Path;
|
||
import java.nio.file.Paths;
|
||
import java.util.List;
|
||
import java.util.concurrent.CopyOnWriteArrayList;
|
||
import java.util.stream.Collectors;
|
||
import java.util.stream.Stream;
|
||
|
||
/**
|
||
* 知识库索引服务
|
||
* 负责 L0 精确匹配索引的管理
|
||
*/
|
||
@Slf4j
|
||
@Service
|
||
public class KnowledgeIndexService {
|
||
|
||
@Value("${knowledge.base-path}")
|
||
private String knowledgeBasePath;
|
||
|
||
@Autowired
|
||
private FrontmatterParser frontmatterParser;
|
||
|
||
/**
|
||
* 内存索引(线程安全)
|
||
*/
|
||
private final List<KnowledgeEntry> knowledgeIndex = new CopyOnWriteArrayList<>();
|
||
|
||
/**
|
||
* 启动时扫描知识库目录,构建索引
|
||
*/
|
||
@PostConstruct
|
||
public void loadIndex() {
|
||
log.info("开始扫描知识库目录: {}", knowledgeBasePath);
|
||
|
||
try {
|
||
Path basePath = Paths.get(knowledgeBasePath);
|
||
|
||
// 目录不存在时自动创建
|
||
if (!Files.exists(basePath)) {
|
||
Files.createDirectories(basePath);
|
||
log.info("知识库目录已创建: {}", basePath.toAbsolutePath());
|
||
}
|
||
|
||
// 递归扫描 .md 文件
|
||
try (Stream<Path> paths = Files.walk(basePath)) {
|
||
paths.filter(p -> p.toString().endsWith(".md"))
|
||
.forEach(this::indexFile);
|
||
}
|
||
|
||
log.info("知识库索引加载完成,共 {} 个文档", knowledgeIndex.size());
|
||
|
||
} catch (IOException e) {
|
||
log.error("知识库索引加载失败", e);
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 索引单个文件
|
||
*
|
||
* @param filePath 文件路径
|
||
*/
|
||
private void indexFile(Path filePath) {
|
||
try {
|
||
// 读取文件内容
|
||
String content = Files.readString(filePath);
|
||
|
||
// 解析 frontmatter
|
||
Frontmatter frontmatter = frontmatterParser.parse(content);
|
||
if (frontmatter == null) {
|
||
log.debug("跳过文件(无有效 frontmatter): {}", filePath);
|
||
return;
|
||
}
|
||
|
||
// 提取 category(从路径中获取)
|
||
String category = extractCategoryFromPath(filePath.toString());
|
||
|
||
// 构建索引条目
|
||
KnowledgeEntry entry = KnowledgeEntry.builder()
|
||
.filePath(filePath.toString())
|
||
.title(frontmatter.getTitle())
|
||
.keywords(frontmatter.getKeywords())
|
||
.summary(frontmatter.getSummary())
|
||
.category(category)
|
||
.sections(frontmatter.getSections())
|
||
.build();
|
||
|
||
knowledgeIndex.add(entry);
|
||
log.debug("文档已加入索引: title={}, filePath={}", entry.getTitle(), filePath);
|
||
|
||
} catch (IOException e) {
|
||
log.warn("读取文件失败: {}", filePath, e);
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 从文件路径中提取 category
|
||
* 例如:knowledge_base/api/test.md -> api
|
||
*/
|
||
private String extractCategoryFromPath(String filePath) {
|
||
String normalized = filePath.replace("\\", "/");
|
||
String[] parts = normalized.split("/");
|
||
|
||
// 查找 knowledge_base 后的第一个目录
|
||
for (int i = 0; i < parts.length - 1; i++) {
|
||
if (parts[i].equals("knowledge_base") && i + 1 < parts.length) {
|
||
return parts[i + 1];
|
||
}
|
||
}
|
||
|
||
return "default";
|
||
}
|
||
|
||
/**
|
||
* L0 精确匹配
|
||
*
|
||
* @param query 查询关键词
|
||
* @return 匹配的文档列表
|
||
*/
|
||
public List<KnowledgeEntry> exactMatch(String query) {
|
||
long startTime = System.currentTimeMillis();
|
||
|
||
if (query == null || query.trim().isEmpty()) {
|
||
log.debug("查询关键词为空,返回空结果");
|
||
return List.of();
|
||
}
|
||
|
||
String queryLower = query.toLowerCase();
|
||
|
||
List<KnowledgeEntry> results = knowledgeIndex.stream()
|
||
.filter(entry -> matchesKeywords(entry, queryLower))
|
||
.collect(Collectors.toList());
|
||
|
||
long elapsedTime = System.currentTimeMillis() - startTime;
|
||
log.debug("L0精确匹配: query={}, matches={}, indexSize={}, time={}ms",
|
||
query, results.size(), knowledgeIndex.size(), elapsedTime);
|
||
|
||
return results;
|
||
}
|
||
|
||
/**
|
||
* 关键词匹配逻辑(不区分大小写)
|
||
*
|
||
* @param entry 索引条目
|
||
* @param query 查询关键词(小写)
|
||
* @return true 如果匹配
|
||
*/
|
||
private boolean matchesKeywords(KnowledgeEntry entry, String query) {
|
||
if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) {
|
||
return false;
|
||
}
|
||
|
||
for (String keyword : entry.getKeywords()) {
|
||
String keywordLower = keyword.toLowerCase();
|
||
// query 包含 keyword 或 keyword 包含 query
|
||
if (query.contains(keywordLower) || keywordLower.contains(query)) {
|
||
return true;
|
||
}
|
||
}
|
||
|
||
return false;
|
||
}
|
||
|
||
/**
|
||
* 读取文档内容
|
||
*
|
||
* @param filePath 文件路径
|
||
* @param maxChars 最大字符数
|
||
* @return 文档内容(前 maxChars 字符),失败返回 null
|
||
*/
|
||
public String readDocument(String filePath, int maxChars) {
|
||
try {
|
||
String content = Files.readString(Paths.get(filePath));
|
||
|
||
if (content.length() > maxChars) {
|
||
return content.substring(0, maxChars) + "...";
|
||
}
|
||
|
||
return content;
|
||
|
||
} catch (IOException e) {
|
||
log.error("读取文档失败: {}", filePath, e);
|
||
return null;
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 添加文档到索引(上传时调用)
|
||
*
|
||
* @param entry 知识库条目
|
||
*/
|
||
public void addToIndex(KnowledgeEntry entry) {
|
||
knowledgeIndex.add(entry);
|
||
log.debug("文档已添加到 L0 索引: title={}", entry.getTitle());
|
||
}
|
||
|
||
/**
|
||
* 从索引中移除文档(删除时调用)
|
||
*
|
||
* @param filePath 文件路径
|
||
*/
|
||
public void removeFromIndex(String filePath) {
|
||
knowledgeIndex.removeIf(e -> e.getFilePath().equals(filePath));
|
||
log.debug("文档已从 L0 索引移除: {}", filePath);
|
||
}
|
||
|
||
/**
|
||
* 获取索引大小
|
||
*
|
||
* @return 索引中的文档数量
|
||
*/
|
||
public int getIndexSize() {
|
||
return knowledgeIndex.size();
|
||
}
|
||
}
|