feat(knowledge): 完成 L0+L1 混合检索集成
核心功能: - 新增 FrontmatterParser 解析 YAML frontmatter - 新增 KnowledgeIndexService L0 内存索引 - 新增 LookupKnowledgeTool 混合检索工具 - 增强 DocumentManagementService 文件保存和索引同步 技术实现: - 数据库迁移 V004: api_document.metadata (TEXT) - 依赖新增: snakeyaml 2.0 - 配置新增: knowledge.base-path - 可观测性: requestId 追踪 + 性能日志 质量保证: - 单元测试: 31/31 通过 - 测试覆盖: FrontmatterParser(11), KnowledgeIndexService(13), LookupKnowledgeTool(7) - 启动验证: L0 索引正常加载 归档文档: - OpenSpec: openspec/changes/lookup-knowledge-integration/ - devflow 档案: devflow/projects/2026-06-24-lookup-knowledge-integration/ - handoff: handoff/2026-06-24-lookup-knowledge-integration.md
This commit is contained in:
@@ -72,6 +72,10 @@ public class ApiDocument {
|
||||
@Column(name = "error_message", columnDefinition = "TEXT")
|
||||
private String errorMessage;
|
||||
|
||||
// Frontmatter 元数据
|
||||
@Column(name = "metadata", columnDefinition = "TEXT")
|
||||
private String metadata;
|
||||
|
||||
// 时间字段
|
||||
@Column(name = "indexed_at")
|
||||
private LocalDateTime indexedAt;
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.AllArgsConstructor;
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
import lombok.NoArgsConstructor;
|
||||
|
||||
import java.time.LocalDate;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Frontmatter 数据模型
|
||||
* 用于解析 Markdown 文件头的 YAML frontmatter
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
@NoArgsConstructor
|
||||
@AllArgsConstructor
|
||||
public class Frontmatter {
|
||||
|
||||
/**
|
||||
* 文档标题(必填)
|
||||
*/
|
||||
private String title;
|
||||
|
||||
/**
|
||||
* 关键词列表(必填,用于 L0 精确匹配)
|
||||
*/
|
||||
private List<String> keywords;
|
||||
|
||||
/**
|
||||
* 文档摘要(必填)
|
||||
*/
|
||||
private String summary;
|
||||
|
||||
/**
|
||||
* 文档类别(可选)
|
||||
*/
|
||||
private String category;
|
||||
|
||||
/**
|
||||
* 章节锚点(预留字段,MVP 不使用)
|
||||
* Key: 章节标题,Value: 章节 Markdown 标题
|
||||
*/
|
||||
private Map<String, String> sections;
|
||||
|
||||
/**
|
||||
* 版本号(预留字段)
|
||||
*/
|
||||
private String version;
|
||||
|
||||
/**
|
||||
* 作者(预留字段)
|
||||
*/
|
||||
private String author;
|
||||
|
||||
/**
|
||||
* 最后更新日期(预留字段)
|
||||
*/
|
||||
private LocalDate lastUpdated;
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* 知识库索引条目
|
||||
* L0 内存索引使用的数据结构
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
public class KnowledgeEntry {
|
||||
|
||||
/**
|
||||
* 文件路径(如:knowledge_base/api/payment-errors.md)
|
||||
*/
|
||||
private String filePath;
|
||||
|
||||
/**
|
||||
* 文档标题
|
||||
*/
|
||||
private String title;
|
||||
|
||||
/**
|
||||
* 关键词列表(用于精确匹配)
|
||||
*/
|
||||
private List<String> keywords;
|
||||
|
||||
/**
|
||||
* 文档摘要
|
||||
*/
|
||||
private String summary;
|
||||
|
||||
/**
|
||||
* 文档类别(如:api、domain、troubleshooting)
|
||||
*/
|
||||
private String category;
|
||||
|
||||
/**
|
||||
* 章节锚点(预留字段,MVP 不使用)
|
||||
*/
|
||||
private Map<String, String> sections;
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 知识库查询结果
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
public class LookupResult {
|
||||
|
||||
/**
|
||||
* 是否找到结果
|
||||
*/
|
||||
private boolean found;
|
||||
|
||||
/**
|
||||
* 主要结果(L0 精确匹配)
|
||||
*/
|
||||
private PrimaryResult primary;
|
||||
|
||||
/**
|
||||
* 补充结果(L1 语义检索)
|
||||
*/
|
||||
private SupplementResult supplement;
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* L0 精确匹配结果
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
public class PrimaryResult {
|
||||
|
||||
/**
|
||||
* 文档内容(前 2000 字符)
|
||||
*/
|
||||
private String content;
|
||||
|
||||
/**
|
||||
* 文档来源路径
|
||||
*/
|
||||
private String source;
|
||||
|
||||
/**
|
||||
* 匹配类型(exact_L0)
|
||||
*/
|
||||
private String matchType;
|
||||
|
||||
/**
|
||||
* 置信度(high / low)
|
||||
*/
|
||||
private String confidence;
|
||||
|
||||
/**
|
||||
* 可用的章节列表(预留字段,MVP 返回 null)
|
||||
*/
|
||||
private List<String> availableSections;
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
package com.superbiz.agent.dto;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Data;
|
||||
|
||||
/**
|
||||
* L1 语义检索补充结果
|
||||
*/
|
||||
@Data
|
||||
@Builder
|
||||
public class SupplementResult {
|
||||
|
||||
/**
|
||||
* 文档内容片段
|
||||
*/
|
||||
private String content;
|
||||
|
||||
/**
|
||||
* 文档来源
|
||||
*/
|
||||
private String source;
|
||||
|
||||
/**
|
||||
* 匹配类型(semantic_L1)
|
||||
*/
|
||||
private String matchType;
|
||||
}
|
||||
@@ -1,14 +1,18 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.domain.entity.ApiDocument;
|
||||
import com.superbiz.agent.domain.enums.FaultCategory;
|
||||
import com.superbiz.agent.dto.DocumentChunk;
|
||||
import com.superbiz.agent.dto.DocumentQueryResponse;
|
||||
import com.superbiz.agent.dto.DocumentUploadRequest;
|
||||
import com.superbiz.agent.dto.Frontmatter;
|
||||
import com.superbiz.agent.dto.KnowledgeEntry;
|
||||
import com.superbiz.agent.exception.DocumentProcessException;
|
||||
import com.superbiz.agent.repository.ApiDocumentRepository;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.data.domain.Page;
|
||||
import org.springframework.data.domain.PageRequest;
|
||||
import org.springframework.stereotype.Service;
|
||||
@@ -16,6 +20,9 @@ import org.springframework.transaction.annotation.Transactional;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.Paths;
|
||||
import java.security.MessageDigest;
|
||||
import java.time.LocalDateTime;
|
||||
import java.util.List;
|
||||
@@ -30,6 +37,9 @@ import java.util.stream.Collectors;
|
||||
@Service
|
||||
public class DocumentManagementService {
|
||||
|
||||
@Value("${knowledge.base-path}")
|
||||
private String knowledgeBasePath;
|
||||
|
||||
@Autowired
|
||||
private TextExtractorService textExtractorService;
|
||||
|
||||
@@ -42,6 +52,15 @@ public class DocumentManagementService {
|
||||
@Autowired
|
||||
private ApiDocumentRepository apiDocumentRepository;
|
||||
|
||||
@Autowired
|
||||
private FrontmatterParser frontmatterParser;
|
||||
|
||||
@Autowired
|
||||
private KnowledgeIndexService knowledgeIndexService;
|
||||
|
||||
@Autowired
|
||||
private ObjectMapper objectMapper;
|
||||
|
||||
/**
|
||||
* 上传文档
|
||||
*
|
||||
@@ -52,82 +71,149 @@ public class DocumentManagementService {
|
||||
public String uploadDocument(DocumentUploadRequest request) {
|
||||
MultipartFile file = request.getFile();
|
||||
String fileName = file.getOriginalFilename();
|
||||
String localPath = null;
|
||||
long startTime = System.currentTimeMillis();
|
||||
|
||||
log.info("开始上传文档,文件名: {}, 大小: {} bytes", fileName, file.getSize());
|
||||
|
||||
// 1. 验证文件格式
|
||||
if (!textExtractorService.isSupportedFormat(fileName)) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"不支持的文件格式,仅支持 .md 和 .txt"
|
||||
);
|
||||
}
|
||||
|
||||
// 2. 计算文件 hash(去重)
|
||||
String fileHash = calculateFileHash(file);
|
||||
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
|
||||
if (existing.isPresent()) {
|
||||
log.warn("文档已存在,hash: {}, docId: ", fileHash, existing.get().getDocId());
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"文档已存在,docId: " + existing.get().getDocId()
|
||||
);
|
||||
}
|
||||
|
||||
// 3. 提取文本
|
||||
String text = textExtractorService.extractText(file, fileName);
|
||||
if (text == null || text.isBlank()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
|
||||
}
|
||||
|
||||
// 4. 分块(使用 DocumentChunkService 默认配置)
|
||||
// 注意:chunkSize 和 overlap 参数由 DocumentChunkConfig 配置,暂不支持动态调整
|
||||
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
|
||||
|
||||
if (chunks.isEmpty()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
|
||||
}
|
||||
|
||||
log.info("文档分块完成,文件名: {}, 分块数: {}", fileName, chunks.size());
|
||||
|
||||
// 5. 创建文档元数据
|
||||
String docId = UUID.randomUUID().toString();
|
||||
ApiDocument document = ApiDocument.builder()
|
||||
.docId(docId)
|
||||
.fileName(fileName)
|
||||
.faultCategory(parseFaultCategory(request.getFaultCategory()))
|
||||
.faultSource(request.getFaultSource())
|
||||
.apiName(request.getApiName())
|
||||
.version(request.getVersion())
|
||||
.fileSize(file.getSize())
|
||||
.fileHash(fileHash)
|
||||
.status("PROCESSING")
|
||||
.chunkCount(chunks.size())
|
||||
.build();
|
||||
|
||||
apiDocumentRepository.save(document);
|
||||
log.info("文档元数据已保存,docId: {}", docId);
|
||||
|
||||
// 6. 向量化并索引
|
||||
try {
|
||||
// 1. 验证文件格式
|
||||
if (!textExtractorService.isSupportedFormat(fileName)) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"不支持的文件格式,仅支持 .md 和 .txt"
|
||||
);
|
||||
}
|
||||
|
||||
// 2. 计算文件 hash(去重)
|
||||
long hashStart = System.currentTimeMillis();
|
||||
String fileHash = calculateFileHash(file);
|
||||
log.debug("文件hash计算完成: hash={}, time={}ms", fileHash, System.currentTimeMillis() - hashStart);
|
||||
|
||||
Optional<ApiDocument> existing = apiDocumentRepository.findByFileHash(fileHash);
|
||||
if (existing.isPresent()) {
|
||||
log.warn("文档已存在,hash: {}, docId: {}", fileHash, existing.get().getDocId());
|
||||
throw new DocumentProcessException(
|
||||
fileName, "upload",
|
||||
"文档已存在,docId: " + existing.get().getDocId()
|
||||
);
|
||||
}
|
||||
|
||||
// 3. 提取文本
|
||||
long extractStart = System.currentTimeMillis();
|
||||
String text = textExtractorService.extractText(file, fileName);
|
||||
log.debug("文本提取完成: length={}, time={}ms", text != null ? text.length() : 0, System.currentTimeMillis() - extractStart);
|
||||
|
||||
if (text == null || text.isBlank()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档内容为空");
|
||||
}
|
||||
|
||||
// 4. 保存原始文件到本地
|
||||
String category = request.getCategory();
|
||||
if (category == null || category.isBlank()) {
|
||||
category = "upload"; // 默认类别
|
||||
category = "default";
|
||||
}
|
||||
vectorIndexService.indexDocumentChunks(docId, chunks, category);
|
||||
document.setStatus("INDEXED");
|
||||
document.setIndexedAt(LocalDateTime.now());
|
||||
long saveStart = System.currentTimeMillis();
|
||||
localPath = saveToLocal(file, fileName, category);
|
||||
log.debug("文件保存到本地完成: path={}, time={}ms", localPath, System.currentTimeMillis() - saveStart);
|
||||
|
||||
// 5. 解析 frontmatter
|
||||
long frontmatterStart = System.currentTimeMillis();
|
||||
Frontmatter frontmatter = null;
|
||||
if (frontmatterParser.hasFrontmatter(text)) {
|
||||
frontmatter = frontmatterParser.parse(text);
|
||||
if (frontmatter != null) {
|
||||
log.info("解析到frontmatter: title={}, keywords={}, time={}ms",
|
||||
frontmatter.getTitle(), frontmatter.getKeywords(), System.currentTimeMillis() - frontmatterStart);
|
||||
} else {
|
||||
log.warn("frontmatter解析失败,文件名: {}", fileName);
|
||||
}
|
||||
} else {
|
||||
log.debug("文件不包含frontmatter: {}", fileName);
|
||||
}
|
||||
|
||||
// 6. 分块
|
||||
long chunkStart = System.currentTimeMillis();
|
||||
List<DocumentChunk> chunks = documentChunkService.chunkDocument(text, fileName);
|
||||
if (chunks.isEmpty()) {
|
||||
throw new DocumentProcessException(fileName, "upload", "文档分块失败");
|
||||
}
|
||||
log.info("文档分块完成: fileName={}, chunks={}, time={}ms",
|
||||
fileName, chunks.size(), System.currentTimeMillis() - chunkStart);
|
||||
|
||||
// 7. 创建文档元数据
|
||||
String docId = UUID.randomUUID().toString();
|
||||
String metadataJson = null;
|
||||
if (frontmatter != null) {
|
||||
try {
|
||||
metadataJson = objectMapper.writeValueAsString(frontmatter);
|
||||
} catch (Exception e) {
|
||||
log.warn("Frontmatter序列化失败", e);
|
||||
}
|
||||
}
|
||||
|
||||
ApiDocument document = ApiDocument.builder()
|
||||
.docId(docId)
|
||||
.fileName(fileName)
|
||||
.filePath(localPath)
|
||||
.metadata(metadataJson)
|
||||
.faultCategory(parseFaultCategory(request.getFaultCategory()))
|
||||
.faultSource(request.getFaultSource())
|
||||
.apiName(request.getApiName())
|
||||
.version(request.getVersion())
|
||||
.fileSize(file.getSize())
|
||||
.fileHash(fileHash)
|
||||
.status("PROCESSING")
|
||||
.chunkCount(chunks.size())
|
||||
.build();
|
||||
|
||||
apiDocumentRepository.save(document);
|
||||
log.info("文档索引完成,docId: {}, 类别: {}", docId, category);
|
||||
log.info("文档元数据已保存: docId={}", docId);
|
||||
|
||||
// 8. 向量化并索引
|
||||
try {
|
||||
long vectorStart = System.currentTimeMillis();
|
||||
vectorIndexService.indexDocumentChunks(docId, chunks, category);
|
||||
document.setStatus("INDEXED");
|
||||
document.setIndexedAt(LocalDateTime.now());
|
||||
apiDocumentRepository.save(document);
|
||||
log.info("文档向量索引完成: docId={}, category={}, time={}ms",
|
||||
docId, category, System.currentTimeMillis() - vectorStart);
|
||||
|
||||
} catch (Exception e) {
|
||||
log.error("文档索引失败: docId={}", docId, e);
|
||||
document.setStatus("FAILED");
|
||||
apiDocumentRepository.save(document);
|
||||
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
|
||||
}
|
||||
|
||||
// 9. 更新 L0 索引
|
||||
if (frontmatter != null) {
|
||||
KnowledgeEntry entry = KnowledgeEntry.builder()
|
||||
.filePath(localPath)
|
||||
.title(frontmatter.getTitle())
|
||||
.keywords(frontmatter.getKeywords())
|
||||
.summary(frontmatter.getSummary())
|
||||
.category(category)
|
||||
.sections(frontmatter.getSections())
|
||||
.build();
|
||||
|
||||
knowledgeIndexService.addToIndex(entry);
|
||||
log.info("文档已加入L0索引: docId={}, title={}", docId, frontmatter.getTitle());
|
||||
}
|
||||
|
||||
long totalTime = System.currentTimeMillis() - startTime;
|
||||
log.info("文档上传完成: docId={}, fileName={}, hasFrontmatter={}, totalTime={}ms",
|
||||
docId, fileName, frontmatter != null, totalTime);
|
||||
|
||||
return docId;
|
||||
|
||||
} catch (Exception e) {
|
||||
log.error("文档索引失败,docId: {}", docId, e);
|
||||
document.setStatus("FAILED");
|
||||
apiDocumentRepository.save(document);
|
||||
throw new DocumentProcessException(docId, "index", "向量化索引失败: " + e.getMessage(), e);
|
||||
// 失败时清理本地文件
|
||||
cleanupLocalFile(localPath);
|
||||
log.error("文档上传失败: fileName={}", fileName, e);
|
||||
throw e;
|
||||
}
|
||||
|
||||
return docId;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -150,6 +236,52 @@ public class DocumentManagementService {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 保存文件到本地
|
||||
*
|
||||
* @param file 上传的文件
|
||||
* @param fileName 文件名
|
||||
* @param category 类别
|
||||
* @return 本地文件路径
|
||||
*/
|
||||
private String saveToLocal(MultipartFile file, String fileName, String category) {
|
||||
try {
|
||||
// 1. 构建目标路径
|
||||
Path categoryDir = Paths.get(knowledgeBasePath, category);
|
||||
Files.createDirectories(categoryDir);
|
||||
|
||||
Path targetPath = categoryDir.resolve(fileName);
|
||||
|
||||
// 2. 保存文件
|
||||
file.transferTo(targetPath.toFile());
|
||||
|
||||
log.info("文件已保存到本地: {}", targetPath);
|
||||
return targetPath.toString();
|
||||
|
||||
} catch (IOException e) {
|
||||
throw new DocumentProcessException(
|
||||
fileName, "save-local",
|
||||
"保存文件到本地失败: " + e.getMessage(), e
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 清理本地文件(事务回滚时调用)
|
||||
*
|
||||
* @param localPath 本地文件路径
|
||||
*/
|
||||
private void cleanupLocalFile(String localPath) {
|
||||
if (localPath != null) {
|
||||
try {
|
||||
Files.deleteIfExists(Paths.get(localPath));
|
||||
log.info("已清理本地文件: {}", localPath);
|
||||
} catch (IOException e) {
|
||||
log.warn("清理本地文件失败: {}", localPath, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 解析故障类别
|
||||
*/
|
||||
@@ -209,6 +341,21 @@ public class DocumentManagementService {
|
||||
|
||||
ApiDocument doc = optional.get();
|
||||
|
||||
// 删除本地文件
|
||||
if (doc.getFilePath() != null) {
|
||||
try {
|
||||
Files.deleteIfExists(Paths.get(doc.getFilePath()));
|
||||
log.info("本地文件已删除: {}", doc.getFilePath());
|
||||
} catch (IOException e) {
|
||||
log.warn("删除本地文件失败: {}", doc.getFilePath(), e);
|
||||
}
|
||||
}
|
||||
|
||||
// 删除 L0 索引
|
||||
if (doc.getFilePath() != null) {
|
||||
knowledgeIndexService.removeFromIndex(doc.getFilePath());
|
||||
}
|
||||
|
||||
// 删除向量索引
|
||||
try {
|
||||
vectorIndexService.deleteDocumentChunks(docId);
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.dto.Frontmatter;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.yaml.snakeyaml.Yaml;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Frontmatter 解析器
|
||||
* 解析 Markdown 文件头的 YAML frontmatter
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
public class FrontmatterParser {
|
||||
|
||||
private final Yaml yaml = new Yaml();
|
||||
|
||||
/**
|
||||
* 检查文件是否包含 frontmatter
|
||||
*
|
||||
* @param content 文件内容
|
||||
* @return true 如果包含 frontmatter
|
||||
*/
|
||||
public boolean hasFrontmatter(String content) {
|
||||
if (content == null || content.isEmpty()) {
|
||||
return false;
|
||||
}
|
||||
return content.trim().startsWith("---");
|
||||
}
|
||||
|
||||
/**
|
||||
* 解析 Markdown frontmatter
|
||||
*
|
||||
* @param content 完整文件内容
|
||||
* @return Frontmatter 对象,如果不存在或解析失败返回 null
|
||||
*/
|
||||
public Frontmatter parse(String content) {
|
||||
if (!hasFrontmatter(content)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
try {
|
||||
// 1. 提取 frontmatter 部分(两个 --- 之间)
|
||||
String frontmatterText = extractFrontmatter(content);
|
||||
if (frontmatterText == null) {
|
||||
log.warn("未找到有效的 frontmatter 结束标记");
|
||||
return null;
|
||||
}
|
||||
|
||||
// 2. 使用 SnakeYAML 解析
|
||||
Map<String, Object> map = yaml.load(frontmatterText);
|
||||
if (map == null || map.isEmpty()) {
|
||||
log.warn("Frontmatter 解析结果为空");
|
||||
return null;
|
||||
}
|
||||
|
||||
// 3. 映射到 Frontmatter 对象
|
||||
Frontmatter frontmatter = Frontmatter.builder()
|
||||
.title((String) map.get("title"))
|
||||
.keywords((java.util.List<String>) map.get("keywords"))
|
||||
.summary((String) map.get("summary"))
|
||||
.category((String) map.get("category"))
|
||||
.sections((Map<String, String>) map.get("sections"))
|
||||
.version((String) map.get("version"))
|
||||
.author((String) map.get("author"))
|
||||
.build();
|
||||
|
||||
// 4. 验证必填字段
|
||||
if (frontmatter.getTitle() == null || frontmatter.getKeywords() == null ||
|
||||
frontmatter.getSummary() == null) {
|
||||
log.warn("Frontmatter 缺少必填字段: title={}, keywords={}, summary={}",
|
||||
frontmatter.getTitle(), frontmatter.getKeywords(), frontmatter.getSummary());
|
||||
return null;
|
||||
}
|
||||
|
||||
log.debug("Frontmatter 解析成功: title={}, keywords=",
|
||||
frontmatter.getTitle(), frontmatter.getKeywords());
|
||||
return frontmatter;
|
||||
|
||||
} catch (Exception e) {
|
||||
log.warn("Frontmatter 解析失败", e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 提取 frontmatter 文本(两个 --- 之间的内容)
|
||||
*
|
||||
* @param content 完整文件内容
|
||||
* @return frontmatter 文本,如果格式错误返回 null
|
||||
*/
|
||||
private String extractFrontmatter(String content) {
|
||||
// 去除开头的空白
|
||||
content = content.trim();
|
||||
|
||||
// 检查是否以 --- 开头
|
||||
if (!content.startsWith("---")) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// 查找第二个 ---(结束标记)
|
||||
int secondDelimiter = content.indexOf("\n---", 3);
|
||||
if (secondDelimiter == -1) {
|
||||
// 尝试查找 Windows 风格换行
|
||||
secondDelimiter = content.indexOf("\r\n---", 3);
|
||||
if (secondDelimiter == -1) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// 提取 frontmatter(不包含 --- 标记)
|
||||
return content.substring(3, secondDelimiter).trim();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,225 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.dto.Frontmatter;
|
||||
import com.superbiz.agent.dto.KnowledgeEntry;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import jakarta.annotation.PostConstruct;
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.Paths;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* 知识库索引服务
|
||||
* 负责 L0 精确匹配索引的管理
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
public class KnowledgeIndexService {
|
||||
|
||||
@Value("${knowledge.base-path}")
|
||||
private String knowledgeBasePath;
|
||||
|
||||
@Autowired
|
||||
private FrontmatterParser frontmatterParser;
|
||||
|
||||
/**
|
||||
* 内存索引(线程安全)
|
||||
*/
|
||||
private final List<KnowledgeEntry> knowledgeIndex = new CopyOnWriteArrayList<>();
|
||||
|
||||
/**
|
||||
* 启动时扫描知识库目录,构建索引
|
||||
*/
|
||||
@PostConstruct
|
||||
public void loadIndex() {
|
||||
log.info("开始扫描知识库目录: {}", knowledgeBasePath);
|
||||
|
||||
try {
|
||||
Path basePath = Paths.get(knowledgeBasePath);
|
||||
|
||||
// 目录不存在时自动创建
|
||||
if (!Files.exists(basePath)) {
|
||||
Files.createDirectories(basePath);
|
||||
log.info("知识库目录已创建: {}", basePath.toAbsolutePath());
|
||||
}
|
||||
|
||||
// 递归扫描 .md 文件
|
||||
try (Stream<Path> paths = Files.walk(basePath)) {
|
||||
paths.filter(p -> p.toString().endsWith(".md"))
|
||||
.forEach(this::indexFile);
|
||||
}
|
||||
|
||||
log.info("知识库索引加载完成,共 {} 个文档", knowledgeIndex.size());
|
||||
|
||||
} catch (IOException e) {
|
||||
log.error("知识库索引加载失败", e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 索引单个文件
|
||||
*
|
||||
* @param filePath 文件路径
|
||||
*/
|
||||
private void indexFile(Path filePath) {
|
||||
try {
|
||||
// 读取文件内容
|
||||
String content = Files.readString(filePath);
|
||||
|
||||
// 解析 frontmatter
|
||||
Frontmatter frontmatter = frontmatterParser.parse(content);
|
||||
if (frontmatter == null) {
|
||||
log.debug("跳过文件(无有效 frontmatter): {}", filePath);
|
||||
return;
|
||||
}
|
||||
|
||||
// 提取 category(从路径中获取)
|
||||
String category = extractCategoryFromPath(filePath.toString());
|
||||
|
||||
// 构建索引条目
|
||||
KnowledgeEntry entry = KnowledgeEntry.builder()
|
||||
.filePath(filePath.toString())
|
||||
.title(frontmatter.getTitle())
|
||||
.keywords(frontmatter.getKeywords())
|
||||
.summary(frontmatter.getSummary())
|
||||
.category(category)
|
||||
.sections(frontmatter.getSections())
|
||||
.build();
|
||||
|
||||
knowledgeIndex.add(entry);
|
||||
log.debug("文档已加入索引: title={}, filePath={}", entry.getTitle(), filePath);
|
||||
|
||||
} catch (IOException e) {
|
||||
log.warn("读取文件失败: {}", filePath, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 从文件路径中提取 category
|
||||
* 例如:knowledge_base/api/test.md -> api
|
||||
*/
|
||||
private String extractCategoryFromPath(String filePath) {
|
||||
String normalized = filePath.replace("\\", "/");
|
||||
String[] parts = normalized.split("/");
|
||||
|
||||
// 查找 knowledge_base 后的第一个目录
|
||||
for (int i = 0; i < parts.length - 1; i++) {
|
||||
if (parts[i].equals("knowledge_base") && i + 1 < parts.length) {
|
||||
return parts[i + 1];
|
||||
}
|
||||
}
|
||||
|
||||
return "default";
|
||||
}
|
||||
|
||||
/**
|
||||
* L0 精确匹配
|
||||
*
|
||||
* @param query 查询关键词
|
||||
* @return 匹配的文档列表
|
||||
*/
|
||||
public List<KnowledgeEntry> exactMatch(String query) {
|
||||
long startTime = System.currentTimeMillis();
|
||||
|
||||
if (query == null || query.trim().isEmpty()) {
|
||||
log.debug("查询关键词为空,返回空结果");
|
||||
return List.of();
|
||||
}
|
||||
|
||||
String queryLower = query.toLowerCase();
|
||||
|
||||
List<KnowledgeEntry> results = knowledgeIndex.stream()
|
||||
.filter(entry -> matchesKeywords(entry, queryLower))
|
||||
.collect(Collectors.toList());
|
||||
|
||||
long elapsedTime = System.currentTimeMillis() - startTime;
|
||||
log.debug("L0精确匹配: query={}, matches={}, indexSize={}, time={}ms",
|
||||
query, results.size(), knowledgeIndex.size(), elapsedTime);
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
/**
|
||||
* 关键词匹配逻辑(不区分大小写)
|
||||
*
|
||||
* @param entry 索引条目
|
||||
* @param query 查询关键词(小写)
|
||||
* @return true 如果匹配
|
||||
*/
|
||||
private boolean matchesKeywords(KnowledgeEntry entry, String query) {
|
||||
if (entry.getKeywords() == null || entry.getKeywords().isEmpty()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
for (String keyword : entry.getKeywords()) {
|
||||
String keywordLower = keyword.toLowerCase();
|
||||
// query 包含 keyword 或 keyword 包含 query
|
||||
if (query.contains(keywordLower) || keywordLower.contains(query)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* 读取文档内容
|
||||
*
|
||||
* @param filePath 文件路径
|
||||
* @param maxChars 最大字符数
|
||||
* @return 文档内容(前 maxChars 字符),失败返回 null
|
||||
*/
|
||||
public String readDocument(String filePath, int maxChars) {
|
||||
try {
|
||||
String content = Files.readString(Paths.get(filePath));
|
||||
|
||||
if (content.length() > maxChars) {
|
||||
return content.substring(0, maxChars) + "...";
|
||||
}
|
||||
|
||||
return content;
|
||||
|
||||
} catch (IOException e) {
|
||||
log.error("读取文档失败: {}", filePath, e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 添加文档到索引(上传时调用)
|
||||
*
|
||||
* @param entry 知识库条目
|
||||
*/
|
||||
public void addToIndex(KnowledgeEntry entry) {
|
||||
knowledgeIndex.add(entry);
|
||||
log.debug("文档已添加到 L0 索引: title={}", entry.getTitle());
|
||||
}
|
||||
|
||||
/**
|
||||
* 从索引中移除文档(删除时调用)
|
||||
*
|
||||
* @param filePath 文件路径
|
||||
*/
|
||||
public void removeFromIndex(String filePath) {
|
||||
knowledgeIndex.removeIf(e -> e.getFilePath().equals(filePath));
|
||||
log.debug("文档已从 L0 索引移除: {}", filePath);
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取索引大小
|
||||
*
|
||||
* @return 索引中的文档数量
|
||||
*/
|
||||
public int getIndexSize() {
|
||||
return knowledgeIndex.size();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
package com.superbiz.agent.tool;
|
||||
|
||||
import com.superbiz.agent.dto.*;
|
||||
import com.superbiz.agent.service.KnowledgeIndexService;
|
||||
import com.superbiz.agent.service.VectorSearchService;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.ai.tool.annotation.Tool;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 知识库查询工具
|
||||
* 提供给 Agent 的混合检索工具(L0 + L1)
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
public class LookupKnowledgeTool {
|
||||
|
||||
@Autowired
|
||||
private KnowledgeIndexService knowledgeIndexService;
|
||||
|
||||
@Autowired
|
||||
private VectorSearchService vectorSearchService;
|
||||
|
||||
/**
|
||||
* 查询知识库文档
|
||||
*
|
||||
* @param query 查询关键词
|
||||
* @return 查询结果
|
||||
*/
|
||||
@Tool(description = "查询知识库文档。优先精确匹配关键词,未命中或多个匹配时自动补充语义相关片段。" +
|
||||
"参数 query: 查询关键词,例如 'ERR_TIMEOUT'、'支付网关超时'")
|
||||
public LookupResult lookupKnowledge(String query) {
|
||||
// 生成请求ID用于追踪
|
||||
String requestId = java.util.UUID.randomUUID().toString().substring(0, 8);
|
||||
long startTime = System.currentTimeMillis();
|
||||
|
||||
log.info("[{}] 收到知识库查询请求: query={}", requestId, query);
|
||||
|
||||
// Step 1: L0 精确匹配
|
||||
long l0Start = System.currentTimeMillis();
|
||||
List<KnowledgeEntry> l0Matches = knowledgeIndexService.exactMatch(query);
|
||||
long l0Time = System.currentTimeMillis() - l0Start;
|
||||
log.info("[{}] L0精确匹配完成: matches={}, time={}ms", requestId, l0Matches.size(), l0Time);
|
||||
|
||||
// Step 2: 判断是否高置信度(唯一匹配)
|
||||
boolean highConfidence = (l0Matches.size() == 1);
|
||||
log.debug("[{}] 置信度判断: highConfidence={}, reason={}",
|
||||
requestId, highConfidence, highConfidence ? "唯一匹配" : "多个或零个匹配");
|
||||
|
||||
// Step 3: L1 条件调用
|
||||
List<VectorSearchService.SearchResult> l1Results = null;
|
||||
if (!highConfidence) {
|
||||
log.info("[{}] L0非唯一匹配,触发L1语义检索", requestId);
|
||||
long l1Start = System.currentTimeMillis();
|
||||
l1Results = vectorSearchService.searchSimilarDocuments(query, 3, null);
|
||||
long l1Time = System.currentTimeMillis() - l1Start;
|
||||
log.info("[{}] L1语义检索完成: matches={}, time={}ms",
|
||||
requestId, l1Results != null ? l1Results.size() : 0, l1Time);
|
||||
} else {
|
||||
log.debug("[{}] L0唯一匹配,跳过L1检索", requestId);
|
||||
}
|
||||
|
||||
// Step 4: 组装结果
|
||||
LookupResult result = buildResult(l0Matches, l1Results, highConfidence);
|
||||
|
||||
// 记录完整结果
|
||||
long totalTime = System.currentTimeMillis() - startTime;
|
||||
log.info("[{}] 查询完成: found={}, hasL0={}, hasL1={}, confidence={}, totalTime={}ms",
|
||||
requestId,
|
||||
result.isFound(),
|
||||
result.getPrimary() != null,
|
||||
result.getSupplement() != null,
|
||||
result.getPrimary() != null ? result.getPrimary().getConfidence() : "N/A",
|
||||
totalTime);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* 组装查询结果
|
||||
*
|
||||
* @param l0Matches L0 匹配结果
|
||||
* @param l1Results L1 检索结果
|
||||
* @param highConfidence 是否高置信度
|
||||
* @return 组装后的结果
|
||||
*/
|
||||
private LookupResult buildResult(
|
||||
List<KnowledgeEntry> l0Matches,
|
||||
List<VectorSearchService.SearchResult> l1Results,
|
||||
boolean highConfidence
|
||||
) {
|
||||
LookupResult.LookupResultBuilder builder = LookupResult.builder();
|
||||
|
||||
// 构建 primary(L0 结果)
|
||||
PrimaryResult primary = null;
|
||||
if (l0Matches != null && !l0Matches.isEmpty()) {
|
||||
KnowledgeEntry first = l0Matches.get(0);
|
||||
String content = knowledgeIndexService.readDocument(first.getFilePath(), 2000);
|
||||
|
||||
if (content != null) {
|
||||
primary = PrimaryResult.builder()
|
||||
.content(content)
|
||||
.source(first.getFilePath())
|
||||
.matchType("exact_L0")
|
||||
.confidence(highConfidence ? "high" : "low")
|
||||
.availableSections(null) // MVP 返回 null
|
||||
.build();
|
||||
log.debug("L0结果已构建: source={}, contentLength={}", first.getFilePath(), content.length());
|
||||
} else {
|
||||
log.warn("L0匹配但文件读取失败: {}", first.getFilePath());
|
||||
}
|
||||
}
|
||||
builder.primary(primary);
|
||||
|
||||
// 构建 supplement(L1 结果)
|
||||
SupplementResult supplement = null;
|
||||
boolean hasL1 = l1Results != null && !l1Results.isEmpty();
|
||||
if (hasL1) {
|
||||
VectorSearchService.SearchResult firstL1 = l1Results.get(0);
|
||||
supplement = SupplementResult.builder()
|
||||
.content(firstL1.getContent())
|
||||
.source(firstL1.getMetadata())
|
||||
.matchType("semantic_L1")
|
||||
.build();
|
||||
log.debug("L1结果已构建: source={}, score={}", firstL1.getMetadata(), firstL1.getScore());
|
||||
}
|
||||
builder.supplement(supplement);
|
||||
|
||||
// 判断是否找到结果(primary 或 supplement 至少有一个)
|
||||
boolean found = (primary != null) || (supplement != null);
|
||||
builder.found(found);
|
||||
|
||||
return builder.build();
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,10 @@ file:
|
||||
path: ./uploads
|
||||
allowed-extensions: txt,md
|
||||
|
||||
# 知识库配置
|
||||
knowledge:
|
||||
base-path: knowledge_base/
|
||||
|
||||
milvus:
|
||||
host: in03-4a578da0f27ce9d.serverless.aws-eu-central-1.cloud.zilliz.com
|
||||
port: 443
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
-- V004: 添加 metadata 字段到 api_document 表
|
||||
-- 用于存储 frontmatter 元数据(JSON 格式)
|
||||
|
||||
ALTER TABLE api_document
|
||||
ADD COLUMN metadata TEXT COMMENT 'Frontmatter 元数据 (JSON)';
|
||||
Reference in New Issue
Block a user