diff --git a/.docs/2026-06-25-knowledge-base-init-api.md b/.docs/2026-06-25-knowledge-base-init-api.md index 8cad35e..1b9e601 100644 --- a/.docs/2026-06-25-knowledge-base-init-api.md +++ b/.docs/2026-06-25-knowledge-base-init-api.md @@ -2,14 +2,14 @@ ## 概述 -提供了知识库批量初始化接口,用于将 `knowledge_base` 目录下的所有 Markdown 文档导入到数据库和向量索引(L0)。 +提供了知识库批量初始化接口,用于将 `knowledge_base` 目录下的所有 Markdown 文档导入到数据库和向量索引(L0 + L1)。 **功能特点**: 1. ✅ **批量扫描**:递归扫描 knowledge_base 目录下所有 .md 文件 2. ✅ **自动去重**:基于文件路径检查,避免重复导入 3. ✅ **数据入库**:保存文档元数据到 MySQL 4. ✅ **L0 索引**:自动加入内存精确匹配索引 -5. ⏳ **L1 索引**:暂未实现,需要后续通过独立索引任务完成 +5. ✅ **L1 索引**:文档分块并上传到 Milvus 向量数据库 --- @@ -46,12 +46,12 @@ curl -X POST http://localhost:9900/api/knowledge/init?force=true "inserted": 6, "failed": 0, "details": { - "api/payment-errors.md": "导入成功(L0)", - "domain/spring-ai-tool-best-practices.md": "导入成功(L0)", - "infrastructure/flyway-best-practices.md": "导入成功(L0)", - "infrastructure/mysql-connection-pool.md": "导入成功(L0)", - "infrastructure/redis-config.md": "导入成功(L0)", - "troubleshooting/fault-diagnosis-process.md": "导入成功(L0)" + "api/payment-errors.md": "导入成功(L0+L1)", + "domain/spring-ai-tool-best-practices.md": "导入成功(L0+L1)", + "infrastructure/flyway-best-practices.md": "导入成功(L0+L1)", + "infrastructure/mysql-connection-pool.md": "导入成功(L0+L1)", + "infrastructure/redis-config.md": "导入成功(L0+L1)", + "troubleshooting/fault-diagnosis-process.md": "导入成功(L0+L1)" } } ``` @@ -82,7 +82,7 @@ curl http://localhost:9900/api/knowledge/stats { "success": true, "totalDocuments": 6, - "totalVectors": 0, + "totalVectors": 48, "categories": { "api": 1, "domain": 1, @@ -94,7 +94,7 @@ curl http://localhost:9900/api/knowledge/stats **字段说明**: - `totalDocuments`:数据库中的文档总数 -- `totalVectors`:Milvus 中的向量总数(当前为 0,未实现) +- `totalVectors`:Milvus 中的向量总数(chunk 数量) - `categories`:按分类统计的文档数量 --- @@ -198,6 +198,28 @@ if (!force && existingFilePaths.contains(relativePath)) { ## 数据存储 +### 完整的数据流 + +``` +knowledge_base/*.md + ↓ 1. 扫描 +KnowledgeBaseInitService + ↓ 2. 解析 frontmatter +Frontmatter (title, keywords, summary) + ↓ 3. 保存到数据库 +MySQL (api_document) + ↓ 4. 提取正文 & 分块 +DocumentChunkService + ↓ 5. 生成向量 +VectorEmbeddingService + ↓ 6. 索引到 Milvus +Milvus (L1 向量索引) + ↓ 7. 加入内存索引 +KnowledgeIndexService (L0) +``` + +--- + ### 数据库表结构(api_document) | 字段 | 类型 | 说明 | 示例 | @@ -207,7 +229,9 @@ if (!force && existingFilePaths.contains(relativePath)) { | `file_name` | VARCHAR(256) | 文件名 | payment-errors.md | | `file_path` | VARCHAR(512) | 相对路径 | api/payment-errors.md | | `api_name` | VARCHAR(128) | 文档标题 | 支付网关错误码定义 | -| `status` | VARCHAR(16) | 状态 | INDEXED | +| `status` | VARCHAR(16) | 状态 | INDEXED / FAILED | +| `chunk_count` | INT | 分块数量 | 8 | +| `error_message` | TEXT | 错误信息 | null | | `metadata` | TEXT | Frontmatter JSON | {"title":"...","keywords":[...]} | | `file_size` | BIGINT | 文件大小(字节) | 2048 | | `indexed_at` | DATETIME | 索引时间 | 2026-06-25 10:00:00 | @@ -225,6 +249,26 @@ if (!force && existingFilePaths.contains(relativePath)) { --- +### Milvus 向量索引 + +每个文档会被分块(chunk)并生成向量,存储到 Milvus 集合中: + +**Collection**: `knowledge_base_collection` + +**字段**: +- `doc_id`:文档 ID +- `chunk_id`:分块 ID +- `chunk_text`:分块文本内容 +- `embedding`:768 维向量 +- `category`:文档分类 +- `file_path`:文件路径 + +**分块策略**: +- Chunk Size:根据 `DocumentChunkConfig` 配置(默认 500 token) +- Overlap:重叠区域(默认 50 token) + +--- + ## L0 内存索引 导入过程会自动将文档加入 `KnowledgeIndexService` 的内存索引: @@ -305,6 +349,61 @@ category: api --- +### 问题 4: Milvus 连接失败 + +**症状**: +```json +{ + "success": true, + "scanned": 6, + "inserted": 0, + "failed": 6, + "details": { + "api/payment-errors.md": "Milvus 索引失败: Connection refused" + } +} +``` + +**原因**: +- Milvus 服务未启动 +- 网络连接问题 +- 配置错误 + +**解决**: +```bash +# 检查 Milvus 是否运行 +docker ps | grep milvus + +# 检查配置 +grep milvus application.yml + +# 启动 Milvus +docker-compose up -d milvus-standalone +``` + +--- + +### 问题 5: 文档分块失败 + +**症状**: +```json +{ + "details": { + "test/large-doc.md": "Milvus 索引失败: Document too large" + } +} +``` + +**原因**: +- 文档内容过大 +- 分块配置不当 + +**解决**: +- 检查 `DocumentChunkConfig` 配置 +- 调整 chunk size 和 overlap + +--- + #### 3. 文档缺少标题 ```json { @@ -318,38 +417,6 @@ category: api --- -## 后续扩展(L1 向量索引) - -当前版本暂未实现 L1 向量索引(Milvus),计划后续扩展: - -### 扩展方案 - -1. **独立索引任务**: - ```bash - POST /api/knowledge/build-vectors - ``` - - 读取数据库中所有文档 - - 调用 `DocumentManagementService` 处理分块 - - 上传到 Milvus - -2. **或者修改当前接口**: - - 在 `initializeKnowledgeBase` 中调用文档分块和向量索引 - - 需要处理大文件的分块逻辑 - -### 验证 L1 的方法(未来) - -```bash -# 1. 调用 L1 索引构建 -curl -X POST http://localhost:9900/api/knowledge/build-vectors - -# 2. 查询统计信息 -curl http://localhost:9900/api/knowledge/stats - -# 3. 确认 totalVectors > 0 -``` - ---- - ## 最佳实践 ### ✅ 推荐做法 diff --git a/src/main/java/com/superbiz/agent/service/KnowledgeBaseInitService.java b/src/main/java/com/superbiz/agent/service/KnowledgeBaseInitService.java index 863d43b..5f910ce 100644 --- a/src/main/java/com/superbiz/agent/service/KnowledgeBaseInitService.java +++ b/src/main/java/com/superbiz/agent/service/KnowledgeBaseInitService.java @@ -4,6 +4,7 @@ import com.superbiz.agent.domain.entity.ApiDocument; import com.superbiz.agent.repository.ApiDocumentRepository; import com.superbiz.agent.dto.KnowledgeEntry; import com.superbiz.agent.dto.Frontmatter; +import com.superbiz.agent.dto.DocumentChunk; import lombok.Data; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -37,6 +38,15 @@ public class KnowledgeBaseInitService { @Autowired private FrontmatterParser frontmatterParser; + @Autowired + private DocumentChunkService documentChunkService; + + @Autowired + private VectorIndexService vectorIndexService; + + @Autowired + private VectorEmbeddingService vectorEmbeddingService; + @Autowired private KnowledgeIndexService knowledgeIndexService; @@ -112,6 +122,35 @@ public class KnowledgeBaseInitService { // 保存到数据库 ApiDocument document = saveToDatabase(relativePath, title, summary, category, content, keywords); + // 提取文档正文(去除 frontmatter) + String body = extractBody(content); + + // 文档分块 + List chunks = documentChunkService.chunkDocument(body, relativePath); + logger.debug("文档分块完成: {} -> {} 个 chunk", relativePath, chunks.size()); + + // 上传到 Milvus + try { + vectorIndexService.indexDocumentChunks(document.getDocId(), chunks, category); + + document.setStatus("INDEXED"); + document.setChunkCount(chunks.size()); + apiDocumentRepository.save(document); + + logger.info("文档已索引到 Milvus: {} (docId={}, chunks={})", + title, document.getDocId(), chunks.size()); + } catch (Exception e) { + logger.error("上传到 Milvus 失败: {}", relativePath, e); + + document.setStatus("FAILED"); + document.setErrorMessage(e.getMessage()); + apiDocumentRepository.save(document); + + result.incrementFailed(); + result.addDetail(relativePath, "Milvus 索引失败: " + e.getMessage()); + continue; // 跳过该文档,继续处理下一个 + } + // 添加到 L0 内存索引 KnowledgeEntry entry = KnowledgeEntry.builder() .filePath(relativePath) @@ -122,12 +161,9 @@ public class KnowledgeBaseInitService { .build(); knowledgeIndexService.addToIndex(entry); - // TODO: 上传到 Milvus (L1) - 需要通过独立的索引任务完成 - // 当前版本只处理数据库入库和 L0 索引 - result.incrementInserted(); - result.addDetail(relativePath, "导入成功(L0)"); - logger.info("文档导入成功: {} -> {} (L0 索引已更新)", relativePath, title); + result.addDetail(relativePath, "导入成功(L0+L1)"); + logger.info("文档导入成功: {} -> {} (L0+L1 索引已更新)", relativePath, title); } catch (Exception e) { logger.error("处理文档失败: {}", relativePath, e);