feat: include breadcrumb in embedding text
This commit is contained in:
@@ -170,7 +170,7 @@ Known gaps:
|
||||
- Metadata taxonomy still needs cleanup, for example `database` vs `infrastructure`.
|
||||
- Abstract design questions may need query rewriting or better indexed interview/devflow documents.
|
||||
- Chunk context reconstruction is still limited when one logical section spans multiple chunks.
|
||||
- `breadcrumb` should be used more strongly in embedding text, retrieval metadata, and evidence packing.
|
||||
- `breadcrumb` now participates in embedding text, but it can still be used more strongly in context expansion, rerank, and evidence packing.
|
||||
- Rerank, RRF, BM25, and hybrid retrieval are not implemented yet.
|
||||
- Indexing writes still use SDK.
|
||||
|
||||
|
||||
@@ -148,7 +148,7 @@ public class VectorIndexService {
|
||||
|
||||
try {
|
||||
// 生成向量
|
||||
List<Float> vector = embeddingService.generateEmbedding(chunk.getContent());
|
||||
List<Float> vector = embeddingService.generateEmbedding(buildEmbeddingText(chunk));
|
||||
|
||||
// 构建元数据(包含文件信息)
|
||||
Map<String, Object> metadata = buildMetadata(path.toString(), chunk, chunks.size());
|
||||
@@ -191,7 +191,7 @@ public class VectorIndexService {
|
||||
|
||||
try {
|
||||
// 生成向量
|
||||
List<Float> vector = embeddingService.generateEmbedding(chunk.getContent());
|
||||
List<Float> vector = embeddingService.generateEmbedding(buildEmbeddingText(chunk));
|
||||
|
||||
// 构建元数据(使用 docId 和 category)
|
||||
Map<String, Object> metadata = buildDocumentMetadata(docId, chunk, chunks.size(), category);
|
||||
@@ -280,6 +280,30 @@ public class VectorIndexService {
|
||||
return metadata;
|
||||
}
|
||||
|
||||
static String buildEmbeddingText(DocumentChunk chunk) {
|
||||
String content = trimToEmpty(chunk.getContent());
|
||||
String title = trimToEmpty(chunk.getTitle());
|
||||
String breadcrumb = trimToEmpty(chunk.getBreadcrumb());
|
||||
|
||||
if (title.isEmpty() && breadcrumb.isEmpty()) {
|
||||
return content;
|
||||
}
|
||||
|
||||
StringBuilder text = new StringBuilder();
|
||||
if (!title.isEmpty()) {
|
||||
text.append("Title: ").append(title).append("\n");
|
||||
}
|
||||
if (!breadcrumb.isEmpty()) {
|
||||
text.append("Path: ").append(breadcrumb).append("\n");
|
||||
}
|
||||
text.append("Content:\n").append(content);
|
||||
return text.toString();
|
||||
}
|
||||
|
||||
private static String trimToEmpty(String value) {
|
||||
return value == null ? "" : value.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* 删除文件的旧数据(根据 metadata._source)
|
||||
*/
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
package com.superbiz.agent.service;
|
||||
|
||||
import com.superbiz.agent.dto.DocumentChunk;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
class VectorIndexServiceTest {
|
||||
|
||||
@Test
|
||||
void buildEmbeddingTextIncludesTitleAndBreadcrumb() {
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.title("Connection Pool")
|
||||
.breadcrumb("Database > MySQL > Connection Pool")
|
||||
.content("Check active connections and leak detection.")
|
||||
.build();
|
||||
|
||||
String embeddingText = VectorIndexService.buildEmbeddingText(chunk);
|
||||
|
||||
assertEquals("""
|
||||
Title: Connection Pool
|
||||
Path: Database > MySQL > Connection Pool
|
||||
Content:
|
||||
Check active connections and leak detection.""", embeddingText);
|
||||
}
|
||||
|
||||
@Test
|
||||
void buildEmbeddingTextKeepsPlainContentWhenNoStructureExists() {
|
||||
DocumentChunk chunk = DocumentChunk.builder()
|
||||
.title(" ")
|
||||
.breadcrumb(null)
|
||||
.content("Plain chunk content.")
|
||||
.build();
|
||||
|
||||
assertEquals("Plain chunk content.", VectorIndexService.buildEmbeddingText(chunk));
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user