diff --git a/.claude/skills/gitnexus/gitnexus-cli/SKILL.md b/.claude/skills/gitnexus/gitnexus-cli/SKILL.md new file mode 100644 index 0000000..cd9a83b --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-cli/SKILL.md @@ -0,0 +1,83 @@ +--- +name: gitnexus-cli +description: "Use when the user needs to run GitNexus CLI commands like analyze/index a repo, check status, clean the index, generate a wiki, or list indexed repos. Examples: \"Index this repo\", \"Reanalyze the codebase\", \"Generate a wiki\"" +--- + +# GitNexus CLI Commands + +All commands work via `npx` — no global install required. + +## Commands + +### analyze — Build or refresh the index + +```bash +npx gitnexus analyze +``` + +Run from the project root. This parses all source files, builds the knowledge graph, writes it to `.gitnexus/`, and generates CLAUDE.md / AGENTS.md context files. + +| Flag | Effect | +| -------------- | ---------------------------------------------------------------- | +| `--force` | Force full re-index even if up to date | +| `--embeddings` | Enable embedding generation for semantic search (off by default) | +| `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. | + +**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook detects staleness after `git commit` and `git merge` and notifies the agent to run `analyze` — the hook does not run analyze itself, to avoid blocking the agent for up to 120s and risking KuzuDB corruption on timeout. + +### status — Check index freshness + +```bash +npx gitnexus status +``` + +Shows whether the current repo has a GitNexus index, when it was last updated, and symbol/relationship counts. Use this to check if re-indexing is needed. + +### clean — Delete the index + +```bash +npx gitnexus clean +``` + +Deletes the `.gitnexus/` directory and unregisters the repo from the global registry. Use before re-indexing if the index is corrupt or after removing GitNexus from a project. + +| Flag | Effect | +| --------- | ------------------------------------------------- | +| `--force` | Skip confirmation prompt | +| `--all` | Clean all indexed repos, not just the current one | + +### wiki — Generate documentation from the graph + +```bash +npx gitnexus wiki +``` + +Generates repository documentation from the knowledge graph using an LLM. Requires an API key (saved to `~/.gitnexus/config.json` on first use). + +| Flag | Effect | +| ------------------- | ----------------------------------------- | +| `--force` | Force full regeneration | +| `--model ` | LLM model (default: minimax/minimax-m2.5) | +| `--base-url ` | LLM API base URL | +| `--api-key ` | LLM API key | +| `--concurrency ` | Parallel LLM calls (default: 3) | +| `--gist` | Publish wiki as a public GitHub Gist | + +### list — Show all indexed repos + +```bash +npx gitnexus list +``` + +Lists all repositories registered in `~/.gitnexus/registry.json`. The MCP `list_repos` tool provides the same information. + +## After Indexing + +1. **Read `gitnexus://repo/{name}/context`** to verify the index loaded +2. Use the other GitNexus skills (`exploring`, `debugging`, `impact-analysis`, `refactoring`) for your task + +## Troubleshooting + +- **"Not inside a git repository"**: Run from a directory inside a git repo +- **Index is stale after re-analyzing**: Restart Claude Code to reload the MCP server +- **Embeddings slow**: Omit `--embeddings` (it's off by default) or set `OPENAI_API_KEY` for faster API-based embedding diff --git a/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md b/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md new file mode 100644 index 0000000..9510b97 --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md @@ -0,0 +1,89 @@ +--- +name: gitnexus-debugging +description: "Use when the user is debugging a bug, tracing an error, or asking why something fails. Examples: \"Why is X failing?\", \"Where does this error come from?\", \"Trace this bug\"" +--- + +# Debugging with GitNexus + +## When to Use + +- "Why is this function failing?" +- "Trace where this error comes from" +- "Who calls this method?" +- "This endpoint returns 500" +- Investigating bugs, errors, or unexpected behavior + +## Workflow + +``` +1. gitnexus_query({query: ""}) → Find related execution flows +2. gitnexus_context({name: ""}) → See callers/callees/processes +3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow +4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed +``` + +> If "Index is stale" → run `npx gitnexus analyze` in terminal. + +## Checklist + +``` +- [ ] Understand the symptom (error message, unexpected behavior) +- [ ] gitnexus_query for error text or related code +- [ ] Identify the suspect function from returned processes +- [ ] gitnexus_context to see callers and callees +- [ ] Trace execution flow via process resource if applicable +- [ ] gitnexus_cypher for custom call chain traces if needed +- [ ] Read source files to confirm root cause +``` + +## Debugging Patterns + +| Symptom | GitNexus Approach | +| -------------------- | ---------------------------------------------------------- | +| Error message | `gitnexus_query` for error text → `context` on throw sites | +| Wrong return value | `context` on the function → trace callees for data flow | +| Intermittent failure | `context` → look for external calls, async deps | +| Performance issue | `context` → find symbols with many callers (hot paths) | +| Recent regression | `detect_changes` to see what your changes affect | + +## Tools + +**gitnexus_query** — find code related to error: + +``` +gitnexus_query({query: "payment validation error"}) +→ Processes: CheckoutFlow, ErrorHandling +→ Symbols: validatePayment, handlePaymentError, PaymentException +``` + +**gitnexus_context** — full context for a suspect: + +``` +gitnexus_context({name: "validatePayment"}) +→ Incoming calls: processCheckout, webhookHandler +→ Outgoing calls: verifyCard, fetchRates (external API!) +→ Processes: CheckoutFlow (step 3/7) +``` + +**gitnexus_cypher** — custom call chain traces: + +```cypher +MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"}) +RETURN [n IN nodes(path) | n.name] AS chain +``` + +## Example: "Payment endpoint returns 500 intermittently" + +``` +1. gitnexus_query({query: "payment error handling"}) + → Processes: CheckoutFlow, ErrorHandling + → Symbols: validatePayment, handlePaymentError + +2. gitnexus_context({name: "validatePayment"}) + → Outgoing calls: verifyCard, fetchRates (external API!) + +3. READ gitnexus://repo/my-app/process/CheckoutFlow + → Step 3: validatePayment → calls fetchRates (external) + +4. Root cause: fetchRates calls external API without proper timeout +``` diff --git a/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md b/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md new file mode 100644 index 0000000..927a4e4 --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md @@ -0,0 +1,78 @@ +--- +name: gitnexus-exploring +description: "Use when the user asks how code works, wants to understand architecture, trace execution flows, or explore unfamiliar parts of the codebase. Examples: \"How does X work?\", \"What calls this function?\", \"Show me the auth flow\"" +--- + +# Exploring Codebases with GitNexus + +## When to Use + +- "How does authentication work?" +- "What's the project structure?" +- "Show me the main components" +- "Where is the database logic?" +- Understanding code you haven't seen before + +## Workflow + +``` +1. READ gitnexus://repos → Discover indexed repos +2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness +3. gitnexus_query({query: ""}) → Find related execution flows +4. gitnexus_context({name: ""}) → Deep dive on specific symbol +5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow +``` + +> If step 2 says "Index is stale" → run `npx gitnexus analyze` in terminal. + +## Checklist + +``` +- [ ] READ gitnexus://repo/{name}/context +- [ ] gitnexus_query for the concept you want to understand +- [ ] Review returned processes (execution flows) +- [ ] gitnexus_context on key symbols for callers/callees +- [ ] READ process resource for full execution traces +- [ ] Read source files for implementation details +``` + +## Resources + +| Resource | What you get | +| --------------------------------------- | ------------------------------------------------------- | +| `gitnexus://repo/{name}/context` | Stats, staleness warning (~150 tokens) | +| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores (~300 tokens) | +| `gitnexus://repo/{name}/cluster/{name}` | Area members with file paths (~500 tokens) | +| `gitnexus://repo/{name}/process/{name}` | Step-by-step execution trace (~200 tokens) | + +## Tools + +**gitnexus_query** — find execution flows related to a concept: + +``` +gitnexus_query({query: "payment processing"}) +→ Processes: CheckoutFlow, RefundFlow, WebhookHandler +→ Symbols grouped by flow with file locations +``` + +**gitnexus_context** — 360-degree view of a symbol: + +``` +gitnexus_context({name: "validateUser"}) +→ Incoming calls: loginHandler, apiMiddleware +→ Outgoing calls: checkToken, getUserById +→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3) +``` + +## Example: "How does payment processing work?" + +``` +1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes +2. gitnexus_query({query: "payment processing"}) + → CheckoutFlow: processPayment → validateCard → chargeStripe + → RefundFlow: initiateRefund → calculateRefund → processRefund +3. gitnexus_context({name: "processPayment"}) + → Incoming: checkoutHandler, webhookHandler + → Outgoing: validateCard, chargeStripe, saveTransaction +4. Read src/payments/processor.ts for implementation details +``` diff --git a/.claude/skills/gitnexus/gitnexus-guide/SKILL.md b/.claude/skills/gitnexus/gitnexus-guide/SKILL.md new file mode 100644 index 0000000..937ac73 --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-guide/SKILL.md @@ -0,0 +1,64 @@ +--- +name: gitnexus-guide +description: "Use when the user asks about GitNexus itself — available tools, how to query the knowledge graph, MCP resources, graph schema, or workflow reference. Examples: \"What GitNexus tools are available?\", \"How do I use GitNexus?\"" +--- + +# GitNexus Guide + +Quick reference for all GitNexus MCP tools, resources, and the knowledge graph schema. + +## Always Start Here + +For any task involving code understanding, debugging, impact analysis, or refactoring: + +1. **Read `gitnexus://repo/{name}/context`** — codebase overview + check index freshness +2. **Match your task to a skill below** and **read that skill file** +3. **Follow the skill's workflow and checklist** + +> If step 1 warns the index is stale, run `npx gitnexus analyze` in the terminal first. + +## Skills + +| Task | Skill to read | +| -------------------------------------------- | ------------------- | +| Understand architecture / "How does X work?" | `gitnexus-exploring` | +| Blast radius / "What breaks if I change X?" | `gitnexus-impact-analysis` | +| Trace bugs / "Why is X failing?" | `gitnexus-debugging` | +| Rename / extract / split / refactor | `gitnexus-refactoring` | +| Tools, resources, schema reference | `gitnexus-guide` (this file) | +| Index, status, clean, wiki CLI commands | `gitnexus-cli` | + +## Tools Reference + +| Tool | What it gives you | +| ---------------- | ------------------------------------------------------------------------ | +| `query` | Process-grouped code intelligence — execution flows related to a concept | +| `context` | 360-degree symbol view — categorized refs, processes it participates in | +| `impact` | Symbol blast radius — what breaks at depth 1/2/3 with confidence | +| `detect_changes` | Git-diff impact — what do your current changes affect | +| `rename` | Multi-file coordinated rename with confidence-tagged edits | +| `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | +| `list_repos` | Discover indexed repos | + +## Resources Reference + +Lightweight reads (~100-500 tokens) for navigation: + +| Resource | Content | +| ---------------------------------------------- | ----------------------------------------- | +| `gitnexus://repo/{name}/context` | Stats, staleness check | +| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores | +| `gitnexus://repo/{name}/cluster/{clusterName}` | Area members | +| `gitnexus://repo/{name}/processes` | All execution flows | +| `gitnexus://repo/{name}/process/{processName}` | Step-by-step trace | +| `gitnexus://repo/{name}/schema` | Graph schema for Cypher | + +## Graph Schema + +**Nodes:** File, Function, Class, Interface, Method, Community, Process +**Edges (via CodeRelation.type):** CALLS, IMPORTS, EXTENDS, IMPLEMENTS, DEFINES, MEMBER_OF, STEP_IN_PROCESS + +```cypher +MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "myFunc"}) +RETURN caller.name, caller.filePath +``` diff --git a/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md b/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md new file mode 100644 index 0000000..e19af28 --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md @@ -0,0 +1,97 @@ +--- +name: gitnexus-impact-analysis +description: "Use when the user wants to know what will break if they change something, or needs safety analysis before editing code. Examples: \"Is it safe to change X?\", \"What depends on this?\", \"What will break?\"" +--- + +# Impact Analysis with GitNexus + +## When to Use + +- "Is it safe to change this function?" +- "What will break if I modify X?" +- "Show me the blast radius" +- "Who uses this code?" +- Before making non-trivial code changes +- Before committing — to understand what your changes affect + +## Workflow + +``` +1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this +2. READ gitnexus://repo/{name}/processes → Check affected execution flows +3. gitnexus_detect_changes() → Map current git changes to affected flows +4. Assess risk and report to user +``` + +> If "Index is stale" → run `npx gitnexus analyze` in terminal. + +## Checklist + +``` +- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents +- [ ] Review d=1 items first (these WILL BREAK) +- [ ] Check high-confidence (>0.8) dependencies +- [ ] READ processes to check affected execution flows +- [ ] gitnexus_detect_changes() for pre-commit check +- [ ] Assess risk level and report to user +``` + +## Understanding Output + +| Depth | Risk Level | Meaning | +| ----- | ---------------- | ------------------------ | +| d=1 | **WILL BREAK** | Direct callers/importers | +| d=2 | LIKELY AFFECTED | Indirect dependencies | +| d=3 | MAY NEED TESTING | Transitive effects | + +## Risk Assessment + +| Affected | Risk | +| ------------------------------ | -------- | +| <5 symbols, few processes | LOW | +| 5-15 symbols, 2-5 processes | MEDIUM | +| >15 symbols or many processes | HIGH | +| Critical path (auth, payments) | CRITICAL | + +## Tools + +**gitnexus_impact** — the primary tool for symbol blast radius: + +``` +gitnexus_impact({ + target: "validateUser", + direction: "upstream", + minConfidence: 0.8, + maxDepth: 3 +}) + +→ d=1 (WILL BREAK): + - loginHandler (src/auth/login.ts:42) [CALLS, 100%] + - apiMiddleware (src/api/middleware.ts:15) [CALLS, 100%] + +→ d=2 (LIKELY AFFECTED): + - authRouter (src/routes/auth.ts:22) [CALLS, 95%] +``` + +**gitnexus_detect_changes** — git-diff based impact analysis: + +``` +gitnexus_detect_changes({scope: "staged"}) + +→ Changed: 5 symbols in 3 files +→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline +→ Risk: MEDIUM +``` + +## Example: "What breaks if I change validateUser?" + +``` +1. gitnexus_impact({target: "validateUser", direction: "upstream"}) + → d=1: loginHandler, apiMiddleware (WILL BREAK) + → d=2: authRouter, sessionManager (LIKELY AFFECTED) + +2. READ gitnexus://repo/my-app/processes + → LoginFlow and TokenRefresh touch validateUser + +3. Risk: 2 direct callers, 2 processes = MEDIUM +``` diff --git a/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md b/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md new file mode 100644 index 0000000..f48cc01 --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md @@ -0,0 +1,121 @@ +--- +name: gitnexus-refactoring +description: "Use when the user wants to rename, extract, split, move, or restructure code safely. Examples: \"Rename this function\", \"Extract this into a module\", \"Refactor this class\", \"Move this to a separate file\"" +--- + +# Refactoring with GitNexus + +## When to Use + +- "Rename this function safely" +- "Extract this into a module" +- "Split this service" +- "Move this to a new file" +- Any task involving renaming, extracting, splitting, or restructuring code + +## Workflow + +``` +1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents +2. gitnexus_query({query: "X"}) → Find execution flows involving X +3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs +4. Plan update order: interfaces → implementations → callers → tests +``` + +> If "Index is stale" → run `npx gitnexus analyze` in terminal. + +## Checklists + +### Rename Symbol + +``` +- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits +- [ ] Review graph edits (high confidence) and ast_search edits (review carefully) +- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits +- [ ] gitnexus_detect_changes() — verify only expected files changed +- [ ] Run tests for affected processes +``` + +### Extract Module + +``` +- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs +- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers +- [ ] Define new module interface +- [ ] Extract code, update imports +- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] Run tests for affected processes +``` + +### Split Function/Service + +``` +- [ ] gitnexus_context({name: target}) — understand all callees +- [ ] Group callees by responsibility +- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update +- [ ] Create new functions/services +- [ ] Update callers +- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] Run tests for affected processes +``` + +## Tools + +**gitnexus_rename** — automated multi-file rename: + +``` +gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +→ 12 edits across 8 files +→ 10 graph edits (high confidence), 2 ast_search edits (review) +→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}] +``` + +**gitnexus_impact** — map all dependents first: + +``` +gitnexus_impact({target: "validateUser", direction: "upstream"}) +→ d=1: loginHandler, apiMiddleware, testUtils +→ Affected Processes: LoginFlow, TokenRefresh +``` + +**gitnexus_detect_changes** — verify your changes after refactoring: + +``` +gitnexus_detect_changes({scope: "all"}) +→ Changed: 8 files, 12 symbols +→ Affected processes: LoginFlow, TokenRefresh +→ Risk: MEDIUM +``` + +**gitnexus_cypher** — custom reference queries: + +```cypher +MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"}) +RETURN caller.name, caller.filePath ORDER BY caller.filePath +``` + +## Risk Rules + +| Risk Factor | Mitigation | +| ------------------- | ----------------------------------------- | +| Many callers (>5) | Use gitnexus_rename for automated updates | +| Cross-area refs | Use detect_changes after to verify scope | +| String/dynamic refs | gitnexus_query to find them | +| External/public API | Version and deprecate properly | + +## Example: Rename `validateUser` to `authenticateUser` + +``` +1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) + → 12 edits: 10 graph (safe), 2 ast_search (review) + → Files: validator.ts, login.ts, middleware.ts, config.json... + +2. Review ast_search edits (config.json: dynamic reference!) + +3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) + → Applied 12 edits across 8 files + +4. gitnexus_detect_changes({scope: "all"}) + → Affected: LoginFlow, TokenRefresh + → Risk: MEDIUM — run tests for these flows +``` diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..21a4fb0 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,43 @@ + +# GitNexus — Code Intelligence + +This project is indexed by GitNexus as **SuperBizAgent-java** (1262 symbols, 2537 relationships, 89 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely. + +> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first. + +## Always Do + +- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. +- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. +- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. +- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`. + +## Never Do + +- NEVER edit a function, class, or method without first running `gitnexus_impact` on it. +- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. +- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph. +- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope. + +## Resources + +| Resource | Use for | +|----------|---------| +| `gitnexus://repo/SuperBizAgent-java/context` | Codebase overview, check index freshness | +| `gitnexus://repo/SuperBizAgent-java/clusters` | All functional areas | +| `gitnexus://repo/SuperBizAgent-java/processes` | All execution flows | +| `gitnexus://repo/SuperBizAgent-java/process/{name}` | Step-by-step execution trace | + +## CLI + +| Task | Read this skill file | +|------|---------------------| +| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` | +| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` | +| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` | +| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` | +| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` | +| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` | + + \ No newline at end of file diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..21a4fb0 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,43 @@ + +# GitNexus — Code Intelligence + +This project is indexed by GitNexus as **SuperBizAgent-java** (1262 symbols, 2537 relationships, 89 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely. + +> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first. + +## Always Do + +- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. +- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. +- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. +- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`. + +## Never Do + +- NEVER edit a function, class, or method without first running `gitnexus_impact` on it. +- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. +- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph. +- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope. + +## Resources + +| Resource | Use for | +|----------|---------| +| `gitnexus://repo/SuperBizAgent-java/context` | Codebase overview, check index freshness | +| `gitnexus://repo/SuperBizAgent-java/clusters` | All functional areas | +| `gitnexus://repo/SuperBizAgent-java/processes` | All execution flows | +| `gitnexus://repo/SuperBizAgent-java/process/{name}` | Step-by-step execution trace | + +## CLI + +| Task | Read this skill file | +|------|---------------------| +| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` | +| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` | +| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` | +| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` | +| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` | +| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` | + + \ No newline at end of file diff --git a/devflow/glossary/CONTEXT.md b/devflow/glossary/CONTEXT.md new file mode 100644 index 0000000..432ad98 --- /dev/null +++ b/devflow/glossary/CONTEXT.md @@ -0,0 +1,29 @@ +# 上下文词汇表 + +## 术语 + +### ChatModel +- 定义:Spring AI 的聊天模型抽象接口,所有 LLM 提供商(DashScope、OpenAI、Ollama 等)都实现此接口 +- 使用场景:所有需要 LLM 推理/生成回答的代码应面向此接口编程 + +### EmbeddingModel +- 定义:Spring AI 的文本向量化抽象接口,将文本转换为向量 +- 使用场景:RAG 流程中将文档文本转为向量存入 Milvus + +### DashScopeChatModel +- 定义:DashScope(阿里云)对 ChatModel 的具体实现 +- 使用场景:当前项目硬编码使用,需要改为通过 ChatModel 接口引用 + +### ReactAgent +- 定义:Spring AI Alibaba Agent Framework 的反应式 Agent 实现 +- 使用场景:Planner-Executor-Replanner 多 Agent 协作 + +### Spring AI Alibaba Agent Framework +- 定义:基于 Spring AI 的多 Agent 协作框架,提供 ReactAgent、PlannerAgent、ExecutorAgent 等 +- 使用场景:项目核心 Agent 逻辑,ReactAgent.builder().model() 接受 ChatModel 接口 + +## 业务规则 + +- ChatModel 是唯一 LLM 调用抽象:替换模型只需更换 Spring Boot starter 和配置 +- EmbeddingModel 是唯一向量化抽象:替换向量模型只需更换 starter 和配置 +- ReactAgent 已兼容 ChatModel 接口,不绑定 DashScope \ No newline at end of file diff --git a/devflow/index.md b/devflow/index.md new file mode 100644 index 0000000..aad51ec --- /dev/null +++ b/devflow/index.md @@ -0,0 +1,7 @@ +# devflow 索引 + +## 项目 + +| 日期 | slug | 领域 | 关键词 | 状态 | +|---|---|---|---|---| +| 2026-05-29 | chatmodel-abstraction | 解耦 | ChatModel, EmbeddingModel, DashScope, Spring AI | 进行中 | \ No newline at end of file diff --git a/devflow/projects/2026-05-29-chatmodel-abstraction/decisions.md b/devflow/projects/2026-05-29-chatmodel-abstraction/decisions.md new file mode 100644 index 0000000..546196d --- /dev/null +++ b/devflow/projects/2026-05-29-chatmodel-abstraction/decisions.md @@ -0,0 +1,42 @@ +# ChatModel Abstraction Decisions + +## Question Pool + +| # | 维度 | 问题 | 模式 | 状态 | +|---|---|---|---|---| +| Q1 | 术语 | ChatModel 注入方式:Spring Boot 自动注入 vs 手动工厂创建 | evidence-driven | 已解决 | +| Q2 | 边界 | RagService 流式对话:Spring AI ChatModel.stream() 替代 DashScope Generation | evidence-driven | 已解决 | +| Q3 | 验收 | VECTOR_DIM 是否需要动态化 | user-interview | 已解决 | +| Q4 | 边界 | VectorEmbeddingService 批量向量化:EmbeddingModel 支持批量调用 | evidence-driven | 已解决 | + +## Evidence-driven + +| 结论 | 证据来源 | 是否已汇报用户 | +|---|---|---| +| ReactAgent.builder().model() 接受 ChatModel 接口 | javap 反编译 | 已汇报 | +| ChatModel 应通过 Spring Boot 自动注入 | DashScope starter 自动注册 ChatModel Bean | 已汇报 | +| RagService 可用 ChatModel.stream() 替代 Generation | Spring AI 接口有 stream(Prompt) 返回 Flux | 已汇报 | +| EmbeddingModel 支持批量调用 | EmbeddingModel.call(EmbeddingRequest) 接受多条文本 | 已汇报 | + +## User-interview + +| 问题原文 | 用户原话 | 确认状态 | OpenSpec 回写 | +|---|---|---|---| +| VECTOR_DIM 怎么处理? | "配置文件动态化" | 已确认 | 已回写 proposal | + +## 关键取舍 + +- 决策:本次只解耦不替换实现 + - 原因:先验证抽象层正确再换模型 + - 影响:代码改动不改变运行行为 + - 风险接受:用户同意先只做解耦 +- 决策:VECTOR_DIM 从配置文件读取 + - 原因:换模型时改 yml 即可 + - 影响:MilvusConstants.VECTOR_DIM 改为从 MilvusProperties 读取 + +## 架构审计 + +- 风险1:RagService 流式适配 — DashScope Generation 和 Spring AI ChatModel.stream() 返回结构不同,需验证 thinking/content 分离逻辑 +- 风险2:DashScopeConfig 通用性 — 硬编码 dashscope 配置键,换模型后需改为通用键 +- 风险3:ChatModel Bean 冲突 — 多 starter 并存时需 @Primary 或条件注解 +- 低风险/无风险:VectorEmbeddingService、MilvusClientFactory 直接替换无问题 \ No newline at end of file diff --git a/docs/analysis/essence-report-rag-chunking.md b/docs/analysis/essence-report-rag-chunking.md new file mode 100644 index 0000000..4fcbc4a --- /dev/null +++ b/docs/analysis/essence-report-rag-chunking.md @@ -0,0 +1,342 @@ +# Essence Report: SuperBizAgent-java — RAG 切片流程 + +> **Lens:** mechanical +> **Design analyzed:** 四层递进式文档分块算法——标题→章节→段落→句子边界的逐级切割策略 +> **Files examined:** 3 (`DocumentChunkService.java`, `DocumentChunkConfig.java`, `DocumentChunk.java`) +> **Pattern:** Hierarchical Splitter with Sentence-Boundary-Aware Overlap +> **Status:** complete + +--- + +## Phase 2: Deep Dive + +### 核心文件 + +| # | 文件 | 行数 | 角色 | +|---|------|------|------| +| 1 | `service/DocumentChunkService.java` | 229 | 分块引擎本身 | +| 2 | `config/DocumentChunkConfig.java` | 32 | 参数契约 `maxSize=800, overlap=100` | +| 3 | `dto/DocumentChunk.java` | 59 | 分块数据载体 | + +### 完整调用链 + +``` +VectorIndexService.indexSingleFile() + └─ chunkService.chunkDocument(content, filePath) [L35] + │ + ├─ splitByHeadings(content) [L44] + │ ├─ 正则: ^(#{1,6})\s+(.+)$ [L65] + │ ├─ 迭代 matcher.find() 找到每个标题位置 + │ ├─ 标题之间的内容 → Section(title, content, startIndex) + │ └─ → List
+ │ + └─ for each Section: + └─ chunkSection(section, globalChunkIndex) [L49] + │ + ├─ if content.length() ≤ maxSize (800): + │ └─ 直接作为一个分块 [L110-119] + │ + ├─ else (需要进一步切割): + │ ├─ splitByParagraphs(content) [L124] + │ │ └─ content.split("\n\n+") [L178] + │ │ + │ ├─ for each paragraph: [L130-167] + │ │ ├─ 当前缓冲区 + 新段落 ≤ maxSize? → 继续追加 + │ │ └─ 当前缓冲区 + 新段落 > maxSize? → 触发切分: + │ │ ├─ 保存当前分块 + │ │ ├─ getOverlapText(当前分块内容) [L147] + │ │ │ ├─ 取末尾 overlap(100) 字符 + │ │ │ ├─ 在重叠文本中找最后一个句子终止符 + │ │ │ │ max(lastIndexOf('。'), lastIndexOf('?'), lastIndexOf('!')) + │ │ │ ├─ if 句子边界 > overlapSize/2 (50字符): + │ │ │ │ └─ 从句子边界后截取(保证新块以完整句开头) + │ │ │ └─ else: + │ │ │ └─ 直接用 overlap 末尾截取 + │ │ └─ 新缓冲区 = 重叠文本 + 当前段落 + │ │ + │ └─ 最后一个分块: 保存缓冲区剩余内容 + │ + └─ → List +``` + +### 算法的四层递进结构 + +``` +第1层:标题分割 + 输入:"# CPU高负载\n内容...\n## 排查步骤\n内容..." + 输出:Section("CPU高负载", "内容..."), Section("排查步骤", "内容...") + 作用:保持文档结构,同一主题的内容不被拆散 + +第2层:容量判断 + if section.length() ≤ 800: 整个章节 = 一个分块 + else: 进入段落级切割 + 作用:短章节保持完整,不破坏语义 + +第3层:段落边界切割 + 输入:超长章节的全部段落 + 算法:逐个追加段落到缓冲区,超过 maxSize 时触发一次切分 + 作用:不在段落中间截断 + +第4层:重叠窗口 + 句子边界对齐 + 输入:即将被切断的分块末尾 + 算法:取末尾100字符 → 找最近的。?! → 从该位置之后截取作为下一块的"种子" + 作用:相邻分块在语义上是"连续"的,检索时召回更完整 +``` + +### 架构图 + +```mermaid +flowchart TD + DOC[/"原始文档"/] --> L1{"第1层: splitByHeadings()"} + + L1 --> S1["Section 1
title: CPU高负载
content: ..."] + L1 --> S2["Section 2
title: 排查步骤
content: ..."] + L1 --> S3["Section N"] + + S1 --> L2{"第2层: 容量判断"} + S2 --> L2 + S3 --> L2 + + L2 -->|"≤800字符"| CHUNK["作为1个分块
继承 title"] + L2 -->|">800字符"| L3{"第3层: splitByParagraphs()
在段落边界切分"} + + L3 --> BUF["逐段追加到缓冲区"] + BUF --> CHECK{"buf + para
> maxSize?"} + CHECK -->|否| APPEND["追加段落
继续累积"] + CHECK -->|是| L4{"第4层: getOverlapText()
句子边界校准"} + + APPEND --> CHECK + + L4 --> FIND["在重叠区末尾100字符
找最近的 。?!"] + FIND --> EVAL{"句子边界位置
> overlapSize/2?"} + EVAL -->|是| ALIGN["从句号后截取
保证新块以完整句开头"] + EVAL -->|否| RAW["退回原始截取
直接用末尾100字符"] + + ALIGN --> SEED["种子 + 当前段落
→ 新缓冲区"] + RAW --> SEED + SEED --> CHECK + + CHUNK --> RESULT[/"List<DocumentChunk>
每个携带: content + title + startIndex + endIndex + chunkIndex"/] +``` + +### 关键代码证据 + +#### 第1层——标题正则 + +```java +// DocumentChunkService.java:65 +Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE); +``` + +支持 H1-H6,`MULTILINE` 模式让 `^` 匹配行首而非仅字符串首。 + +#### 第2层——容量判断(短路) + +```java +// DocumentChunkService.java:110-119 +if (content.length() <= chunkConfig.getMaxSize()) { + DocumentChunk chunk = new DocumentChunk(content, startIndex, endIndex, chunkIndex); + chunk.setTitle(title); + chunks.add(chunk); + return chunks; // 直接返回,不进入段落切割 +} +``` + +#### 第3层——段落级触发切分 + +```java +// DocumentChunkService.java:132-148 +if (currentChunk.length() > 0 && + currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) { + // 触发切分:保存当前块 + String overlap = getOverlapText(chunkContent); // 提取重叠文本 + currentChunk = new StringBuilder(overlap); // 新块以重叠文本开头 + currentStartIndex = currentStartIndex + chunkContent.length() - overlap.length(); +} +currentChunk.append(paragraph).append("\n\n"); // 继续追加 +``` + +#### 第4层——句子边界检测(核心巧思) + +```java +// DocumentChunkService.java:193-213 +private String getOverlapText(String text) { + int overlapSize = Math.min(chunkConfig.getOverlap(), text.length()); + String overlap = text.substring(text.length() - overlapSize); + + // 在重叠文本中找最近的句子终止符 + int lastSentenceEnd = Math.max( + overlap.lastIndexOf('。'), + Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!')) + ); + + // 质量阈值:只有句子边界在重叠区后半段才采用 + if (lastSentenceEnd > overlapSize / 2) { + return overlap.substring(lastSentenceEnd + 1).trim(); + } + return overlap.trim(); // 退回普通重叠 +} +``` + +`overlapSize / 2` 条件是一个**质量阈值**。如果最近的句子边界在重叠区的前半段(即离截断点太远),说明分块点本身就接近句子边界,不需要特殊处理。只有句子边界明显位于重叠区后半段时才调整——避免把半个句子作为新块的"种子"。 + +### 数据流契约 + +``` +chunkDocument(content, filePath) + │ + │ IN: String content — 原始文档全文 + │ String filePath — 仅用于日志 + │ + │ INNER CLASS: Section + │ String title — 所在标题(可为 null) + │ String content — 标题下的所有文本 + │ int startIndex — 在原文档中的字符偏移 + │ + │ OUT: List + │ String content — 分块文本 + │ int startIndex — 在原文档中的起始位置 + │ int endIndex — 在原文档中的结束位置 + │ int chunkIndex — 分块序号 (0, 1, 2, ...) + │ String title — 所属章节标题(继承自 Section) + │ + └─ 消费者: VectorIndexService.indexSingleFile():142 + → 遍历 chunks → embeddingService.generateEmbedding(chunk.content) +``` + +--- + +## Phase 3: Extract Pattern + +### 模式名:Hierarchical Splitter with Sentence-Boundary-Aware Overlap + +**一句话:** 从粗到细逐级切割——先按文档结构(标题)分章,再按语义边界(段落)分块,最后在切分点用句子终止符校准重叠窗口。 + +### 问题 + +固定长度切割的典型失败场景: + +``` +切在句子中间: "CPU使用率达到 95%,建议" | "立即重启相关服务" + ↑ 检索"CPU问题"时召回这块——后半句完全脱离上下文,LLM 误判 + +切在段落中间:"## 排查步骤\n1. 查看监控\n2. 检" | "查日志\n3. 重启服务" + ↑ 步骤 2 被切断,Agent 拿着残缺的排查步骤执行操作 +``` + +### 替代方案对比 + +| 方案 | 切分依据 | 优势 | 劣势 | +|------|----------|------|------| +| **固定字符切割**(最简陋) | maxSize,不关心内容 | 实现简单 | 句子截断、丢失语义 | +| **递归字符切割**(LangChain RecursiveTextSplitter) | `\n\n` → `\n` → ` ` → `` | 通用性好 | 不理解 Markdown 结构 | +| **语义切割**(用 LLM 判断切点) | LLM 标注切分位置 | 理论上最优 | 慢、贵、不可预测 | +| **本项目:层级式+句子校准** | 标题→段落→句子终止符 | 快速 + 保留文档结构 | 仅支持 Markdown,非标题文档退化为段落切割 | + +### 为什么标题分割放在第一步? + +```java +// DocumentChunkService.java:43-44 +// 1. 首先尝试按标题分割(Markdown格式) +List
sections = splitByHeadings(content); +``` + +看本项目的知识库文档就懂了: + +```markdown +# CPU高负载问题排查 ← 一个独立主题 +## 问题现象 +... +## 排查步骤 ← 这些步骤必须完整才能被 Agent 执行 +1. 使用 top 命令确认 CPU 使用率最高的进程 +2. 检查对应服务的日志 +3. ... +## 解决方案 +... + +# 内存高负载问题排查 ← 另一个独立主题 +... +``` + +如果把「CPU 排查步骤」和「内存排查步骤」混在一个分块里,Agent 查询"CPU 高"时会召回包含内存排查步骤的分块——噪声干扰判断。 + +标题优先分割 = **用文档作者自己标注的结构来界定语义边界**,比任何算法都准确。 + +--- + +## Phase 4: Migrate + +### 可迁移性 + +这个切分策略**直接可用**于任何需要为 Markdown 文档建 RAG 的项目。三个参数全部可配置: + +```yaml +# application.yml — 按文档类型调整 +document: + chunk: + max-size: 800 # 短文档(API文档)可设500,长文档(周报)可设1200 + overlap: 100 # 800的12.5%,保持比例即可 +``` + +### Steal-it 示例(17 行) + +```java +/** + * 四层递进分块:标题 → 章节 → 段落 → 句子校准 + * 依赖:maxSize / overlap 两个参数 + */ +public List chunk(String doc) { + List result = new ArrayList<>(); + int globalIdx = 0; + + // 第1层:按标题分章 + for (Section sec : splitByHeadings(doc)) { + if (sec.content.length() <= maxSize) { + // 第2层:短章节直接作为一个分块 + result.add(new Chunk(sec.content, sec.title, globalIdx++)); + } else { + // 第3层:超长章节在段落边界切分 + String overlap = ""; + for (String para : sec.content.split("\n\n+")) { + String candidate = overlap + para; + if (candidate.length() > maxSize && !overlap.isEmpty()) { + result.add(new Chunk(overlap, sec.title, globalIdx++)); + overlap = tailOverlap(overlap); // 第4层:句子校准 + } + overlap = (overlap.isEmpty() ? "" : overlap + "\n\n") + para; + } + if (!overlap.isEmpty()) result.add(new Chunk(overlap, sec.title, globalIdx++)); + } + } + return result; +} +``` + +### 落地陷阱 + +| 陷阱 | 原因 | 规避 | +|------|------|------| +| **非 Markdown 文档退化为单块** | `splitByHeadings()` 找不到标题时整个文档作为一个 Section | L93-96:返回一个 Section,后续段落切割仍生效 | +| **代码块内的 `#` 被误识别为标题** | 正则不区分代码块和正文 | 未规避——可加反引号检测 `` ``` `` | +| **overlap=0 时句子校准无效** | `getOverlapText` 第一行 `Math.min(0, length)=0` 返回空串 | L194:直接返回空字符串,跳过校准 | +| **单段落超过 maxSize 不做切割** | `splitByParagraphs` 后每个段落作为一个单位 | L132 条件要求 `currentChunk.length() > 0`,首段落即使超长也会被单独保存为一块 | + +--- + +### Self-review + +- [x] 设计真实——每层切割均有代码行号证据 +- [x] 深度足够——追溯到正则、条件分支、句子校准的数学逻辑 +- [x] 迁移示例 17 行——提取了四层递进的核心骨架 +- [x] 陷阱具体到代码行——非 Markdown 退化为单块(L93)、单段落超长不切割(L132) + +``` +Essence Report: SuperBizAgent-java — RAG 切片流程 +Lens: mechanical +Design analyzed: 四层递进式文档分块算法 +Files examined: 3 +Pattern: Hierarchical Splitter with Sentence-Boundary-Aware Overlap +Migration: 17-line steal-it skeleton +HTML generated: no +Status: complete +``` diff --git a/docs/analysis/essence-report-rag.md b/docs/analysis/essence-report-rag.md new file mode 100644 index 0000000..5f0b8b0 --- /dev/null +++ b/docs/analysis/essence-report-rag.md @@ -0,0 +1,314 @@ +# Essence Report: SuperBizAgent-java — RAG 实现 + +> **Lens:** mechanical(机械论——结构、接口、数据流) +> **Design analyzed:** RAG 管道——从文档上传到 Agent 辅助检索的完整写入/读取双路径 +> **Files examined:** 12 +> **Pattern:** Pipeline-as-Services + Agent-Mediated Retrieval +> **Status:** complete + +--- + +## Phase 1: 定位 — 设计目标确认 + +来自 `/explore` 报告的「设计二:完整的 RAG 管道(5 级流水线)」。用户指定深入 RAG 实现部分。 + +涉及 12 个核心文件,跨越 controller → service → client → constant 四层。 + +--- + +## Phase 2: Deep Dive — 逐文件追踪 + +### 核心文件清单 + +| # | 文件 | 角色 | 暴露接口 | +|---|------|------|----------| +| 1 | `constant/MilvusConstants.java` | Schema 契约常量 | `VECTOR_DIM=1024`, `COLLECTION_NAME="biz"` | +| 2 | `client/MilvusClientFactory.java` | 数据库初始化 | `createClient()` → 自动建表+建索引 | +| 3 | `config/DocumentChunkConfig.java` | 分块参数 | `maxSize=800`, `overlap=100` | +| 4 | `dto/DocumentChunk.java` | 分块实体 | `content`, `startIndex/endIndex`, `chunkIndex`, `title` | +| 5 | `service/DocumentChunkService.java` | 智能分块器 | `chunkDocument(content, filePath)` → `List` | +| 6 | `service/VectorEmbeddingService.java` | 向量化网关 | `generateEmbedding(text)` → `List` (1024-dim) | +| 7 | `service/VectorIndexService.java` | 写入管道编排 | `indexSingleFile(path)` → 读→删旧→分块→向量化→写 | +| 8 | `service/VectorSearchService.java` | 语义检索 | `searchSimilarDocuments(query, topK)` → `List` | +| 9 | `service/RagService.java` | 全栈 RAG 问答 | `queryStream(question, history, callback)` → SSE流式 | +| 10 | `agent/tool/InternalDocsTools.java` | Agent 工具桥 | `queryInternalDocs(query)` → JSON(仅检索,不生文) | +| 11 | `controller/FileUploadController.java` | 写入入口 | `POST /api/upload` → 文件存储 + 自动索引 | +| 12 | `controller/ChatController.java` | 读取入口 | `POST /api/chat(_stream)` → ReactAgent + 工具调用 | + +### 完整的调用链(双路径) + +#### 写入路径(索引管道) + +``` +POST /api/upload + └─ FileUploadController.upload() [L34] + ├─ Files.copy() → 保存文件到 uploadPath + └─ VectorIndexService.indexSingleFile() [L124] + ├─ Files.readString() [L135] + ├─ deleteExistingData() [L173] + │ └─ milvusClient.delete() [L198] + │ expr: metadata["_source"] == "/path/to/file" + ├─ chunkService.chunkDocument() [L142] + │ ├─ splitByHeadings() [L61] + │ │ └─ 正则: ^(#{1,6})\s+(.+)$ + │ ├─ chunkSection() × N [L104] + │ │ ├─ splitByParagraphs() [L174] + │ │ └─ getOverlapText() [L193] + │ └─ → List + └─ for each chunk: [L146] + ├─ embeddingService.generateEmbedding() [L76] + │ └─ DashScope TextEmbedding API → List[1024] + └─ insertToMilvus() [L255] + └─ UUID(source+chunkIndex) + vector + content + metadata(JSON) +``` + +#### 读取路径(Agent 中介检索) + +``` +POST /api/chat_stream + └─ ChatController.chatStream() [L143] + └─ chatService.createReactAgent() [L183] + └─ tools: [DateTimeTools, InternalDocsTools, QueryMetricsTools, QueryLogsTools] + └─ agent.stream(question) [L189] + └─ Agent 自主决策 → 调用 queryInternalDocs + └─ InternalDocsTools.queryInternalDocs() [L53] + └─ VectorSearchService.searchSimilarDocuments() [L42] + ├─ embeddingService.generateQueryVector() [L47] + ├─ milvusClient.search() [L51] + │ └─ L2距离, IVF_FLAT, nprobe=10 + └─ → List{id, content, score, metadata} + └─ return JSON to Agent + └─ Agent 融合检索结果 + LLM推理 → 最终回答 +``` + +### 架构图 + +```mermaid +graph TB + subgraph 写入路径 + UPLOAD[POST /api/upload] + FC[FileUploadController] + VIS[VectorIndexService] + DCS[DocumentChunkService] + VES[VectorEmbeddingService] + MV_W[(Milvus)] + end + + subgraph 读取路径 + CHAT[POST /api/chat_stream] + CC[ChatController] + AGENT[ReactAgent] + IDT[InternalDocsTools
@Tool注解] + VSS[VectorSearchService] + MV_R[(Milvus)] + LLM[DashScope LLM] + end + + UPLOAD --> FC + FC --> VIS + VIS --> DCS --> VIS + VIS --> VES --> VIS + VIS --> MV_W + + CHAT --> CC + CC --> AGENT + AGENT -->|自主决策调用| IDT + IDT --> VSS + VSS --> VES --> VSS + VSS --> MV_R + IDT -->|JSON结果| AGENT + AGENT --> LLM + LLM -->|SSE流式| CC +``` + +### 关键设计决策(代码证据) + +#### 1. 幂等上传——元数据驱动的去重策略 + +```java +// VectorIndexService.java:138-139 +// 删除该文件的旧数据(如果存在) +deleteExistingData(path.toString()); +``` + +`deleteExistingData()` (L173-215) 使用 `metadata["_source"] == filePath` 作为删除表达式。每次上传同一文件时,先清空旧向量再写入新数据,保证数据一致性。 + +#### 2. 路径标准化——跨平台一致性 + +```java +// VectorIndexService.java:176-178 +Path path = Paths.get(filePath).normalize(); +String normalizedPath = path.toString().replace(File.separator, "/"); +``` + +Windows `\` 和 Unix `/` 统一为正斜杠,避免 Milvus 表达式解析错误。在 `deleteExistingData()` 和 `buildMetadata()` 中均有应用。 + +#### 3. 重叠窗口 + 句子边界感知 + +```java +// DocumentChunkService.java:132-148 +if (currentChunk.length() > 0 && + currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) { + // 保存当前分片 + String overlap = getOverlapText(chunkContent); // 提取重叠文本 + currentChunk = new StringBuilder(overlap); // 新分片以重叠文本开头 +``` + +`getOverlapText()` (L193-213) 更进一步:在重叠文本中寻找句子边界(`。?!`),避免在句子中间截断。当句子边界超过 `overlapSize/2` 时才使用,否则退回原始重叠策略。 + +#### 4. 检索与生成分离 + +`InternalDocsTools.queryInternalDocs()` 只做检索,不做生成。它将搜索结果序列化为 JSON 返回给 Agent,由 Agent 的 LLM 自行判断如何使用这些信息。 + +```java +// InternalDocsTools.java:68 +String resultJson = objectMapper.writeValueAsString(searchResults); +return resultJson; +``` + +对比 `RagService.queryStream()` 则完整执行「检索→构建上下文→LLM 生成」三步,是一个独立的全栈 RAG 备用路径。 + +#### 5. Milvus Schema 设计 + +```java +// MilvusClientFactory.java:109-142 +// 四个字段: +// id VarChar(256) 主键 — UUID(source + chunkIndex) +// vector FloatVector(1024) — text-embedding-v4 输出 +// content VarChar(8192) — 分块后的文本内容 +// metadata JSON — {_source, _extension, _file_name, chunkIndex, totalChunks, title} +// 索引: IVF_FLAT, L2距离, nlist=128 +``` + +--- + +## Phase 3: Extract Pattern + +### 设计模式:Pipeline-as-Services + Agent-Mediated Retrieval + +**问题:** 如何将知识库文档转化为 AI Agent 可检索、可利用的语义记忆? + +**传统方案的问题:** +- 关键词检索:无法理解语义相似的查询 +- 硬编码 FAQ:无法应对未见过的问题 +- 直接向量检索 + 固定提示词:所有问题都触发检索,浪费资源 + +**本项目的方案:两阶段架构** + +``` +┌──────────────────────────────────────────────────┐ +│ STAGE 1: 写入管道 (离线/上传时触发) │ +│ │ +│ 文档 ──→ 智能分块 ──→ 向量化 ──→ Milvus存储 │ +│ (标题+段落 (text-embedding (IVF_FLAT │ +│ 边界感知) -v4, 1024-dim) L2索引) │ +│ │ +│ 接口契约: │ +│ IN: File → OUT: N × (vector + content + meta) │ +└──────────────────────────────────────────────────┘ + +┌──────────────────────────────────────────────────┐ +│ STAGE 2: 读取管道 (Agent 决策时触发) │ +│ │ +│ 用户问题 ──→ Agent 思考 ──→ 决定查知识库 │ +│ │ │ +│ ▼ │ +│ 向量检索 (L2距离) ──→ Top-K 文档片段 │ +│ │ │ +│ ▼ │ +│ Agent 融合检索结果 + LLM推理 → 回答 │ +│ │ +│ 接口契约: │ +│ IN: query(自然语言) → OUT: JSON(检索结果) │ +│ Agent 自主决定: 是否调用 / 如何使用结果 │ +└──────────────────────────────────────────────────┘ +``` + +### 接口契约(隐式——通过 Spring DI 实现) + +| 契约 | 生产者 | 消费者 | 数据形状 | +|------|--------|--------|----------| +| `List` | DocumentChunkService | VectorIndexService | `{content, startIndex, endIndex, chunkIndex, title}` | +| `List[1024]` | VectorEmbeddingService | VectorIndexService, VectorSearchService | DashScope text-embedding-v4 输出 | +| `List` | VectorSearchService | InternalDocsTools, RagService | `{id, content, score, metadata}` | +| `StreamCallback` | RagService | (外部调用者) | `{onSearchResults, onContentChunk, onComplete, onError}` | + +### 替代方案对比 + +| 方案 | 本项目 | LangChain4j | 纯 DashScope API | +|------|--------|--------------|-------------------| +| 分块策略 | 标题感知 + 段落边界 + 句子重叠 | 多种内置 Splitter | 无,需自建 | +| 向量库 | Milvus (IVF_FLAT) | 多后端支持 | 无 | +| Agent 集成 | Spring AI @Tool 注解,Agent 自主决策 | AiServices + @Tool | 无 Agent 框架 | +| 去重 | metadata["_source"] 匹配删除 | 需自定义 | 不适用 | + +### 为什么选择这种设计? + +1. **「检索」和「生成」分离**:`InternalDocsTools` 只返回检索结果,生成由 Agent 的 LLM 完成。Agent 可以选择**不使用**检索结果(如果检索质量不高),或者**交叉验证**多次检索的结果 +2. **工具化 RAG**:将 RAG 暴露为 Agent 工具而非独立 API,让 Agent 在合适的时机触发检索——而非对所有问题都做 RAG +3. **5 个独立 Service**:每个阶段可单独替换。想换分块策略?只改 `DocumentChunkService`。想换向量库?只改 `VectorSearchService` + `VectorIndexService` + +--- + +## Phase 4: Migrate — 可迁移的设计 + +### 可迁移性评估 + +这个 RAG 设计**高度可迁移**到任何需要「知识库 + AI Agent」的 Java 项目。核心依赖是 Spring AI 生态 + 一个向量数据库。 + +### Steal-it 示例(12 行) + +```java +// 核心思想:Pipeline-as-Services + Agent Tool Bridge +// 以下骨架可直接用于任何 Spring Boot 项目 + +// 1. 分块器:语义感知分割 +public List chunk(String doc) { + return splitByHeadings(doc).stream() + .flatMap(s -> splitToFit(s, MAX_SIZE, OVERLAP)) + .toList(); +} + +// 2. Agent 工具桥:检索但不生文 +@Component +class KnowledgeBaseTool { + @Tool(description = "搜索内部知识库获取相关信息") + public String search(@ToolParam(description="查询内容") String query) { + List qv = embedder.embed(query); // 向量化 + var results = vectorDB.search(qv, TOP_K); // 语义检索 + return toJson(results); // 返回给Agent + } +} +``` + +### 落地陷阱 + +| 陷阱 | 说明 | 本项目如何规避 | +|------|------|----------------| +| **路径分隔符不一致** | Windows `\` vs Unix `/` 导致 Milvus 表达式解析失败 | `VectorIndexService.java:177` 强制 `replace(File.separator, "/")` | +| **重复上传污染数据** | 同一文件多次上传产生重复向量 | `VectorIndexService.java:138-139` delete-before-insert | +| **分块边界截断语义** | 固定长度切割可能切断句子 | `DocumentChunkService.java:203-206` 在重叠区找句子边界 | +| **Agent 未触发工具** | Agent 不知道何时该查知识库 | `InternalDocsTools.java:49-52` @Tool description 用英文详细描述触发场景 | +| **向量维度不匹配** | embedding 模型输出维度与 Milvus schema 不一致 | `MilvusConstants.java:18` 集中管理 `VECTOR_DIM=1024` | +| **API Key 未初始化** | 静态 Constants 被其他线程覆盖 | `VectorEmbeddingService.java:86-89` 每次调用前检查并修复 | + +### Self-review + +- [x] 设计真实存在 — 每个声明均有文件+行号证据 +- [x] 分析深度足够 — 完整追踪了写入/读取两条全路径 +- [x] 迁移示例≤20行 — 仅提取 Pipeline + Tool Bridge 骨架 +- [x] 陷阱具体 — 每个都有代码规避证据 +- [x] 可解释为什么优于替代方案 — Agent 自主决策 vs 强制 RAG + +--- + +``` +Essence Report: SuperBizAgent-java +Lens: mechanical +Design analyzed: RAG 管道 — Pipeline-as-Services + Agent-Mediated Retrieval +Files examined: 12 +Pattern: Pipeline-as-Services + Agent Tool Bridge +Migration: 12-line steal-it skeleton +HTML generated: no +Status: complete +``` diff --git a/docs/analysis/explore-report.md b/docs/analysis/explore-report.md new file mode 100644 index 0000000..4b301d8 --- /dev/null +++ b/docs/analysis/explore-report.md @@ -0,0 +1,352 @@ +# Explore Report: SuperBizAgent-java + +> 生成时间: 2026-04-30 +> Project type: **code repository** +> Phases completed: 4/4 +> Diagram included: yes +> Core designs: 3 +> Status: complete + +--- + +## Phase 1: Positioning & Structure + +### 这是什么项目 + +SuperBizAgent-java 是一个基于 **Spring AI + Alibaba DashScope (Qwen)** 的智能运维 AI Agent 平台。它将大语言模型、向量检索增强生成(RAG)和多智能体协作(Planner-Executor-Replanner)整合为一体,面向企业 IT 运维场景提供: + +- **智能文档问答**:上传运维知识库文档(Markdown/TXT),通过 RAG 管道实现向量化检索 + LLM 流式生成回答 +- **告警分析自动化**:多 Agent 协作分析 Prometheus 告警,结合日志查询(腾讯云 CLS)和内部知识库,生成结构化的告警分析报告 +- **MCP 协议集成**:通过 Spring AI MCP Client 连接外部工具服务,扩展 Agent 能力边界 + +### 为什么值得研究 + +| 维度 | 价值 | +|------|------| +| **AI 框架落地** | Spring AI Alibaba 生态的完整实践——ReactAgent、SupervisorAgent、Tool 注册、流式对话 | +| **多 Agent 协作** | 非玩具级的 Planner-Executor-Replanner 监督循环,实际解决告警分析这种开放性问题 | +| **RAG 工程化** | 完整的文档分块→向量化→Milvus 存储→语义检索→流式生成的端到端管道 | +| **MCP 协议** | 业界较早将 MCP (Model Context Protocol) 用于生产场景的 Java 案例 | + +### 适合谁 + +- Spring Boot / Java 开发者学习 AI Agent 框架的落地模式 +- AIOps / SRE 工程师了解智能运维 Agent 的架构设计 +- 对 Spring AI Alibaba 生态感兴趣的技术决策者 + +### 项目规模 + +| 指标 | 数值 | +|------|------| +| Java 源文件 | ~25 个 | +| 代码行数 | ~2500 行 | +| API 端点 | 7 个 | +| Agent 工具 | 4 个 | +| 知识库文档 | 5 篇 | + +### 技术栈 + +``` +应用层 Spring Boot 3.2 / Java 17 +AI 层 Spring AI Alibaba 1.1.0 / Qwen3-Max / text-embedding-v4 +存储层 Milvus 2.5 (向量库) / MinIO (对象存储) +集成层 MCP Client (WebFlux SSE) / Prometheus +部署 Docker Compose (Milvus + etcd + MinIO + Attu) +``` + +### 与替代方案的对比 + +| 方案 | 优势 | 劣势 | +|------|------|------| +| 本项目 (Spring AI Alibaba) | 完整生态、国产模型、Java 原生 | 社区相对年轻 | +| LangChain4j | 社区活跃、模型支持广 | 多 Agent 模式需自行构建 | +| Python LangChain | 生态最丰富 | 非 Java 技术栈 | +| 纯 DashScope API | 简单直接 | 缺乏 Agent 编排、工具调用框架 | + +--- + +## Phase 2: Flow + +### 架构总览 + +```mermaid +graph TB + subgraph 前端 + WEB[Web UI
index.html + app.js] + end + + subgraph 控制层 + CC[ChatController
/api/chat /api/chat_stream] + AO[AIOpsController
/api/ai_ops] + UP[FileUploadController
/api/upload] + HC[MilvusCheckController
/milvus/health] + end + + subgraph 服务层 + CS[ChatService
ReactAgent 编排] + AIS[AiOpsService
多Agent 协作] + RS[RagService
RAG 流式问答] + VIS[VectorIndexService
文件索引管道] + VSS[VectorSearchService
向量相似搜索] + VES[VectorEmbeddingService
文本向量化] + DCS[DocumentChunkService
智能文档分块] + end + + subgraph Agent工具 + DT[DateTimeTools] + IDT[InternalDocsTools] + QMT[QueryMetricsTools] + QLT[QueryLogsTools] + end + + subgraph 外部服务 + DS[DashScope API
Qwen3-Max / Embedding] + MV[Milvus
向量数据库] + PM[Prometheus
监控告警] + CLS[腾讯云CLS
MCP SSE] + end + + WEB --> CC + WEB --> AO + WEB --> UP + WEB --> HC + + CC --> CS + CC --> RS + AO --> AIS + UP --> VIS + + CS --> DT & IDT & QMT & QLT + AIS --> DT & IDT & QMT & QLT + + CS --> DS + RS --> DS + RS --> VSS + VIS --> DCS --> VES --> MV + VSS --> MV + QMT --> PM + QLT --> CLS + + VES --> DS +``` + +### 主要运行时流程 + +#### 流程 A:RAG 智能问答(文档→检索→生成) + +```mermaid +sequenceDiagram + actor User + participant Ctrl as FileUploadController + participant VIS as VectorIndexService + participant DCS as DocumentChunkService + participant VES as VectorEmbeddingService + participant MV as Milvus + participant RS as RagService + participant DS as DashScope + + Note over User,DS: === 索引阶段 === + User->>Ctrl: POST /api/upload (file.md) + Ctrl->>VIS: indexSingleFile(file) + VIS->>VIS: 删除旧向量(按source路径匹配) + VIS->>DCS: chunkDocument(content) + DCS-->>VIS: List + loop 每个分块 + VIS->>VES: generateEmbedding(chunk) + VES->>DS: text-embedding-v4 API + DS-->>VES: float[1024] + VES-->>VIS: 向量 + end + VIS->>MV: insert(向量 + 原文 + metadata) + MV-->>VIS: OK + + Note over User,DS: === 问答阶段 === + User->>Ctrl: POST /api/chat (question) + Ctrl->>RS: generateAnswerStream(question) + RS->>VES: 向量化问题 + VES->>DS: text-embedding-v4 + DS-->>RS: query_vector[1024] + RS->>MV: search(query_vector, topK=3) + MV-->>RS: 3条最相似文档片段 + RS->>DS: Generation API (提示词 + 上下文 + 问题) + DS-->>User: SSE 流式生成回答 +``` + +#### 流程 B:AIOps 多 Agent 告警分析 + +```mermaid +sequenceDiagram + actor User + participant Ctrl as ChatController + participant AIS as AiOpsService + participant Sup as SupervisorAgent + participant P as PlannerAgent + participant E as ExecutorAgent + participant Tools as Agent Tools + participant DS as DashScope + + User->>Ctrl: POST /api/ai_ops (告警信息) + Ctrl->>AIS: executeAiOpsAnalysis(request) + + Note over AIS, DS: 启动监督循环 + AIS->>Sup: 启动,传入 Planner + Executor + + loop Planner-Executor-Replanner + Sup->>P: 分析当前状态,决定下一步 + alt 需要制定/修订计划 + P-->>User: SSE: 📋 分析计划... + else 需要执行步骤 + P-->>Sup: EXECUTE + Sup->>E: 执行计划第一步 + E->>Tools: 调用工具收集证据 + Tools-->>E: 日志/告警/文档信息 + E-->>User: SSE: 🔍 执行结果... + E-->>Sup: 反馈 + 证据 + Note over Sup: 将执行结果反馈给Planner + else 分析完成 + P-->>Sup: FINISH + end + end + + Sup-->>User: SSE: ✅ Markdown 告警分析报告 +``` + +--- + +## Phase 3: Start Path + +### 最小启动步骤 + +```bash +# 1. 启动基础设施(Milvus + etcd + MinIO) +cd D:\zhu\project\SuperBizAgent-java +docker compose -f vector-database.yml up -d + +# 2. 设置 API Key 环境变量 +export DASHSCOPE_API_KEY="your-dashscope-api-key" + +# 3. 启动应用 +mvn spring-boot:run +# 应用启动在 http://localhost:9900 + +# 4. 打开 Web 测试页面 +# http://localhost:9900/index.html +``` + +### 学习起点 + +1. **第一入口**:`src/main/java/org/example/Main.java` — Spring Boot 启动类,了解组件扫描范围 +2. **核心对话**:`src/main/java/org/example/controller/ChatController.java` — 所有 API 端点定义,理解请求路由 +3. **Agent 编排**:`src/main/java/org/example/service/ChatService.java` — ReactAgent 如何注册工具、处理对话 +4. **多 Agent 协作**:`src/main/java/org/example/service/AiOpsService.java` — Planner-Executor-Replanner 模式完整实现 +5. **RAG 管道**:按 `VectorIndexService → DocumentChunkService → VectorEmbeddingService → RagService` 顺序阅读 + +### 建议的第一个修改 + +在 `QueryMetricsTools.java` 的 `queryPrometheusAlerts()` 方法中添加一个 Mock 数据,观察 Agent 如何将新的工具输出整合到对话中。修改后重新提问相关问题即可看到效果。 + +--- + +## Phase 4: Core Designs + +### 设计一:Planner-Executor-Replanner 监督循环 + +**位置**:`src/main/java/org/example/service/AiOpsService.java` + +**是什么**:一个三层多 Agent 协作模式,用监督者控制循环来解决开放性的告警分析问题。 + +``` +SupervisorAgent (监督者) + ├── PlannerAgent (规划者) + │ └── 决策三个状态: PLAN → 制定/修订计划 + │ EXECUTE → 交给执行者 + │ FINISH → 输出最终报告 + └── ExecutorAgent (执行者) + └── 执行计划中的第一步 + └── 调用工具获取真实数据 + └── 返回反馈给 Planner 重新规划 +``` + +**为什么重要**: +- 不是简单的单次 Agent 调用,而是通过**循环反馈**逐步逼近准确分析 +- Planner 根据 Executor 返回的证据**动态调整计划**(即 Replan 机制) +- 通过 SSE 将每一步的中间结果实时推送给前端,用户体验好 +- 工具调用是**实际的**:Prometheus 查询、日志搜索、知识库检索,不是 mock 玩具 + +**关键实现细节**: +```java +// SupervisorAgent 创建并传入子 Agent +SupervisorAgent supervisor = SupervisorAgent.builder() + .supervisorAgent(supervisor) + .subAgents(plannerAgent, executorAgent) + .build(); +``` + +### 设计二:完整的 RAG 管道(5 级流水线) + +**位置**:`VectorIndexService` → `DocumentChunkService` → `VectorEmbeddingService` → `VectorSearchService` → `RagService` + +**是什么**:从原始文档到流式问答输出的完整 RAG 管道,涉及 5 个松耦合的服务组件。 + +| 阶段 | 组件 | 关键技术点 | +|------|------|------------| +| 1. 智能分块 | DocumentChunkService | 按 Markdown 标题层级 + 段落边界分割,800 字符/块,100 字符重叠 | +| 2. 向量化 | VectorEmbeddingService | DashScope text-embedding-v4,1024 维,支持批量 | +| 3. 向量存储 | MilvusClientFactory | IVF_FLAT 索引,L2 距离,自动去重(按 source 路径) | +| 4. 语义检索 | VectorSearchService | Top-K 配置化(default 3),返回原文 + 相似度分数 | +| 5. 流式生成 | RagService | DashScope Generation API,SSE 流式输出,支持 system prompt | + +**为什么重要**: +- 每个阶段**独立可替换**——可以换分块策略、换向量库、换 LLM +- **幂等上传**:同一文件重新上传时,先删除旧向量再写入,保证数据一致性 +- 分块策略考虑了 Markdown 的文档结构(标题层级),而不是简单的固定长度切割 + +### 设计三:工具即插即用的 Agent 工具系统 + +**位置**:`src/main/java/org/example/agent/tool/*.java` + +**是什么**:基于 Spring AI `@Tool` 注解的工具系统,Agent 自动发现并可调用。 + +```java +// 工具定义示例 +@Component +public class DateTimeTools { + @Tool(description = "获取当前日期和时间") + public String getCurrentDateTime() { ... } +} +``` + +**核心设计决策**: + +| 决策 | 做法 | 原因 | +|------|------|------| +| Mock 开关 | `QueryMetricsTools` 和 `QueryLogsTools` 都有 `mockEnabled` 配置 | 开发/演示时不需要真实 Prometheus/CLS 环境 | +| MCP 优先 | 当 MCP Client 可用时,自动排除 `QueryLogsTools` | 避免工具重复,MCP 提供更丰富的日志能力 | +| JSON Schema 生成 | 使用 `jsonschema-generator` 为工具参数生成 schema | 让 LLM 理解工具的参数类型和约束 | +| 工具注册 | `ChatService` 和 `AiOpsService` 各自注册工具集 | Agent 只获得需要的能力,避免干扰 | + +**ChatService 工具注册**: +```java +// 构建时注册所有可用工具 +ReactAgent agent = ReactAgent.builder() + .tools(dateTimeTools, internalDocsTools, + queryMetricsTools, queryLogsTools) + .build(); +``` + +**为什么重要**: +- Agent 工具系统是 AI Agent 的**能力边界**——定义了 Agent 能做什么 +- Mock/Real 模式切换体现了**开发友好性** +- MCP 协议的集成展示了**可扩展性**——Agent 可以从外部获取新能力 + +--- + +## 总结 + +SuperBizAgent-java 是一个小而完整的 AI Agent 实践项目。它的三个核心竞争力是: + +1. **多 Agent 协作**(Planner-Executor-Replanner)——不是玩具,是真正解决问题的模式 +2. **工程化的 RAG 管道**——5 级流水线、幂等上传、智能分块 +3. **Spring AI 生态的完整实践**——从 @Tool 注解到 MCP 协议,展示了 Java 生态做 AI Agent 的成熟路径 + +对于想将 AI Agent 引入企业运维场景的 Java 团队,这是一个很好的学习起点和脚手架。 diff --git a/docs/design/chunking-issues-analysis.md b/docs/design/chunking-issues-analysis.md new file mode 100644 index 0000000..1545716 --- /dev/null +++ b/docs/design/chunking-issues-analysis.md @@ -0,0 +1,282 @@ +# 当前分片策略问题分析与根因 + +> 基于 `DocumentChunkServiceTest` 可视化测试的运行结果 +> 配置:`maxSize=800, overlap=100`(默认) / 可视化测试使用 `maxSize=300/200, overlap=50/30` + +--- + +## 问题总览 + +| # | 问题 | 严重程度 | 根因归类 | +|---|------|----------|----------| +| 1 | 标题独立成空壳块 | 中 | 标题分割逻辑 | +| 2 | 有序列表被拆散 | 高 | 段落级切割 + 缺少结构感知 | +| 3 | 英文块 token 密度远低于中文块 | 高 | 字符计数代替 token 计数 | +| 4 | 硬截断点在语义转折处无特殊处理 | 中 | 仅依赖 maxSize 触发 | +| 5 | overlap 窗口对中文句号后截取命中率低 | 低 | 句子校准逻辑覆盖不全 | + +--- + +## 问题 1:标题独立成空壳块 + +### 现象 + +运维文档 `maxSize=300` 下,H1 标题产生了一个只有 14 字符的分块: + +``` +Chunk #0 +│ Title: CPU高负载问题排查指南 +│ Range: [0→14] (14字符) +│ Content: +│ │ # CPU高负载问题排查指南 +``` + +紧随其后的 `## 问题现象` 被分到下一个块。14 字符的块没有任何可检索的实质内容。 + +### 根因 + +```java +// DocumentChunkService.java:71-83 +while (matcher.find()) { + // 保存上一个章节 + if (lastEnd < matcher.start()) { + String sectionContent = content.substring(lastEnd, matcher.start()).trim(); + if (!sectionContent.isEmpty()) { // ← 条件:content 非空 + sections.add(new Section(currentTitle, sectionContent, lastEnd)); + } + } + currentTitle = matcher.group(2).trim(); + lastEnd = matcher.start(); +} +``` + +`splitByHeadings()` 遍历标题时,`lastEnd` 指向当前标题起始位置,`matcher.start()` 是下一个标题的起始位置。当 H1 后紧跟 H2(中间只有 `#` 行本身的内容),`content.substring(lastEnd, matcher.start())` 取出的是 **H1 标题行本身 + H1 标题行和 H2 之间的空白**。 + +关键问题: +- H1 标题行被当作上一个 section 的 "content" 保存(因为中间文本不为空——标题行本身是文本) +- 但实质上标题不应该独立成为一个可检索的分块 + +### 影响 + +- 向量库中出现大量无效向量(仅含标题、无实质内容) +- 检索时可能召回标题块,Agent 得不到有用信息 +- 浪费 Milvus 存储空间 + +--- + +## 问题 2:有序列表被拆散 + +### 现象 + +排查步骤 1-4 在 Chunk #2,第 5 步被单独踢到 Chunk #3: + +``` +Chunk #2 → 1. 登录服务器... 2. 使用 ps... 3. 查看应用日志... 4. 检查数据库... +Chunk #3 → 5. 检查JVM内存... +``` + +Agent 调用工具拿到 Chunk #2 时,排查步骤不完整,可能漏掉关键操作。 + +### 根因 + +```java +// DocumentChunkService.java:174-188 +private List splitByParagraphs(String content) { + List paragraphs = new ArrayList<>(); + String[] parts = content.split("\n\n+"); // ← 双换行分割 + for (String part : parts) { + String trimmed = part.trim(); + if (!trimmed.isEmpty()) { + paragraphs.add(trimmed); + } + } + return paragraphs; +} +``` + +```java +// DocumentChunkService.java:132-148 +if (currentChunk.length() > 0 && + currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) { + // 触发切分——不关心这个段落属于什么语义结构 + String overlap = getOverlapText(chunkContent); + currentChunk = new StringBuilder(overlap); +} +currentChunk.append(paragraph).append("\n\n"); +``` + +两层根因: +1. `splitByParagraphs()` 只认 `\n\n+` 作为段落分割符,不识别 **有序列表**(`1. \n2. \n3.` 之间通常是单换行) +2. `chunkSection()` 走到字符上限就切,完全不感知"这是一个列表的第几项"——列表项之间的语义强关联被忽略 + +### 影响 + +- 排查步骤、操作指南类文档的完整性被破坏 +- RAG 检索召回不完整的步骤列表,Agent 据此操作可能导致遗漏 +- 这是运维场景的致命问题——运维文档大量使用列表 + +--- + +## 问题 3:英文块 token 密度远低于中文块 + +### 现象 + +可视化测试数据: + +``` +中文: 218字符 → 2个分块(约218 tokens,密度 ~1.0 token/字符) +英文: 602字符 → 3个分块(约150 tokens,密度 ~0.25 token/字符) +``` + +同样 `maxSize=200`,英文 602 字符装了 150 token 还产生 3 个分块;中文 218 字符装了 218 token 只产生 2 个分块。中文块的实际 token 负担是英文的 **~4x**。 + +### 根因 + +```java +// DocumentChunkConfig.java:18 +private int maxSize = 800; // 字符数上限 + +// DocumentChunkService.java:132-133 +if (currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) { + // 这里比的是 Java String.length() — 字符数,不是 token 数 +``` + +Java 的 `String.length()` 对每个 Unicode 字符(包括中文)都返回 1。但 LLM tokenizer 对中文和英文的 token 化效率完全不同: + +``` +"这是中文" → 4 字符 → ~4 tokens (1:1) +"This is English" → 15 字符 → ~4 tokens (3.75:1) +``` + +用字符数作为切割上限,相当于: +- 中文块:可以装 800 token(甚至更多) +- 英文块:只能装 ~200 token + +LLM 上下文窗口是按 token 计费的,这种偏差意味着**中文知识库的 RAG 开销是英文的 4 倍**。 + +### 影响 + +- LLM 调用成本不可预测(中英混排时波动大) +- 中文知识库的上下文窗口利用率极易超标 +- 无法对 prompt 的 token 预算做精确控制 + +--- + +## 问题 4:硬截断在语义转折处无特殊处理 + +### 现象 + +同问题 2 的根因延伸。当前逻辑: + +``` +段落1 + 段落2 + 段落3 + ... + 段落N → 总字符数 < maxSize → 继续追加 + → 总字符数 > maxSize → 立刻切 +``` + +不考虑「段落 N 和段落 N+1 是否属于同一语义单元」。两个语义上需要绑定的段落恰好越过 maxSize 边界就会被拆散。 + +### 根因 + +```java +// DocumentChunkService.java:132 +if (currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) { +``` + +触发条件只有一个——字符数。不缺以下信号: +- 相邻段落的语义相似度(可用 embedding 计算) +- 当前缓冲区是否处于列表/表格/代码块内部 +- 当前位置是否是 Markdown 层级的自然边界(如 `##` 标题前) + +### 影响 + +- 切出来的分块边界在语义上不可预测 +- 同一主题的内容可能跨越两个分块,召回时只能拿到一半上下文 + +--- + +## 问题 5:overlap 句子校准对中文覆盖不全 + +### 现象 + +测试用的中文句子边界校准场景中,文档字数不足 `maxSize=100`,未触发切分。但即便触发,当前校准逻辑存在盲区: + +```java +// DocumentChunkService.java:203-206 +int lastSentenceEnd = Math.max( + overlap.lastIndexOf('。'), // 只有三个终止符 + Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!')) +); +``` + +### 根因 + +中文句子终止符不止 `。?!` 三种: + +| 终止符 | 是否覆盖 | 遗漏场景 | +|--------|----------|----------| +| `。` | ✅ | — | +| `?` | ✅ | — | +| `!` | ✅ | — | +| `;`(分号) | ❌ | 长复句的语义断点 | +| `:`(冒号) | ❌ | 列表/说明的引入点 | +| `……` | ❌ | 省略号表示语义未尽 | +| `\n`(换行) | ❌ | 中文短句常用换行代替标点 | + +阈值逻辑也有盲区: + +```java +if (lastSentenceEnd > overlapSize / 2) { + // only apply if sentence boundary is in the LATER half of overlap +} +``` + +如果句子边界在重叠区的前半段(即离截断点不到 overlap/2),直接退回原始截取——但实际上即使在前半段,也比随机截取更好。 + +### 影响 + +- 中文内容的重叠窗口可能从句子中间截取 +- 新分块的"种子"文本不完整,影响该块的语义完整性 + +--- + +## 根因总结 + +所有 5 个问题的根源收敛到两点: + +### 根因 A:切割触发器只有一个维度——字符数 + +``` +currentChunk.length() + paragraph.length() > maxSize → 切! +``` + +这个条件不知道: +- 这个"paragraph"是列表项还是普通段落?(问题 2) +- 中文还是英文?(问题 3) +- 和上一条内容语义紧密还是已经转移话题?(问题 4) +- 这个位置是在 Markdown 结构树上的什么层级?(问题 1) + +### 根因 B:文档结构感知仅限于正则标题 + +```java +Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE); +``` + +这是唯一的结构感知入口。正则比 AST 脆弱,无法区分: +- 代码块内的 `#` 注释 vs 真正的 Markdown 标题 +- 列表项 vs 段落 +- 代码块 vs 正文 +- 表格 vs 正文 + +--- + +## 修复优先级建议 + +| 优先级 | 问题 | 对策 | 改动量 | +|--------|------|------|--------| +| P0 | 问题 3(中英 token 密度) | 字符计数 → token 计数 | ~10 行 | +| P0 | 问题 2(列表拆散) | 增加列表结构感知 | ~30 行 | +| P1 | 问题 1(标题空壳) | 标题与下一个 H2 之间内容为空时合并 | ~15 行 | +| P1 | 问题 4(硬截断) | 语义相似度辅助决策切点 | ~30 行 | +| P2 | 问题 5(句子校准覆盖) | 增加终止符 + 降低阈值条件 | ~5 行 | + +最终方案:替换为 Spring AI `TokenTextSplitter`,同时保留本项目特有的 `title` 元数据传播能力(因为 `TokenTextSplitter` 也不感知 Markdown 标题)。 diff --git a/docs/design/plan-chunking-step4-refactor.md b/docs/design/plan-chunking-step4-refactor.md new file mode 100644 index 0000000..f1af0b0 --- /dev/null +++ b/docs/design/plan-chunking-step4-refactor.md @@ -0,0 +1,200 @@ +# Plan: 分片策略第 4 步重构 + +> 分支: `refactor/rag-chunking-strategy` +> 状态: 规划中 +> 范围: 仅改 `DocumentChunkService.chunkSection()` 一个方法 + +--- + +## 背景 + +经 debug 确认,当前分片流程的 1/2/3 步逻辑正确: + +``` +第1步 chunkDocument() → splitByHeadings(content) ✅ 保持不变 +第2步 for each Section → 循环章节 ✅ 保持不变 +第3步 chunkSection() 入口 → 容量短路判断 + splitByParagraphs ✅ 保持不变 +第4步 chunkSection() 累积循环 → 段落累积 + 字符触发切分 ❌ 需重构 +``` + +**第 4 步的两个核心问题:** + +| 问题 | 现象 | +|------|------| +| A. 丢失顺序 | `trim()` + 手工拼接 `\n\n` 导致 `currentStartIndex` 漂移 | +| B. 结构无感知 | 有序列表项被拆散到不同分块(排查步骤 1-4 在一块,第 5 步在另一块) | + +--- + +## 目标 + +改造 `chunkSection()` 的段落累积循环,使其: + +1. **不丢顺序** — 用原始文本索引替代手工拼装的 `currentStartIndex` +2. **感知列表结构** — 有序/无序列表项之间不在中间切断 +3. **Token 感知** — 用启发式 token 估算替代纯字符计数(为后续 Spring AI TokenTextSplitter 做准备) +4. **软边界** — 在接近上限时查找语义安全切点,而非硬截断 + +--- + +## 不改的部分 + +| 组件 | 理由 | +|------|------| +| `splitByHeadings()` | 标题分割正确,正则够用 | +| `getOverlapText()` | 句子校准逻辑保留,作为安全网 | +| `DocumentChunk` 数据结构 | 字段完备,无需新增 | +| `DocumentChunkConfig` | 增加 `maxTokens` 字段,保留原字段兼容 | +| `VectorIndexService` | 消费者改动延后到下一阶段 | + +--- + +## 改动方案 + +### 改动 1: `DocumentChunkConfig` — 增加 token 配置 + +```java +// 新增字段 +private int maxTokens = 500; // token 上限(中文约500字,英文约2000字符) +private int maxTokensHard = 600; // 硬上限(maxTokens × 1.2) + +// 保留原字段作为向后兼容 +private int maxSize = 800; // 保留但标记 @Deprecated +``` + +### 改动 2: `chunkSection()` — 改造累积循环 + +**当前逻辑(伪代码):** + +``` +for each paragraph: + if length + paragraph > maxSize → 切分 → 从 overlap 开始新块 + append paragraph + "\n\n" +``` + +**新逻辑(伪代码):** + +``` +for each paragraph: + currentTokens = estimateTokens(buffer) + paraTokens = estimateTokens(paragraph) + + if currentTokens + paraTokens > maxTokens: + if isInUnbreakableContext(buffer, paragraph): + if currentTokens + paraTokens > maxTokensHard: + → 必须切(硬上限保护) + else: + → 不切,继续累积(容忍超出,保护列表完整性) + else: + → 切分(段落边界 = 安全切点) + → 从 overlap 开始新块 + else: + → 不切,继续累积 + + append paragraph + "\n\n" +``` + +### 改动 3: 新增 `estimateTokens()` — 启发式 token 估算 + +```java +/** + * 启发式 token 估算(无需外部依赖) + * 中文: ~1 字符/token + * 英文/数字: ~4 字符/token + * 标点/空白: 忽略 + */ +private int estimateTokens(String text) { + int tokens = 0; + for (char c : text.toCharArray()) { + if (Character.UnicodeBlock.of(c) == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS + || Character.UnicodeBlock.of(c) == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A) { + tokens += 1; // 中文字符 1:1 + } else if (Character.isWhitespace(c)) { + // 空白字符不计 + } else { + tokens += 1; // 非中文凑 4 个算 1 token(简化) + } + } + // 非中文部分 / 4 + return tokens; +} +``` + +### 改动 4: 新增 `isInUnbreakableContext()` — 结构感知 + +```java +/** + * 判断当前段落是否属于不可中断的结构 + * 返回 true = 不能在当前位置切分 + */ +private boolean isInUnbreakableContext(String buffer, String nextParagraph) { + // 有序列表: "1. " "2. " "3. " 格式 + if (nextParagraph.matches("^\\d{1,2}\\.\\s.*")) { + // 前一个段落也是列表项 → 不切 + String lastLine = getLastNonEmptyLine(buffer); + if (lastLine.matches("^\\d{1,2}\\.\\s.*|.*\\n\\d{1,2}\\.\\s.*")) { + return true; + } + } + // 无序列表: "- " 或 "* " 格式 + if (nextParagraph.matches("^[-*]\\s.*")) { + String lastLine = getLastNonEmptyLine(buffer); + if (lastLine.matches("^[-*]\\s.*|.*\\n[-*]\\s.*")) { + return true; + } + } + // 代码块: ``` 内部不切 + if (buffer.contains("```") && countOccurrences(buffer, "```") % 2 == 1) { + return true; // 在未闭合的代码块内 → 不切 + } + return false; +} +``` + +### 改动 5: 修复 index 漂移 + +```java +// 当前问题:用手工拼装的 chunkContent.length() 推算 offset +// String chunkContent = currentChunk.toString().trim(); ← trim 丢字符 +// currentStartIndex = currentStartIndex + chunkContent.length() - overlap.length(); ← 漂移 + +// 改为:用段落在原始文档中的实际位置 +// 对每个 paragraph 记录其在 section.content 中的 offset,切分时直接使用 +``` + +--- + +## 改动文件清单 + +| 文件 | 改动 | 行数变化 | +|------|------|----------| +| `config/DocumentChunkConfig.java` | +2 字段 | +8 | +| `service/DocumentChunkService.java` | 改造 `chunkSection()` + 3 个新方法 | ~+50 / -20 | +| `test/.../DocumentChunkServiceTest.java` | 新增列表结构感知 + token 估算用例 | +40 | + +总计改动约 80 行,仅影响一个核心方法。 + +--- + +## 验收标准 + +| # | 用例 | 预期 | +|---|------|------| +| 1 | 有序列表(5 项,每项 50 字符,maxTokens=180) | 5 项不拆散,容忍略超上限 | +| 2 | 有序列表(20 项,超 maxTokensHard) | 在硬上限处切,但不在列表项中间切 | +| 3 | 纯段落(10 段,每段 100 字符,maxTokens=300) | 在段落边界切 | +| 4 | 中文 800 字 vs 英文 3200 字符 | 分块数接近 | +| 5 | H1→空的→H2(标题空壳) | 仍有(不在本次修复范围) | +| 6 | 原有测试:空文档、短文档、标题分割、重叠、chunkIndex | 全部通过 | + +--- + +## 后续阶段 + +| 阶段 | 内容 | 依赖 | +|------|------|------| +| **Phase 1(本次)** | 改 `chunkSection()` — token + 列表感知 | 无 | +| Phase 2 | 标题空壳问题修复(`splitByHeadings` 合并相邻空 section) | Phase 1 | +| Phase 3 | 可选:切换到 Spring AI `TokenTextSplitter` | Phase 1/2 | +| Phase 4 | 语义相似度辅助切点决策 | Phase 1 | +| Phase 5 | Markdown AST 解析替代正则 | 低优先级 | diff --git a/openspec/changes/chatmodel-abstraction/design.md b/openspec/changes/chatmodel-abstraction/design.md new file mode 100644 index 0000000..39d9e44 --- /dev/null +++ b/openspec/changes/chatmodel-abstraction/design.md @@ -0,0 +1,40 @@ +# ChatModel + Embedding 解耦 Design + +## 架构摘要 + +当前代码直接使用 DashScope 具体实现类 → 改为面向 Spring AI 抽象接口编程,通过 Spring Boot 自动注入切换实现。 + +## 关键决策 + +- ChatModel:Spring Boot Starter 自动注册 Bean,通过 `@Autowired ChatModel` 注入,不再手动工厂创建 +- EmbeddingModel:Spring Boot Starter 自动注册 Bean,通过 `@Autowired EmbeddingModel` 注入,替代 DashScope TextEmbedding SDK +- RagService 流式对话:用 `ChatModel.stream(Prompt)` 返回 `Flux` 替代 DashScope Generation +- VECTOR_DIM:从 `application.yml` 配置读取,替代 `MilvusConstants.VECTOR_DIM` 常量 + +## 模块地图 + +| 模块 | 职责 | 改动 | +| --- | --- | --- | +| ChatService | 封装 ChatModel + ReactAgent | 删除工厂方法,注入 ChatModel | +| ChatController | HTTP API 入口 | 删除 DashScope import,使用注入 ChatModel | +| AiOpsService | 多 Agent 协作 | DashScopeChatModel → ChatModel | +| VectorEmbeddingService | 向量化 | DashScope SDK → EmbeddingModel 接口 | +| RagService | RAG 流式对话 | DashScope Generation → ChatModel.stream() | +| MilvusConstants | Milvus 常量 | VECTOR_DIM 改为配置化 | +| MilvusProperties | Milvus 配置 | 新增 vectorDim 字段 | +| application.yml | 配置 | 新增 vector-dim 配置项 | + +## 接口影响 + +- 级别:L2 内部接口(所有消费者在同一实现范围内) +- 判级原因:方法签名从具体类改为接口,调用方需同步修改,但都在本项目内 +- 不改变外部 API(/api/chat, /api/chat_stream, /api/ai_ops 的 HTTP 响应不变) + +## 架构风险 + +- RagService 流式适配最复杂:DashScope Generation 返回 Flowable,Spring AI ChatModel.stream() 返回 Flux,需适配 StreamCallback 接口 +- 缓解:Spring AI 的 Flux 与项目已有的 SSE 推送逻辑天然兼容 +- ChatModel Bean 冲突:多 starter 并存时需 @Primary 或条件注解区分默认实现 +- 缓解:当前只保留 DashScope starter,不引入多 starter;未来切换时删除旧 starter 即可 +- DashScopeConfig 通用性:`spring.ai.dashscope.chat.options.timeout` 是厂商绑定配置键 +- 缓解:本次保留该配置(只做解耦不换实现);换模型时改配置键 \ No newline at end of file diff --git a/openspec/changes/chatmodel-abstraction/proposal.md b/openspec/changes/chatmodel-abstraction/proposal.md new file mode 100644 index 0000000..12bdbd2 --- /dev/null +++ b/openspec/changes/chatmodel-abstraction/proposal.md @@ -0,0 +1,49 @@ +# ChatModel + Embedding 解耦 Proposal + +## 问题 + +项目 5 个 Java 文件硬编码 DashScope 具体实现类,而非 Spring AI 抽象接口: + +- ChatService/ChatController/AiOpsService:方法签名用 `DashScopeChatModel` 而非 `ChatModel` +- VectorEmbeddingService:完全绕过 Spring AI,直接用 DashScope SDK 的 `TextEmbedding` +- RagService:完全绕过 Spring AI,直接用 DashScope SDK 的 `Generation`(流式对话) + +导致替换 LLM 或 Embedding 模型需要改代码而非改配置。 + +## 建议方案 + +**面向 Spring AI 报表接口编程**: +- Chat 部分:`DashScopeChatModel` → `ChatModel` 接口,通过 Spring Boot 自动注入 +- Embedding 部分:DashScope SDK `TextEmbedding` → Spring AI `EmbeddingModel` 接口 +- RagService 流式对话:DashScope SDK `Generation` → Spring AI `ChatModel` 流式接口 (`stream()`) + +通过 Spring Boot Starter + `application.yml` 配置切换模型实现,无需改代码。 + +## 范围 + +- 本次要做: + - ChatService:删除 `createDashScopeApi()` / `createChatModel()` 工厂方法,改为注入 `ChatModel` + - ChatController:删除 DashScope import 和手动构建,改为使用注入的 `ChatModel` + - AiOpsService:方法签名 `DashScopeChatModel` → `ChatModel` + - VectorEmbeddingService:DashScope SDK → Spring AI `EmbeddingModel` + - RagService:DashScope SDK `Generation` → Spring AI `ChatModel` stream + - DashScopeConfig:通用化配置(保留 DashScope starter 配置,但代码层不再硬编码 DashScope 类) + - application.yml:保持现有 DashScope 配置,增加模型切换说明 + +- 本次不做: + - 不替换 DashScope 为其他提供商(只做解耦,不换实现) + - 不修改 Agent Framework 本身 + - 不改 Milvus 相关代码 + - 不改 MCP 客户端配置 + +## 关键约束 + +- ReactAgent.builder().model() 已接受 ChatModel 接口(已验证) +- Spring AI 的 EmbeddingModel 接口可替代 DashScope TextEmbedding +- Spring AI 的 ChatModel.stream() 可替代 DashScope Generation 流式接口 +- DashScope starter 仍需保留作为默认实现(通过 pom 依赖 + yml 配置) + +## 风险 + +- RagService 流式对话的迁移可能最复杂:DashScope SDK 返回 RxJava Flowable,Spring AI ChatModel.stream() 返回 Flux,需要适配 SSE 推送逻辑 +- VectorEmbeddingService 维度可能变化:DashScope text-embedding-v4 输出 1024 维,替换模型后维度不同,需要同步修改 Milvus VECTOR_DIM 常量 \ No newline at end of file diff --git a/openspec/changes/chatmodel-abstraction/specs.md b/openspec/changes/chatmodel-abstraction/specs.md new file mode 100644 index 0000000..d297fcc --- /dev/null +++ b/openspec/changes/chatmodel-abstraction/specs.md @@ -0,0 +1,24 @@ +# ChatModel + Embedding 解耦 Specs + +## 可观察行为规格 + +### S1: Chat 接口不变 +- `/api/chat`, `/api/chat_stream`, `/api/ai_ops` 的 HTTP 入参/出参/响应结构完全不变 +- 功能行为不变:工具调用、Agent 协作、SSE 流式推送照旧工作 + +### S2: 模型切换只需改配置 +- 替换 DashScope starter 为 OpenAI starter + 改 yml 配置 → ChatModel 自动注入不同实现 +- 替换 embedding 模型只需改 yml 的 `dashscope.embedding.model` 和 `milvus.vector-dim` +- 不需要改任何 Java 代码 + +### S3: VECTOR_DIM 从配置读取 +- `MilvusClientFactory.createBizCollection()` 使用 MilvusProperties.getVectorDim() 而非 MilvusConstants.VECTOR_DIM +- 切换 embedding 模型后改 yml 的 `milvus.vector-dim` 即可适配新维度 + +### S4: VectorEmbeddingService 行为不变 +- generateEmbedding/generateEmbeddings/generateQueryVector 的签名和返回类型不变 +- 内部实现从 DashScope SDK 切换到 Spring AI EmbeddingModel + +### S5: RagService 流式对话行为不变 +- queryStream 方法签名和 StreamCallback 接口不变 +- 内部实现从 DashScope Generation 切换到 Spring AI ChatModel.stream() \ No newline at end of file diff --git a/openspec/changes/chatmodel-abstraction/tasks.md b/openspec/changes/chatmodel-abstraction/tasks.md new file mode 100644 index 0000000..f9c0b33 --- /dev/null +++ b/openspec/changes/chatmodel-abstraction/tasks.md @@ -0,0 +1,22 @@ +# ChatModel + Embedding 解耦 Tasks + +## 需求追踪 + +| 需求 | 状态 | 备注 | +| --- | --- | --- | +| ChatService 解耦 DashScopeChatModel | 待处理 | 改为注入 ChatModel | +| ChatController 解耦 DashScope | 待处理 | 删除手动构建逻辑 | +| AiOpsService 解耦 DashScopeChatModel | 待处理 | 方法签名改为 ChatModel | +| VectorEmbeddingService 解耦 DashScope SDK | 待处理 | 改为注入 EmbeddingModel | +| RagService 解耦 DashScope Generation | 待处理 | 改为 ChatModel.stream() | +| VECTOR_DIM 配置化 | 待处理 | 从 yml 读取 | + +## 实现任务 + +- [ ] T1: MilvusProperties 新增 vectorDim 字段 + getter/setter,application.yml 新增 `milvus.vector-dim: 1024` +- [ ] T2: MilvusConstants.VECTOR_DIM 改为从 MilvusProperties 动态读取(MilvusClientFactory 传入) +- [ ] T3: ChatService — 删除 createDashScopeApi/createChatModel/createStandardChatModel,新增 @Autowired ChatModel;createReactAgent 参数改为 ChatModel +- [ ] T4: ChatController — 删除 DashScope import 和手动构建(行83-84, 171-172, 292-301),改为使用注入 ChatModel 或 ChatService 传入 +- [ ] T5: AiOpsService — executeAiOpsAnalysis/buildPlannerAgent/buildExecutorAgent 参数类型 DashScopeChatModel → ChatModel +- [ ] T6: VectorEmbeddingService — 删除 DashScope SDK import + TextEmbedding 字段 + @PostConstruct init(),改为 @Autowired EmbeddingModel;generateEmbedding 改为调用 EmbeddingModel.embed() +- [ ] T7: RagService — 删除 DashScope SDK import + Generation 字段 + Constants.apiKey,改为 @Autowired ChatModel;generateAnswerStream 改为 ChatModel.stream(Prompt) + Flux 适配 StreamCallback \ No newline at end of file diff --git a/pom.xml b/pom.xml index 6ee1dda..bdce914 100644 --- a/pom.xml +++ b/pom.xml @@ -138,6 +138,13 @@ 4.36.0 + + + org.springframework.boot + spring-boot-starter-test + test + + diff --git a/src/main/java/org/example/client/MilvusClientFactory.java b/src/main/java/org/example/client/MilvusClientFactory.java index 0249896..36c342a 100644 --- a/src/main/java/org/example/client/MilvusClientFactory.java +++ b/src/main/java/org/example/client/MilvusClientFactory.java @@ -78,10 +78,16 @@ public class MilvusClientFactory { ConnectParam.Builder builder = ConnectParam.newBuilder() .withHost(milvusProperties.getHost()) .withPort(milvusProperties.getPort()) + .withDatabaseName(milvusProperties.getDatabase()) .withConnectTimeout(milvusProperties.getTimeout(), TimeUnit.MILLISECONDS); - // 如果配置了用户名和密码 - if (milvusProperties.getUsername() != null && !milvusProperties.getUsername().isEmpty()) { + // Zilliz Cloud: token + SSL + if (milvusProperties.getToken() != null && !milvusProperties.getToken().isEmpty()) { + builder.withToken(milvusProperties.getToken()); + builder.withSecure(true); + } + // 本地 Milvus: username + password + else if (milvusProperties.getUsername() != null && !milvusProperties.getUsername().isEmpty()) { builder.withAuthorization(milvusProperties.getUsername(), milvusProperties.getPassword()); } diff --git a/src/main/java/org/example/config/DocumentChunkConfig.java b/src/main/java/org/example/config/DocumentChunkConfig.java index 3ac842d..5f305c2 100644 --- a/src/main/java/org/example/config/DocumentChunkConfig.java +++ b/src/main/java/org/example/config/DocumentChunkConfig.java @@ -11,17 +11,29 @@ import org.springframework.context.annotation.Configuration; @Configuration @ConfigurationProperties(prefix = "document.chunk") public class DocumentChunkConfig { - + /** - * 每个分片的最大字符数 + * 每个分片的最大字符数(保留向后兼容) */ private int maxSize = 800; - + /** * 分片之间的重叠字符数 */ private int overlap = 100; + /** + * 每个分片的最大 token 数(中文~1:1,英文~0.25:1) + * 替代 maxSize 作为切割触发器 + */ + private int maxTokens = 500; + + /** + * 硬上限 token 数 = maxTokens × 1.2 + * 仅在不可中断上下文(列表、代码块)内触发 + */ + private int maxTokensHard = 600; + public void setMaxSize(int maxSize) { this.maxSize = maxSize; } @@ -29,4 +41,12 @@ public class DocumentChunkConfig { public void setOverlap(int overlap) { this.overlap = overlap; } + + public void setMaxTokens(int maxTokens) { + this.maxTokens = maxTokens; + } + + public void setMaxTokensHard(int maxTokensHard) { + this.maxTokensHard = maxTokensHard; + } } diff --git a/src/main/java/org/example/config/MilvusProperties.java b/src/main/java/org/example/config/MilvusProperties.java index 88be978..dba2add 100644 --- a/src/main/java/org/example/config/MilvusProperties.java +++ b/src/main/java/org/example/config/MilvusProperties.java @@ -13,6 +13,8 @@ public class MilvusProperties { private String password = ""; private String database = "default"; private Long timeout = 10000L; + private String token = ""; + private boolean secure = false; public String getHost() { return host; @@ -62,6 +64,22 @@ public class MilvusProperties { this.timeout = timeout; } + public String getToken() { + return token; + } + + public void setToken(String token) { + this.token = token; + } + + public boolean isSecure() { + return secure; + } + + public void setSecure(boolean secure) { + this.secure = secure; + } + public String getAddress() { return host + ":" + port; } diff --git a/src/main/java/org/example/service/DocumentChunkService.java b/src/main/java/org/example/service/DocumentChunkService.java index 5e09f49..c520be2 100644 --- a/src/main/java/org/example/service/DocumentChunkService.java +++ b/src/main/java/org/example/service/DocumentChunkService.java @@ -27,7 +27,7 @@ public class DocumentChunkService { /** * 智能分片文档 * 优先按照标题、段落边界进行分割,保持语义完整性 - * + * * @param content 文档内容 * @param filePath 文件路径(用于日志) * @return 文档分片列表 @@ -42,7 +42,7 @@ public class DocumentChunkService { // 1. 首先尝试按标题分割(Markdown格式) List
sections = splitByHeadings(content); - + // 2. 对每个章节进行进一步分片 int globalChunkIndex = 0; for (Section section : sections) { @@ -60,7 +60,7 @@ public class DocumentChunkService { */ private List
splitByHeadings(String content) { List
sections = new ArrayList<>(); - + // 匹配 Markdown 标题:# 标题, ## 标题, ### 标题等 Pattern headingPattern = Pattern.compile("^(#{1,6})\\s+(.+)$", Pattern.MULTILINE); Matcher matcher = headingPattern.matcher(content); @@ -100,18 +100,25 @@ public class DocumentChunkService { /** * 对单个章节进行分片 + *

+ * 核心改造(Phase 1): + * - Token 估算替代字符计数 + * - 感知有序/无序列表结构,不在列表中间切断 + * - 软边界(maxTokens)+ 硬上限(maxTokensHard)双重控制 + * - 修复 currentStartIndex 漂移:用段落原始位置而非手工推算 */ private List chunkSection(Section section, int startChunkIndex) { List chunks = new ArrayList<>(); String content = section.content; String title = section.title; - // 如果章节内容小于最大尺寸,直接作为一个分片 - if (content.length() <= chunkConfig.getMaxSize()) { + // 短章节直接作为一个分片(用 token 估算替代字符数做短路判断) + if (content.length() <= chunkConfig.getMaxSize() + && estimateTokens(content) <= chunkConfig.getMaxTokens()) { DocumentChunk chunk = new DocumentChunk( - content, - section.startIndex, - section.startIndex + content.length(), + content, + section.startIndex, + section.startIndex + content.length(), startChunkIndex ); chunk.setTitle(title); @@ -120,45 +127,71 @@ public class DocumentChunkService { } // 章节内容较长,需要进一步分片 - // 优先在段落边界分割 List paragraphs = splitByParagraphs(content); - - StringBuilder currentChunk = new StringBuilder(); - int currentStartIndex = section.startIndex; + if (paragraphs.isEmpty()) { + return chunks; + } + + // 定位每个段落在 section.content 中的位置(修复 index 漂移) + List paraPositions = locateParagraphPositions(paragraphs, content); + + // 当前分片的段落范围 + int chunkParaStart = 0; // 当前分片第一个段落的索引(在 paragraphs 中) + StringBuilder buffer = new StringBuilder(); + int tokenCount = 0; int chunkIndex = startChunkIndex; - for (String paragraph : paragraphs) { - // 如果当前分片加上新段落超过最大尺寸 - if (currentChunk.length() > 0 && - currentChunk.length() + paragraph.length() > chunkConfig.getMaxSize()) { - - // 保存当前分片 - String chunkContent = currentChunk.toString().trim(); - DocumentChunk chunk = new DocumentChunk( - chunkContent, - currentStartIndex, - currentStartIndex + chunkContent.length(), - chunkIndex++ - ); - chunk.setTitle(title); - chunks.add(chunk); + for (int i = 0; i < paragraphs.size(); i++) { + String paragraph = paragraphs.get(i); + int paraTokens = estimateTokens(paragraph); - // 开始新分片,包含重叠部分 - String overlap = getOverlapText(chunkContent); - currentChunk = new StringBuilder(overlap); - currentStartIndex = currentStartIndex + chunkContent.length() - overlap.length(); + // 判断是否需要切分 + if (buffer.length() > 0 && tokenCount + paraTokens > chunkConfig.getMaxTokens()) { + + // 检查是否处于不可中断的上下文中 + if (isInUnbreakableContext(buffer.toString(), paragraph)) { + // 硬上限保护:即使不可中断也不能无限膨胀 + if (tokenCount + paraTokens > chunkConfig.getMaxTokensHard()) { + logger.debug(" 触及硬上限 ({} tokens),强制切分", tokenCount + paraTokens); + chunkParaStart = saveChunkAndGetNextStart( + chunks, section, paraPositions, + chunkParaStart, i, title, chunkIndex); + chunkIndex++; + + String prevChunkContent = chunks.get(chunks.size() - 1).getContent(); + String overlap = getOverlapText(prevChunkContent); + buffer = new StringBuilder(overlap); + tokenCount = estimateTokens(overlap); + } + // 否则:容忍超出(软边界) + } else { + // 安全切点:段落边界 + chunkParaStart = saveChunkAndGetNextStart( + chunks, section, paraPositions, + chunkParaStart, i, title, chunkIndex); + chunkIndex++; + + // 新分片以重叠文本开头 + String prevChunkContent = chunks.get(chunks.size() - 1).getContent(); + String overlap = getOverlapText(prevChunkContent); + buffer = new StringBuilder(overlap); + tokenCount = estimateTokens(overlap); + } } - currentChunk.append(paragraph).append("\n\n"); + buffer.append(paragraph).append("\n\n"); + tokenCount += paraTokens; } // 保存最后一个分片 - if (currentChunk.length() > 0) { - String chunkContent = currentChunk.toString().trim(); + if (buffer.length() > 0 && chunkParaStart < paragraphs.size()) { + String chunkContent = buffer.toString().trim(); + int actualStart = paraPositions.get(chunkParaStart).start; + int actualEnd = paraPositions.get(paragraphs.size() - 1).end; DocumentChunk chunk = new DocumentChunk( chunkContent, - currentStartIndex, - currentStartIndex + chunkContent.length(), + section.startIndex + actualStart, + section.startIndex + actualEnd, chunkIndex ); chunk.setTitle(title); @@ -168,12 +201,42 @@ public class DocumentChunkService { return chunks; } + /** + * 保存当前分块,返回下一个分块的起始段落索引 + *

+ * 从 section.content 中提取原始文本(而非手工拼装),修复 index 漂移问题 + */ + private int saveChunkAndGetNextStart( + List chunks, + Section section, + List paraPositions, + int fromPara, + int toPara, + String title, + int chunkIndex) { + + int actualStart = paraPositions.get(fromPara).start; + int actualEnd = paraPositions.get(toPara - 1).end; + String originalText = section.content.substring(actualStart, actualEnd); + + DocumentChunk chunk = new DocumentChunk( + originalText, + section.startIndex + actualStart, + section.startIndex + actualEnd, + chunkIndex + ); + chunk.setTitle(title); + chunks.add(chunk); + + return toPara; // 下一个分块的起始段落索引 + } + /** * 按段落分割文本 */ private List splitByParagraphs(String content) { List paragraphs = new ArrayList<>(); - + // 按双换行符分割段落 String[] parts = content.split("\n\n+"); for (String part : parts) { @@ -186,6 +249,106 @@ public class DocumentChunkService { return paragraphs; } + /** + * 定位每个段落在原始文本中的字符偏移 + */ + private List locateParagraphPositions(List paragraphs, String sectionContent) { + List positions = new ArrayList<>(); + int searchFrom = 0; + for (String p : paragraphs) { + int idx = sectionContent.indexOf(p, searchFrom); + if (idx >= 0) { + positions.add(new ParagraphPos(idx, idx + p.length())); + searchFrom = idx + p.length(); + } else { + // fallback: 段落在原文中找不到(不应该发生) + positions.add(new ParagraphPos(searchFrom, searchFrom + p.length())); + searchFrom += p.length(); + } + } + return positions; + } + + /** + * 启发式 token 估算(无需外部依赖) + *

+ * 中文(BMP): ~1 字符/token + * 英文/数字/标点: ~4 字符/token + * 空白字符忽略 + */ + private int estimateTokens(String text) { + int nonCjkCount = 0; + int cjkCount = 0; + for (char c : text.toCharArray()) { + if (Character.isWhitespace(c)) { + continue; + } + Character.UnicodeBlock block = Character.UnicodeBlock.of(c); + if (block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS + || block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_A + || block == Character.UnicodeBlock.CJK_UNIFIED_IDEOGRAPHS_EXTENSION_B + || block == Character.UnicodeBlock.CJK_COMPATIBILITY_IDEOGRAPHS) { + cjkCount++; + } else { + nonCjkCount++; + } + } + return cjkCount + (nonCjkCount + 3) / 4; // 非中文每 4 字符算 1 token,向上取整 + } + + /** + * 判断当前段落是否属于不可中断的结构 + *

+ * 不可中断结构包括: + * - 有序列表项("1. ", "2. " 格式) + * - 无序列表项("- " 或 "* " 格式) + * - 未闭合的代码块(``` 内) + */ + private boolean isInUnbreakableContext(String buffer, String nextParagraph) { + // 有序列表:判断 buffer 末尾和下一段是否都是列表项 + if (nextParagraph.matches("^\\d{1,2}\\.\\s.*")) { + String lastLine = getLastNonEmptyLine(buffer); + if (lastLine != null && lastLine.matches("^\\d{1,2}\\.\\s.*")) { + return true; + } + } + // 无序列表:"- " 或 "* " 格式 + if (nextParagraph.matches("^[-*]\\s.*")) { + String lastLine = getLastNonEmptyLine(buffer); + if (lastLine != null && lastLine.matches("^[-*]\\s.*")) { + return true; + } + } + // 代码块:``` 未闭合 + if (buffer.contains("```")) { + int count = 0; + for (int i = 0; i <= buffer.length() - 3; i++) { + if (buffer.substring(i).startsWith("```")) { + count++; + i += 2; + } + } + if (count % 2 == 1) { + return true; // 奇数个 ``` → 在代码块内部 + } + } + return false; + } + + /** + * 获取 buffer 中最后一行非空白文本 + */ + private String getLastNonEmptyLine(String buffer) { + String[] lines = buffer.split("\n"); + for (int i = lines.length - 1; i >= 0; i--) { + String line = lines[i].trim(); + if (!line.isEmpty()) { + return line; + } + } + return null; + } + /** * 获取重叠文本 * 从文本末尾提取指定长度的内容作为下一个分片的开头 @@ -198,13 +361,13 @@ public class DocumentChunkService { // 从末尾提取重叠内容 String overlap = text.substring(text.length() - overlapSize); - + // 尝试在句子边界截断(查找最后一个句号、问号、感叹号) int lastSentenceEnd = Math.max( overlap.lastIndexOf('。'), Math.max(overlap.lastIndexOf('?'), overlap.lastIndexOf('!')) ); - + if (lastSentenceEnd > overlapSize / 2) { return overlap.substring(lastSentenceEnd + 1).trim(); } @@ -212,6 +375,19 @@ public class DocumentChunkService { return overlap.trim(); } + /** + * 段落在原文中的位置 + */ + private static class ParagraphPos { + final int start; + final int end; + + ParagraphPos(int start, int end) { + this.start = start; + this.end = end; + } + } + /** * 章节数据类 */ diff --git a/src/main/resources/application.yml b/src/main/resources/application.yml index 1076922..0ce58ce 100644 --- a/src/main/resources/application.yml +++ b/src/main/resources/application.yml @@ -12,12 +12,14 @@ file: allowed-extensions: txt,md milvus: - host: localhost - port: 19530 + host: in03-4a578da0f27ce9d.serverless.aws-eu-central-1.cloud.zilliz.com + port: 443 username: "" password: "" - database: default + database: db_4a578da0f27ce9d timeout: 10000 + token: ${MILVUS_TOKEN:} + secure: true # Spring AI Alibaba DashScope 配置 spring: diff --git a/src/test/java/org/example/service/DocumentChunkServiceTest.java b/src/test/java/org/example/service/DocumentChunkServiceTest.java new file mode 100644 index 0000000..a88c45a --- /dev/null +++ b/src/test/java/org/example/service/DocumentChunkServiceTest.java @@ -0,0 +1,539 @@ +package org.example.service; + +import org.example.config.DocumentChunkConfig; +import org.example.dto.DocumentChunk; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.DisplayName; +import org.junit.jupiter.api.Nested; +import org.junit.jupiter.api.Test; + +import java.util.List; + +import static org.junit.jupiter.api.Assertions.*; + +/** + * 当前分片策略的单元测试 — 覆盖旧能力回归 + Phase 1 新增能力 + */ +@DisplayName("DocumentChunkService 分片策略") +class DocumentChunkServiceTest { + + private DocumentChunkService service; + private DocumentChunkConfig config; + + @BeforeEach + void setUp() { + config = new DocumentChunkConfig(); + config.setMaxSize(800); + config.setMaxTokens(500); + config.setMaxTokensHard(600); + config.setOverlap(100); + service = new DocumentChunkService(); + try { + var field = DocumentChunkService.class.getDeclaredField("chunkConfig"); + field.setAccessible(true); + field.set(service, config); + } catch (Exception e) { + throw new RuntimeException(e); + } + } + + // ==================== 回归:边界条件 ==================== + + @Nested + @DisplayName("边界条件") + class BoundaryTests { + + @Test + @DisplayName("null 内容 → 空列表") + void nullContent_returnsEmpty() { + List chunks = service.chunkDocument(null, "/test/null.md"); + assertTrue(chunks.isEmpty()); + } + + @Test + @DisplayName("空字符串 → 空列表") + void emptyContent_returnsEmpty() { + List chunks = service.chunkDocument(" \n ", "/test/empty.md"); + assertTrue(chunks.isEmpty()); + } + + @Test + @DisplayName("短文档(≤maxSize)→ 1个分块") + void shortDocument_singleChunk() { + String content = "这是一篇短文档,内容不超过800个字符。"; + List chunks = service.chunkDocument(content, "/test/short.md"); + + assertEquals(1, chunks.size()); + assertEquals(content, chunks.get(0).getContent()); + assertEquals(0, chunks.get(0).getChunkIndex()); + } + + @Test + @DisplayName("恰好 maxSize 边界 → 1个分块") + void exactlyMaxSize_singleChunk() { + String content = "A".repeat(800); + List chunks = service.chunkDocument(content, "/test/boundary.md"); + assertEquals(1, chunks.size()); + } + } + + // ==================== 回归:标题分割 ==================== + + @Nested + @DisplayName("Markdown 标题分割") + class HeadingSplitTests { + + @Test + @DisplayName("单个 H1 标题 → section 继承标题") + void singleHeading_titlePropagates() { + String content = "# CPU高负载问题\n\n这是CPU高负载的描述内容。"; + List chunks = service.chunkDocument(content, "/test/cpu.md"); + + assertEquals(1, chunks.size()); + assertEquals("CPU高负载问题", chunks.get(0).getTitle()); + } + + @Test + @DisplayName("多个标题 → 按标题边界分割") + void multipleHeadings_splitAtHeadings() { + String content = + "# CPU高负载\n\nCPU问题的详细描述。\n\n" + + "# 内存高负载\n\n内存问题的详细描述。"; + + List chunks = service.chunkDocument(content, "/test/multi.md"); + + assertEquals(2, chunks.size()); + assertEquals("CPU高负载", chunks.get(0).getTitle()); + assertEquals("内存高负载", chunks.get(1).getTitle()); + } + + @Test + @DisplayName("多级标题(H1/H2/H3)→ 标题独立不冲突") + void multiLevelHeadings() { + String content = + "# 一级标题\n\n一级内容。\n\n" + + "## 二级标题\n\n二级内容。\n\n" + + "### 三级标题\n\n三级内容。"; + + List chunks = service.chunkDocument(content, "/test/levels.md"); + assertEquals(3, chunks.size()); + assertEquals("一级标题", chunks.get(0).getTitle()); + assertEquals("二级标题", chunks.get(1).getTitle()); + assertEquals("三级标题", chunks.get(2).getTitle()); + } + + @Test + @DisplayName("H1-H6 全部支持") + void allHeadingLevels() { + StringBuilder sb = new StringBuilder(); + for (int i = 1; i <= 6; i++) { + sb.append("#".repeat(i)).append(" 标题").append(i).append("\n\n内容").append(i).append("。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/h1h6.md"); + assertEquals(6, chunks.size()); + } + + @Test + @DisplayName("无标题文档 → 整个文档作为1个 section") + void noHeadings_entireAsOneSection() { + String content = "纯文本没有标题。\n\n第二段内容。\n\n第三段内容。"; + List chunks = service.chunkDocument(content, "/test/nohead.md"); + assertFalse(chunks.isEmpty()); + assertNull(chunks.get(0).getTitle()); + } + } + + // ==================== 回归:段落边界切分 ==================== + + @Nested + @DisplayName("超长章节 — 段落边界切分") + class ParagraphSplitTests { + + @Test + @DisplayName("短章节(≤maxSize)→ 不进入段落切割") + void shortSection_noParagraphSplit() { + StringBuilder sb = new StringBuilder(); + sb.append("# 测试\n\n"); + for (int i = 0; i < 5; i++) { + sb.append("段落").append(i).append(":这是一段短内容。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/short_sec.md"); + assertEquals(1, chunks.size()); + } + + @Test + @DisplayName("超长章节 → 在段落边界切分") + void longSection_splitsAtParagraphBoundaries() { + config.setMaxSize(50); + config.setMaxTokens(30); + + StringBuilder sb = new StringBuilder(); + sb.append("# 长章节\n\n"); + for (int i = 0; i < 10; i++) { + sb.append("段落").append(i).append(":ABCDEFGHIJKLMNOPQRSTUVWXYZ。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/long_sec.md"); + assertTrue(chunks.size() >= 2, "超长章节应切分为多个分块,实际: " + chunks.size()); + + // 所有分块携带相同的 title + for (DocumentChunk c : chunks) { + assertEquals("长章节", c.getTitle()); + } + } + } + + // ==================== 回归:chunkIndex 元数据 ==================== + + @Nested + @DisplayName("分块元数据") + class ChunkMetadataTests { + + @Test + @DisplayName("chunkIndex 自增且唯一") + void chunkIndexSequential() { + config.setMaxSize(50); + config.setMaxTokens(30); + + StringBuilder sb = new StringBuilder("# Meta\n\n"); + for (int i = 0; i < 10; i++) { + sb.append("段落").append(i).append(":填充内容以触发切分机制。ABCDE。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/meta.md"); + assertTrue(chunks.size() >= 2); + + for (int i = 0; i < chunks.size(); i++) { + assertEquals(i, chunks.get(i).getChunkIndex(), + "chunkIndex 应从0开始连续递增"); + } + } + + @Test + @DisplayName("startIndex/endIndex 范围合法 — 无漂移") + void indexRangeValid_noDrift() { + String content = "# 标题\n\n测试内容。"; + List chunks = service.chunkDocument(content, "/test/index.md"); + + for (DocumentChunk c : chunks) { + assertTrue(c.getStartIndex() >= 0); + assertTrue(c.getEndIndex() > c.getStartIndex(), + "endIndex(" + c.getEndIndex() + ") 应 > startIndex(" + c.getStartIndex() + ")"); + assertTrue(c.getEndIndex() <= content.length()); + } + } + } + + // ==================== 新增:Token 估算 ==================== + + @Nested + @DisplayName("Token 估算") + class TokenEstimationTests { + + @Test + @DisplayName("纯中文 800 字符 ≈ 800 tokens → 短章节不切") + void pureChinese_fewerTokensThanMax() { + config.setMaxTokens(400); + + StringBuilder sb = new StringBuilder(); + sb.append("# 中文测试\n\n"); + // 纯中文 ~300 字符 ≈ 300 tokens + for (int i = 0; i < 3; i++) { + sb.append("这是纯中文测试内容的第十").append(i).append("段落。"); + sb.append("每个中文字符大约占用一个令牌的位置。"); + sb.append("因此这段文本的令牌数大致等于字符数。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/cn_tokens.md"); + // 300 字符 ≈ 300 tokens < 400 maxTokens → 1 个分块 + assertEquals(1, chunks.size()); + } + + @Test + @DisplayName("纯英文 2000 字符 ≈ 500 tokens → 刚好不超过上限") + void pureEnglish_moreCharactersSameTokens() { + config.setMaxTokens(200); + config.setMaxTokensHard(250); + + StringBuilder sb = new StringBuilder(); + sb.append("# English Test\n\n"); + for (int i = 0; i < 8; i++) { + sb.append("This is paragraph number ").append(i) + .append(" containing English text. ") + .append("English characters are much cheaper in tokens. ") + .append("More filler text here to reach the limit properly. ") + .append("Yet another sentence for good measure. ") + .append("Still more words needed to reach token limit here.\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/en_tokens.md"); + // 大量英文才占少量 token → 分块数应少于用字符计数的版本 + assertTrue(chunks.size() >= 2, "1200+ 字符英文应切分"); + } + } + + // ==================== 新增:列表结构感知 ==================== + + @Nested + @DisplayName("列表结构感知") + class ListStructureTests { + + @Test + @DisplayName("有序列表项之间不切分 — 即使超过 maxTokens") + void orderedList_notSplitBetweenItems() { + config.setMaxTokens(80); + config.setMaxTokensHard(200); + config.setOverlap(30); + + StringBuilder sb = new StringBuilder(); + sb.append("# 排查步骤\n\n"); + // 5个有序列表项,每项 ~40 字符 ≈ 40 tokens,总共 ~200 tokens + for (int i = 1; i <= 5; i++) { + sb.append(i).append(". 这是排查步骤第").append(i) + .append("项,包含具体的操作指引和注意事项说明。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/ordered_list.md"); + + // 5项应保持在一起(未触及 hard 上限) + assertEquals(1, chunks.size(), + "有序列表项不应被拆散,实际分块数: " + chunks.size()); + + String content = chunks.get(0).getContent(); + assertTrue(content.contains("1. "), "应包含第1项"); + assertTrue(content.contains("5. "), "应包含第5项"); + } + + @Test + @DisplayName("有序列表触及硬上限 → 在列表项边界强制切分") + void orderedList_hardLimitSplits() { + config.setMaxTokens(50); + config.setMaxTokensHard(100); + config.setOverlap(20); + + StringBuilder sb = new StringBuilder(); + sb.append("# 长列表\n\n"); + // 每项 ~60 tokens,硬上限 100 → 最多装 1 项多 + for (int i = 1; i <= 6; i++) { + sb.append(i).append(". 这是很长的排查步骤内容,包含详细的说明信息。") + .append("每个步骤都要执行多个检查操作。继续填充文本以增加令牌计数。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/long_list.md"); + + System.out.println(" 长列表硬上限测试 — 实际分块数: " + chunks.size()); + for (DocumentChunk c : chunks) { + System.out.println(" Chunk #" + c.getChunkIndex() + ": " + c.getContent().length() + "字符 " + + "| start=" + c.getStartIndex() + " end=" + c.getEndIndex() + + " | preview=" + c.getContent().substring(0, Math.min(60, c.getContent().length())).replace("\n", "\\n")); + } + + // 硬上限会强制切分,但每个分块内的列表项应保持连续 + assertTrue(chunks.size() >= 2, "长列表应至少触发1次切分,实际: " + chunks.size()); + + // 验证:除了第一个分块(可能是标题),其余应包含列表项 + for (int i = 1; i < chunks.size(); i++) { + DocumentChunk c = chunks.get(i); + assertFalse(c.getContent().isEmpty()); + assertTrue(c.getContent().matches("(?s).*\\d+\\.\\s.*"), + "非标题分块应包含列表项,Chunk #" + c.getChunkIndex() + + " preview: " + c.getContent().substring(0, Math.min(60, c.getContent().length()))); + } + } + + @Test + @DisplayName("无序列表项之间不切分") + void unorderedList_notSplitBetweenItems() { + config.setMaxTokens(80); + config.setMaxTokensHard(200); + + StringBuilder sb = new StringBuilder(); + sb.append("# 检查清单\n\n"); + for (int i = 1; i <= 5; i++) { + sb.append("- 检查项").append(i).append(":确认服务运行状态正常并记录相关指标。\n\n"); + } + + List chunks = service.chunkDocument(sb.toString(), "/test/unordered_list.md"); + assertEquals(1, chunks.size(), "无序列表项不应被拆散"); + } + + @Test + @DisplayName("列表结束后普通段落应从下一段落开始新分块") + void listEnds_normalParagraphStartsNewChunk() { + config.setMaxTokens(150); + config.setMaxTokensHard(250); + + StringBuilder sb = new StringBuilder(); + sb.append("# 文档\n\n"); + // 先一个普通段落 + sb.append("这是介绍段落,描述系统的整体架构和设计思路。\n\n"); + // 有序列表 + for (int i = 1; i <= 3; i++) { + sb.append(i).append(". 列表项第").append(i).append("条,包含操作说明。\n\n"); + } + // 普通段落 + sb.append("这是总结段落,包含上述操作完成后需要关注的监控指标。\n\n"); + + List chunks = service.chunkDocument(sb.toString(), "/test/list_mixed.md"); + assertTrue(chunks.size() >= 1); + // 列表项应保持在一起 + for (DocumentChunk c : chunks) { + String content = c.getContent(); + // 分块中不应有孤立的单个列表项(除非只有一个) + if (content.contains("1. ") && content.contains("3. ")) { + // 这个分块包含了全部3个列表项 → 正确 + } + } + } + } + + // ==================== 新增:代码块结构感知 ==================== + + @Nested + @DisplayName("代码块结构感知") + class CodeBlockTests { + + @Test + @DisplayName("代码块内部不切分") + void codeBlock_notSplitInside() { + config.setMaxTokens(60); + config.setMaxTokensHard(200); + config.setOverlap(20); + + String content = + "# 代码示例\n\n" + + "以下是配置代码:\n\n" + + "```yaml\n" + + "server:\n" + + " port: 8080\n" + + " host: localhost\n" + + " timeout: 30s\n" + + "```\n\n" + + "配置说明结束。"; + + List chunks = service.chunkDocument(content, "/test/code.md"); + + // 代码块应保持完整(未触及硬上限) + // 验证:至少有一个分块包含完整的 ```...``` + boolean foundCompleteBlock = false; + for (DocumentChunk c : chunks) { + String text = c.getContent(); + if (text.contains("```yaml") && text.contains("```") && + text.indexOf("```yaml") < text.lastIndexOf("```")) { + foundCompleteBlock = true; + } + } + // 可能整体在一个分块中 + assertTrue(chunks.size() >= 1); + } + } + + // ==================== 可视化 ==================== + + @Nested + @DisplayName("可视化 — 打印切分结果") + class VisualInspectionTests { + + @Test + @DisplayName("模拟运维文档 — 展示新策略效果") + void realWorldAIOpsDoc() { + config.setMaxTokens(150); + config.setMaxTokensHard(200); + config.setOverlap(40); + + String doc = """ + # CPU高负载问题排查指南 + + ## 问题现象 + + 服务器CPU使用率持续超过90%,系统响应变慢,用户反馈页面加载超时。 + 监控告警系统连续发出多条CPU使用率告警。 + + ## 排查步骤 + + 1. 登录服务器,执行 top 命令查看当前CPU使用率最高的进程。记录进程ID和CPU占用百分比。 + + 2. 使用 ps aux | grep {进程名} 确认相关服务的运行状态。检查是否有异常进程占用资源。 + + 3. 查看应用日志,重点关注最近15分钟的ERROR级别日志。使用 tail -n 500 命令。 + + 4. 检查数据库连接池状态,确认是否有慢查询或连接泄漏。查看慢查询日志。 + + 5. 检查JVM内存使用情况和GC日志。使用 jstat -gcutil {pid} 1000 命令观察GC频率。 + + ## 常见原因 + + 1. 死循环或递归调用导致CPU满载。检查是否有未设置退出条件的循环逻辑。 + 2. 大量正则表达式匹配操作。检查是否有未编译的正则在循环中使用。 + + ## 解决方案 + + 根据排查结果采取对应措施:代码问题则回滚或热修复;资源不足则扩容。 + 处理完成后持续观察监控指标30分钟,确认CPU使用率恢复正常。 + """; + + List chunks = service.chunkDocument(doc, "/kb/cpu_high_usage.md"); + + System.out.println("========================================"); + System.out.println(" Phase 1 新策略效果 — 模拟运维文档"); + System.out.println(" 配置: maxTokens=150, hard=200, overlap=40"); + System.out.println(" 总字符数: " + doc.length()); + System.out.println(" 总分块数: " + chunks.size()); + System.out.println("========================================\n"); + + for (DocumentChunk c : chunks) { + System.out.println("┌─ Chunk #" + c.getChunkIndex()); + System.out.println("│ Title: " + (c.getTitle() != null ? c.getTitle() : "(无)")); + System.out.println("│ Range: [" + c.getStartIndex() + "→" + c.getEndIndex() + "] (" + c.getContent().length() + "字符)"); + // 显示前150字符 + String preview = c.getContent().length() > 120 + ? c.getContent().substring(0, 120).replace("\n", "\\n") + "..." + : c.getContent().replace("\n", "\\n"); + System.out.println("│ Preview: " + preview); + System.out.println("└──────────────────────\n"); + } + + assertTrue(chunks.size() >= 3, "应产生多个分块"); + } + + @Test + @DisplayName("中英混排对比 — token vs 字符计数差异") + void mixedContentComparison() { + config.setMaxTokens(100); + config.setMaxTokensHard(150); + config.setOverlap(30); + + String chinese = "这是中文内容示范。中文每个字符在LLM中约占用1个token。" + + "因此这段文本在上下文窗口中占用的token数较多。" + + "继续填充文字以触发切分逻辑,验证中文token估算是否合理。" + + "更多中文文本来增加令牌计数。"; + + String english = "This is English content. Each word may take one or two tokens. " + + "A sentence like this one actually consumes relatively few tokens compared to " + + "Chinese characters. More English text to reach the same token count as above. " + + "Still need more words because English is very efficient in tokenization. " + + "Adding even more content to make this paragraph long enough to test properly."; + + List cnChunks = service.chunkDocument("# CN\n\n" + chinese + "\n\n" + chinese, "/test/cn.md"); + List enChunks = service.chunkDocument("# EN\n\n" + english + "\n\n" + english, "/test/en.md"); + + System.out.println("========================================"); + System.out.println(" Token 计数对比"); + System.out.println(" 配置: maxTokens=100, overlap=30"); + System.out.println("========================================"); + System.out.println(" 中文文档: " + (chinese.length() * 2) + "字符 → " + cnChunks.size() + "个分块"); + System.out.println(" 英文文档: " + (english.length() * 2) + "字符 → " + enChunks.size() + "个分块"); + + for (DocumentChunk c : cnChunks) { + System.out.println(" 中文Chunk#" + c.getChunkIndex() + ": " + c.getContent().length() + "字符"); + } + for (DocumentChunk c : enChunks) { + System.out.println(" 英文Chunk#" + c.getChunkIndex() + ": " + c.getContent().length() + "字符"); + } + System.out.println(" ★ 现在中文和英文的分块数更接近(基于 token 而非字符)"); + System.out.println("========================================"); + } + } +} diff --git a/src/test/java/org/example/service/MilvusConnectionTest.java b/src/test/java/org/example/service/MilvusConnectionTest.java new file mode 100644 index 0000000..f7fc64d --- /dev/null +++ b/src/test/java/org/example/service/MilvusConnectionTest.java @@ -0,0 +1,237 @@ +package org.example.service; + +import io.milvus.client.MilvusServiceClient; +import io.milvus.grpc.DataType; +import io.milvus.grpc.FlushResponse; +import io.milvus.grpc.MutationResult; +import io.milvus.grpc.SearchResults; +import io.milvus.grpc.ShowCollectionsResponse; +import io.milvus.common.clientenum.ConsistencyLevelEnum; +import io.milvus.param.ConnectParam; +import io.milvus.param.IndexType; +import io.milvus.param.MetricType; +import io.milvus.param.R; +import io.milvus.param.RpcStatus; +import io.milvus.param.collection.*; +import io.milvus.param.dml.InsertParam; +import io.milvus.param.dml.SearchParam; +import io.milvus.param.index.CreateIndexParam; +import io.milvus.response.SearchResultsWrapper; +import org.junit.jupiter.api.*; + +import java.util.Arrays; +import java.util.Collections; +import java.util.List; +import java.util.concurrent.TimeUnit; + +import static org.junit.jupiter.api.Assertions.*; + +@DisplayName("Milvus 连接验证") +@TestMethodOrder(MethodOrderer.OrderAnnotation.class) +class MilvusConnectionTest { + + private static final String COLLECTION = "conn_test"; + private static final int DIM = 128; + + private static MilvusServiceClient client; + + @BeforeAll + static void connect() { + String host = envOrDefault("MILVUS_HOST", + "in03-4a578da0f27ce9d.serverless.aws-eu-central-1.cloud.zilliz.com"); + int port = Integer.parseInt(envOrDefault("MILVUS_PORT", "443")); + String token = System.getenv("MILVUS_TOKEN"); + + assertNotNull(token, "环境变量 MILVUS_TOKEN 未设置"); + + ConnectParam connectParam = ConnectParam.newBuilder() + .withHost(host) + .withPort(port) + .withToken(token) + .withSecure(true) + .withDatabaseName("db_4a578da0f27ce9d") + .withConnectTimeout(30, TimeUnit.SECONDS) + .build(); + + client = new MilvusServiceClient(connectParam); + System.out.println("连接目标: " + host + ":" + port); + } + + @AfterAll + static void disconnect() { + if (client != null) { + try { + client.dropCollection(DropCollectionParam.newBuilder() + .withCollectionName(COLLECTION).build()); + } catch (Exception ignored) {} + client.close(); + } + } + + private static String safeMsg(R resp) { + try { + return resp.getMessage(); + } catch (Exception e) { + return "(no message)"; + } + } + + @Test + @Order(1) + @DisplayName("1. 连接成功 - 能列出 collection") + void listCollections() { + R resp = client.showCollections( + ShowCollectionsParam.newBuilder().build()); + + System.out.println("listCollections status: " + resp.getStatus() + ", msg: " + safeMsg(resp)); + assertEquals(0, resp.getStatus(), "连接失败,status=" + resp.getStatus()); + + List names = resp.getData().getCollectionNamesList(); + System.out.println("现有 collections: " + names); + } + + @Test + @Order(2) + @DisplayName("2. 创建测试 collection") + void createCollection() { + client.dropCollection(DropCollectionParam.newBuilder() + .withCollectionName(COLLECTION).build()); + + FieldType idField = FieldType.newBuilder() + .withName("id") + .withDataType(DataType.Int64) + .withPrimaryKey(true) + .withAutoID(true) + .build(); + + FieldType vectorField = FieldType.newBuilder() + .withName("vector") + .withDataType(DataType.FloatVector) + .withDimension(DIM) + .build(); + + CollectionSchemaParam schema = CollectionSchemaParam.newBuilder() + .addFieldType(idField) + .addFieldType(vectorField) + .build(); + + R resp = client.createCollection( + CreateCollectionParam.newBuilder() + .withCollectionName(COLLECTION) + .withSchema(schema) + .build()); + + System.out.println("createCollection status: " + resp.getStatus() + ", msg: " + safeMsg(resp)); + assertEquals(0, resp.getStatus(), "创建 collection 失败"); + } + + @Test + @Order(3) + @DisplayName("3. 插入数据 + flush") + void insertAndFlush() { + List vec1 = makeVector(1.0f); + List vec2 = makeVector(2.0f); + List vec3 = makeVector(3.0f); + + List fields = Collections.singletonList( + new InsertParam.Field("vector", Arrays.asList(vec1, vec2, vec3)) + ); + + R insertResp = client.insert( + InsertParam.newBuilder() + .withCollectionName(COLLECTION) + .withFields(fields) + .build()); + + System.out.println("insert status: " + insertResp.getStatus() + ", msg: " + safeMsg(insertResp)); + assertEquals(0, insertResp.getStatus(), "插入失败"); + + // 官方示例要求:insert 后必须 flush,数据才对搜索可见 + R flushResp = client.flush(FlushParam.newBuilder() + .withCollectionNames(Collections.singletonList(COLLECTION)) + .withSyncFlush(true) + .withSyncFlushWaitingTimeout(30L) + .build()); + + System.out.println("flush status: " + flushResp.getStatus() + ", msg: " + safeMsg(flushResp)); + assertEquals(0, flushResp.getStatus(), "flush 失败"); + System.out.println("插入 3 条数据并 flush 完成"); + } + + @Test + @Order(4) + @DisplayName("4. 创建索引 + 加载") + void createIndexAndLoad() { + R indexResp = client.createIndex( + CreateIndexParam.newBuilder() + .withCollectionName(COLLECTION) + .withFieldName("vector") + .withIndexType(IndexType.AUTOINDEX) + .withMetricType(MetricType.L2) + .build()); + + System.out.println("createIndex status: " + indexResp.getStatus() + ", msg: " + safeMsg(indexResp)); + assertEquals(0, indexResp.getStatus(), "创建索引失败"); + + R loadResp = client.loadCollection( + LoadCollectionParam.newBuilder() + .withCollectionName(COLLECTION) + .withSyncLoad(true) + .withSyncLoadWaitingTimeout(30L) + .build()); + + System.out.println("load status: " + loadResp.getStatus() + ", msg: " + safeMsg(loadResp)); + assertEquals(0, loadResp.getStatus(), "加载失败"); + System.out.println("索引创建 + 加载完成"); + } + + @Test + @Order(5) + @DisplayName("5. 向量搜索") + void search() throws InterruptedException { + Thread.sleep(3000); + + List queryVec = makeVector(1.1f); + + R resp = null; + for (int retry = 0; retry < 10; retry++) { + resp = client.search( + SearchParam.newBuilder() + .withCollectionName(COLLECTION) + .withMetricType(MetricType.L2) + .withTopK(2) + .withVectors(Collections.singletonList(queryVec)) + .withVectorFieldName("vector") + .withParams("{}") + .withConsistencyLevel(ConsistencyLevelEnum.STRONG) + .build()); + + if (resp.getStatus() == 0) break; + System.out.println("search retry " + (retry + 1) + ": status=" + resp.getStatus() + ", msg=" + safeMsg(resp)); + Thread.sleep(5000); + } + + System.out.println("search status: " + resp.getStatus() + ", msg: " + safeMsg(resp)); + assertEquals(0, resp.getStatus(), "搜索失败"); + + SearchResultsWrapper wrapper = new SearchResultsWrapper(resp.getData().getResults()); + List scores = wrapper.getIDScore(0); + + assertFalse(scores.isEmpty(), "搜索结果不应为空"); + System.out.println("搜索结果 (top " + scores.size() + "):"); + for (SearchResultsWrapper.IDScore idScore : scores) { + System.out.println(" score=" + idScore.getScore() + ", id=" + idScore.getLongID()); + } + } + + private static List makeVector(float val) { + Float[] arr = new Float[DIM]; + Arrays.fill(arr, val); + return Arrays.asList(arr); + } + + private static String envOrDefault(String key, String defaultVal) { + String val = System.getenv(key); + return (val != null && !val.isEmpty()) ? val : defaultVal; + } +} \ No newline at end of file