diff --git a/eval/rag-retrieval/README.md b/eval/rag-retrieval/README.md index f85cb02..97e7367 100644 --- a/eval/rag-retrieval/README.md +++ b/eval/rag-retrieval/README.md @@ -15,6 +15,7 @@ eval/rag-retrieval/ fixtures/*.json Saved retrieval candidates for each case reports/baseline.json Machine-readable baseline report reports/baseline.md Human-readable baseline report + reports/live-post-reindex.* Optional live acceptance reports ``` ## Run @@ -49,3 +50,37 @@ python scripts/eval_rag_retrieval.py \ This baseline runs fully offline and does not call MySQL, Redis, Milvus, an LLM, or the Spring Boot application. It is a regression harness for retrieval behavior, not a claim that live production retrieval accuracy is complete. + +## Live Post-Reindex Acceptance + +When embedding input changes, existing vectors do not update by themselves. For +example, after adding `title` and `breadcrumb` to the embedding text, the live +Milvus/Zilliz collection must be reindexed before retrieval can reflect that new +semantic signal. + +Use this optional live acceptance flow after the application is running and the +knowledge base has been reindexed: + +```bash +python scripts/eval_rag_live_acceptance.py +``` + +Custom service URL and output paths are supported: + +```bash +python scripts/eval_rag_live_acceptance.py \ + --base-url http://127.0.0.1:9900 \ + --json-report eval/rag-retrieval/reports/live-post-reindex.json \ + --markdown-report eval/rag-retrieval/reports/live-post-reindex.md +``` + +The script calls: + +```text +GET /api/search/similar +``` + +It writes JSON and Markdown reports with query, topK, result count, top +candidates, breadcrumb, score labels, and raw response fields. This is a live +smoke check for environment readiness and post-reindex behavior; it does not +replace the deterministic offline baseline above. diff --git a/interview/rag-breadcrumb-embedding-acceptance.md b/interview/rag-breadcrumb-embedding-acceptance.md new file mode 100644 index 0000000..624b933 --- /dev/null +++ b/interview/rag-breadcrumb-embedding-acceptance.md @@ -0,0 +1,66 @@ +# RAG Breadcrumb Embedding Acceptance + +## What Changed + +The indexing path now builds embedding text from chunk structure plus content: + +```text +Title: {title} +Path: {breadcrumb} +Content: +{content} +``` + +The stored Milvus `content` field remains the original chunk content. This keeps display and evidence output clean while allowing the vector to carry section-level semantics. + +## Why Reindex Is Required + +Embeddings are materialized at index time. Existing vectors were generated from the previous content-only text, so they cannot benefit from `title` and `breadcrumb` until the knowledge base is reindexed. + +This is the key acceptance point: + +```text +code change alone != live retrieval changed +code change + reindex + live query report = accepted behavior +``` + +## How To Validate + +1. Start the Spring Boot application. +2. Reindex the knowledge base through the existing indexing path. +3. Run: + +```bash +python scripts/eval_rag_live_acceptance.py +``` + +The script writes: + +```text +eval/rag-retrieval/reports/live-post-reindex.json +eval/rag-retrieval/reports/live-post-reindex.md +``` + +The default cases cover: + +- RAG chunk context questions where breadcrumb matters. +- Diagnosis flow questions where section path matters. +- `ERR_TIMEOUT` exact error-code retrieval. +- MySQL connection pool troubleshooting. +- AIOps payment-service latency alert retrieval. + +## What To Look For + +For breadcrumb-sensitive cases, inspect whether top candidates expose expected `title` and `breadcrumb` values in the report. + +For core troubleshooting cases, check that result counts and top candidates remain stable. The goal is not to prove a full benchmark; it is to prove that reindexing did not obviously break important demo retrieval paths. + +## Interview Answer + +If asked how I verified the breadcrumb embedding change: + +> I separated deterministic regression from live acceptance. The offline fixture baseline still runs without services. But because embedding changes only affect newly indexed vectors, I added a live post-reindex acceptance script. It calls the real `/api/search/similar` endpoint against representative breadcrumb-sensitive, troubleshooting, and AIOps queries, then writes JSON and Markdown reports. This lets me prove both that the code changed and that the live vector collection was refreshed. + +If asked why the script does not reindex automatically: + +> Reindexing mutates the vector store and depends on environment-specific data. I kept mutation explicit and made the script validation-only. That makes failures easier to diagnose: if retrieval does not improve, I can distinguish code changes, reindex state, and runtime retrieval behavior. diff --git a/openspec/changes/rag-breadcrumb-embedding-acceptance/.openspec.yaml b/openspec/changes/rag-breadcrumb-embedding-acceptance/.openspec.yaml new file mode 100644 index 0000000..e089cfa --- /dev/null +++ b/openspec/changes/rag-breadcrumb-embedding-acceptance/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-05 diff --git a/openspec/changes/rag-breadcrumb-embedding-acceptance/design.md b/openspec/changes/rag-breadcrumb-embedding-acceptance/design.md new file mode 100644 index 0000000..52a8175 --- /dev/null +++ b/openspec/changes/rag-breadcrumb-embedding-acceptance/design.md @@ -0,0 +1,72 @@ +## Context + +The indexing path now builds embeddings from structured text: + +```text +Title: {title} +Path: {breadcrumb} +Content: +{content} +``` + +The persisted Milvus `content` field remains the raw chunk content. This improves semantic recall for section-aware questions, but only after documents are reindexed. Existing vectors were generated from the previous content-only input and cannot reflect the new breadcrumb signal. + +The repository already has an offline fixture-based retrieval baseline. That baseline is useful for deterministic regression checks, but it does not prove that the live Milvus/Zilliz collection has been reindexed or that the running service returns breadcrumb-aware results. + +## Goals / Non-Goals + +**Goals:** + +- Provide an explicit post-reindex live acceptance flow. +- Make the reindex prerequisite visible in documentation. +- Add a small script that calls the live retrieval endpoint with representative queries and writes reviewable reports. +- Keep the live flow optional so unit tests and offline evaluation remain service-free. + +**Non-Goals:** + +- Do not add a new reindex API in this change. +- Do not automatically mutate live Milvus/Zilliz data from the acceptance script. +- Do not change `lookup_knowledge`, VectorStore retrieval, or Milvus schema. +- Do not commit environment-specific live results unless they were intentionally captured for interview evidence. + +## Decisions + +### Decision 1: Keep Reindex Manual And Explicit + +The acceptance flow documents that reindexing must happen before live validation, but it does not perform the reindex itself. + +Rationale: + +- Reindexing is a data mutation and can be slow or environment-specific. +- The existing project already has indexing paths through upload, document management, and knowledge-base initialization. +- Keeping mutation separate from validation makes failures easier to diagnose. + +Alternative considered: add a script that triggers reindex and then validates. This was rejected for now because it would need environment-specific credentials, source selection, and safety controls. + +### Decision 2: Use HTTP Endpoint Validation + +The script calls `/api/search/similar` instead of invoking Java services directly. + +Rationale: + +- It validates the same runtime path used in demos. +- It works across SDK, Spring AI, and auto retrieval modes. +- It produces a simple artifact that can be shown in interview material. + +Alternative considered: add a Java integration test. This was rejected because live Milvus and Spring Boot availability should remain optional. + +### Decision 3: Preserve Offline Baseline Separately + +The existing fixture-based evaluator remains the deterministic baseline. The new live acceptance flow is a smoke/regression companion, not a replacement. + +Rationale: + +- Offline reports are stable and CI-friendly. +- Live reports prove environment readiness and post-reindex behavior. +- Keeping both avoids mixing deterministic fixture checks with external-service validation. + +## Risks / Trade-offs + +- [Risk] Live results vary by environment, indexed documents, and retrieval mode. -> Mitigation: report the base URL, query set, result count, top candidates, score labels, and timestamp. +- [Risk] A developer may run live validation before reindexing. -> Mitigation: document the prerequisite clearly and include a report note. +- [Risk] The script could be mistaken for a benchmark. -> Mitigation: position it as acceptance smoke coverage; keep offline baseline for deterministic metrics. diff --git a/openspec/changes/rag-breadcrumb-embedding-acceptance/proposal.md b/openspec/changes/rag-breadcrumb-embedding-acceptance/proposal.md new file mode 100644 index 0000000..020e669 --- /dev/null +++ b/openspec/changes/rag-breadcrumb-embedding-acceptance/proposal.md @@ -0,0 +1,26 @@ +## Why + +`title` and `breadcrumb` now participate in embedding text, but that improvement only affects newly indexed vectors. We need a repeatable acceptance path that tells us how to reindex the knowledge base and verify live retrieval after the embedding input changes. + +## What Changes + +- Add a live RAG retrieval acceptance flow for breadcrumb-aware embedding changes. +- Document the reindex prerequisite so reviewers understand old vectors do not change automatically. +- Provide a small repeatable script for calling live retrieval cases and writing JSON/Markdown reports. +- Add interview-facing acceptance notes that explain what was verified and what remains manual or environment-dependent. + +## Capabilities + +### New Capabilities + +None. + +### Modified Capabilities + +- `rag-retrieval-evaluation`: Extend retrieval evaluation with an opt-in live acceptance flow for post-reindex verification. + +## Impact + +- Adds scripts and documentation under the retrieval evaluation/interview areas. +- Does not change the Agent runtime path, `lookup_knowledge`, VectorStore search logic, or Milvus schema. +- Live verification depends on a running Spring Boot service and a reindexed Milvus/Zilliz collection. diff --git a/openspec/changes/rag-breadcrumb-embedding-acceptance/specs/rag-retrieval-evaluation/spec.md b/openspec/changes/rag-breadcrumb-embedding-acceptance/specs/rag-retrieval-evaluation/spec.md new file mode 100644 index 0000000..238495a --- /dev/null +++ b/openspec/changes/rag-breadcrumb-embedding-acceptance/specs/rag-retrieval-evaluation/spec.md @@ -0,0 +1,21 @@ +## ADDED Requirements + +### Requirement: Retrieval evaluation SHALL provide live post-reindex acceptance +The retrieval evaluation system SHALL provide an opt-in live acceptance flow for validating retrieval behavior after embedding input changes require a knowledge-base reindex. + +#### Scenario: Live acceptance requires a running service +- **WHEN** live retrieval acceptance is run +- **THEN** it SHALL call the configured Spring Boot retrieval endpoint +- **AND** it SHALL not be required by the offline fixture baseline + +#### Scenario: Live acceptance records retrieval evidence +- **WHEN** a live retrieval case is executed +- **THEN** the report SHALL include the query, requested topK, result count, top candidate titles or sources, score labels, and raw response fields needed for review + +#### Scenario: Reindex prerequisite is documented +- **WHEN** a developer prepares to validate breadcrumb-aware embedding behavior +- **THEN** the repository SHALL explain that existing vectors must be reindexed before live validation can reflect the new embedding text + +#### Scenario: Live report is reviewable +- **WHEN** the live acceptance script completes +- **THEN** it SHALL write JSON and Markdown outputs that can be inspected or attached to interview evidence diff --git a/openspec/changes/rag-breadcrumb-embedding-acceptance/tasks.md b/openspec/changes/rag-breadcrumb-embedding-acceptance/tasks.md new file mode 100644 index 0000000..02499ae --- /dev/null +++ b/openspec/changes/rag-breadcrumb-embedding-acceptance/tasks.md @@ -0,0 +1,14 @@ +## 1. Live Acceptance Tooling + +- [x] 1.1 Add a script that runs representative live `/api/search/similar` queries and writes JSON/Markdown reports. +- [x] 1.2 Include breadcrumb-sensitive and core troubleshooting cases in the default live query set. + +## 2. Documentation + +- [x] 2.1 Document the post-reindex validation flow under `eval/rag-retrieval`. +- [x] 2.2 Add interview-facing acceptance notes for breadcrumb-aware embedding validation. + +## 3. Verification + +- [x] 3.1 Run targeted tests or syntax checks for the new script. +- [x] 3.2 Validate the OpenSpec change and confirm the working tree only contains expected files. diff --git a/scripts/eval_rag_live_acceptance.py b/scripts/eval_rag_live_acceptance.py new file mode 100644 index 0000000..04d296e --- /dev/null +++ b/scripts/eval_rag_live_acceptance.py @@ -0,0 +1,292 @@ +#!/usr/bin/env python3 +"""Live acceptance runner for post-reindex RAG retrieval checks. + +This script calls the running Spring Boot retrieval endpoint. It is intentionally +separate from the offline fixture baseline because it depends on live service and +Milvus/Zilliz state. +""" + +from __future__ import annotations + +import argparse +import json +import sys +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + + +DEFAULT_BASE_URL = "http://127.0.0.1:9900" +DEFAULT_JSON_REPORT = Path("eval/rag-retrieval/reports/live-post-reindex.json") +DEFAULT_MD_REPORT = Path("eval/rag-retrieval/reports/live-post-reindex.md") + + +DEFAULT_CASES: list[dict[str, Any]] = [ + { + "caseId": "breadcrumb-rag-chunk-context", + "query": "If a long RAG section is split into multiple chunks, how do we keep retrieval context?", + "topK": 5, + "purpose": "Breadcrumb-sensitive RAG chunk context retrieval.", + }, + { + "caseId": "breadcrumb-diagnosis-flow", + "query": "What is the standard troubleshooting flow for an application incident?", + "topK": 5, + "purpose": "Process-style retrieval where section path matters.", + }, + { + "caseId": "core-err-timeout", + "query": "ERR_TIMEOUT", + "topK": 3, + "purpose": "Exact error-code retrieval should remain stable.", + }, + { + "caseId": "core-mysql-connection-pool", + "query": "MySQL connection pool is exhausted. How should I diagnose it?", + "topK": 3, + "purpose": "Core infrastructure troubleshooting retrieval.", + }, + { + "caseId": "aiops-payment-latency", + "query": "Alert HighLatency on payment-service with p95 latency above threshold", + "topK": 3, + "purpose": "AIOps alert-style retrieval.", + }, +] + + +@dataclass +class LiveCase: + case_id: str + query: str + top_k: int + purpose: str + category: str | None = None + + @classmethod + def from_json(cls, raw: dict[str, Any]) -> "LiveCase": + return cls( + case_id=str(raw["caseId"]), + query=str(raw["query"]), + top_k=int(raw.get("topK") or 3), + purpose=str(raw.get("purpose") or raw.get("notes") or ""), + category=( + str(raw.get("category")) + if raw.get("category") not in (None, "") + else None + ), + ) + + +def load_cases(path: Path | None) -> list[LiveCase]: + if path is None: + return [LiveCase.from_json(item) for item in DEFAULT_CASES] + with path.open("r", encoding="utf-8") as handle: + payload = json.load(handle) + raw_cases = payload.get("cases", payload) + return [LiveCase.from_json(item) for item in raw_cases] + + +def write_json(path: Path, payload: Any) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="\n") as handle: + json.dump(payload, handle, ensure_ascii=False, indent=2) + handle.write("\n") + + +def write_text(path: Path, content: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="\n") as handle: + handle.write(content) + + +def request_case(base_url: str, case: LiveCase, timeout_seconds: float) -> dict[str, Any]: + endpoint = base_url.rstrip("/") + "/api/search/similar" + params: dict[str, str] = { + "query": case.query, + "topK": str(case.top_k), + } + if case.category: + params["category"] = case.category + url = endpoint + "?" + urllib.parse.urlencode(params) + + started_at = datetime.now(timezone.utc) + try: + with urllib.request.urlopen(url, timeout=timeout_seconds) as response: + body = response.read().decode("utf-8") + payload = json.loads(body) + status = int(getattr(response, "status", 200)) + except (urllib.error.URLError, TimeoutError, json.JSONDecodeError) as exc: + return { + "caseId": case.case_id, + "query": case.query, + "topK": case.top_k, + "category": case.category, + "purpose": case.purpose, + "url": url, + "ok": False, + "error": str(exc), + "resultCount": 0, + "topCandidates": [], + "rawResponse": None, + "startedAt": started_at.isoformat(), + } + + data = payload.get("data") if isinstance(payload, dict) else None + if not isinstance(data, list): + data = [] + + ok = status == 200 and payload.get("code") == 200 + return { + "caseId": case.case_id, + "query": case.query, + "topK": case.top_k, + "category": case.category, + "purpose": case.purpose, + "url": url, + "ok": ok, + "httpStatus": status, + "responseCode": payload.get("code"), + "responseMessage": payload.get("message"), + "resultCount": len(data), + "topCandidates": [summarize_candidate(item, index + 1) for index, item in enumerate(data)], + "rawResponse": payload, + "startedAt": started_at.isoformat(), + } + + +def summarize_candidate(raw: dict[str, Any], rank: int) -> dict[str, Any]: + metadata = parse_metadata(raw.get("metadata")) + return { + "rank": rank, + "id": raw.get("id"), + "title": metadata.get("title"), + "breadcrumb": metadata.get("breadcrumb"), + "category": metadata.get("category"), + "source": metadata.get("_source") or metadata.get("source"), + "score": raw.get("score"), + "rawScore": raw.get("rawScore"), + "scoreLabel": raw.get("scoreLabel"), + "contentPreview": preview(raw.get("content")), + } + + +def parse_metadata(value: Any) -> dict[str, Any]: + if isinstance(value, dict): + return value + if isinstance(value, str) and value.strip(): + try: + parsed = json.loads(value) + return parsed if isinstance(parsed, dict) else {} + except json.JSONDecodeError: + return {} + return {} + + +def preview(value: Any, limit: int = 180) -> str: + text = " ".join(str(value or "").split()) + if len(text) <= limit: + return text + return text[: limit - 3] + "..." + + +def render_markdown(report: dict[str, Any]) -> str: + lines = [ + "# RAG Live Post-Reindex Acceptance", + "", + f"Generated at: `{report['generatedAt']}`", + f"Base URL: `{report['baseUrl']}`", + "", + "> Reindex prerequisite: this report only reflects breadcrumb-aware embedding if the knowledge base was reindexed after the embedding-text change.", + "", + "## Summary", + "", + "| Metric | Value |", + "|---|---:|", + f"| Cases | {report['caseCount']} |", + f"| Successful calls | {report['successfulCalls']} |", + f"| Empty result cases | {report['emptyResultCases']} |", + "", + "## Cases", + "", + "| Case | Purpose | Results | Top Candidates |", + "|---|---|---:|---|", + ] + for item in report["results"]: + top = "
".join(format_candidate(candidate) for candidate in item["topCandidates"]) + if not top and item.get("error"): + top = "ERROR: " + str(item["error"]) + lines.append( + "| {case} | {purpose} | {count} | {top} |".format( + case=item["caseId"], + purpose=item.get("purpose") or "", + count=item["resultCount"], + top=top, + ) + ) + lines.append("") + return "\n".join(lines) + + +def format_candidate(candidate: dict[str, Any]) -> str: + label = candidate.get("title") or candidate.get("source") or candidate.get("id") or "" + breadcrumb = candidate.get("breadcrumb") or "" + score_label = candidate.get("scoreLabel") or "" + score = candidate.get("score") + raw_score = candidate.get("rawScore") + details = f"score={score}" + if raw_score is not None: + details += f", raw={raw_score}" + if score_label: + details += f", label={score_label}" + if breadcrumb: + return f"{candidate['rank']}. {label} ({breadcrumb}; {details})" + return f"{candidate['rank']}. {label} ({details})" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--base-url", default=DEFAULT_BASE_URL) + parser.add_argument("--cases", type=Path, default=None) + parser.add_argument("--json-report", type=Path, default=DEFAULT_JSON_REPORT) + parser.add_argument("--markdown-report", type=Path, default=DEFAULT_MD_REPORT) + parser.add_argument("--timeout-seconds", type=float, default=10.0) + return parser.parse_args() + + +def main() -> int: + args = parse_args() + cases = load_cases(args.cases) + results = [ + request_case(args.base_url, case, args.timeout_seconds) + for case in cases + ] + successful = [item for item in results if item["ok"]] + empty = [item for item in results if item["ok"] and item["resultCount"] == 0] + report = { + "generatedAt": datetime.now(timezone.utc).isoformat(), + "baseUrl": args.base_url, + "caseCount": len(results), + "successfulCalls": len(successful), + "emptyResultCases": len(empty), + "reindexPrerequisite": "Run or trigger knowledge-base reindex before treating this as breadcrumb-aware embedding evidence.", + "results": results, + } + write_json(args.json_report, report) + write_text(args.markdown_report, render_markdown(report)) + print( + "Ran {total} live cases: successful={successful}, empty={empty}".format( + total=len(results), + successful=len(successful), + empty=len(empty), + ) + ) + return 1 if len(successful) != len(results) else 0 + + +if __name__ == "__main__": + raise SystemExit(main())