From 190013c901cb4f3bb1790c103261b5a7d3ff287c Mon Sep 17 00:00:00 2001 From: zhuyongxin Date: Mon, 20 Jul 2026 10:23:27 +0800 Subject: [PATCH] feat(graph): complete stategraph cleanup and acceptance --- devflow/index.md | 1 + .../acceptance.md | 49 ++ .../brief.md | 32 + .../decisions.md | 159 +++++ .../evidence.md | 38 ++ mvp/architecture/README.md | 8 +- mvp/architecture/agent-orchestration.md | 55 +- mvp/architecture/current-mvp-architecture.md | 29 +- .../executor-evidence-pipeline-refactor.md | 36 +- mvp/architecture/feedback-architecture.md | 23 +- mvp/architecture/harness-quality-gates.md | 25 +- mvp/architecture/retrieval-observability.md | 8 +- mvp/architecture/session-trace-lifecycle.md | 17 +- mvp/demo/README.md | 11 +- mvp/demo/scripts/run-interview-demo-check.ps1 | 67 ++- mvp/demo/trace-inspection-checklist.md | 39 +- mvp/eval/README.md | 12 +- mvp/issues/README.md | 4 +- ...chat-diagnosis-stategraph-orchestration.md | 111 ++-- .../.archive-ready | 1 + .../.committed | 3 + .../design.md | 127 ++++ .../proposal.md | 69 +++ .../spec.md | 88 +++ .../specs/mvp-demo-trace-acceptance/spec.md | 39 ++ .../tasks.md | 52 ++ .../spec.md | 91 +++ .../specs/mvp-demo-trace-acceptance/spec.md | 24 +- .../AbstractDiagnosisAgentNodeAdapter.java | 3 +- .../DiagnosisOrchestrationTraceBuilder.java | 51 +- .../diagnosis/EvidenceRetryPrepareNode.java | 2 +- .../agent/graph/diagnosis/FallbackNode.java | 2 +- .../agent/graph/diagnosis/GatekeeperNode.java | 2 +- .../diagnosis/ReactAgentDiagnosisInvoker.java | 19 +- .../graph/diagnosis/VerifiedInputNode.java | 2 +- .../agent/hook/VerifierInputHook.java | 174 ------ .../service/ToolTraceSummaryService.java | 546 ------------------ .../agent/util/VerifierContextHolder.java | 87 --- .../DiagnosisGraphNodeContractTest.java | 8 +- .../diagnosis/DiagnosisGraphWorkflowTest.java | 5 +- ...iagnosisOrchestrationTraceBuilderTest.java | 21 + .../diagnosis/ExecutorNodeAdapterTest.java | 4 +- .../InterviewDemoScriptContractTest.java | 47 ++ .../ReactAgentDiagnosisInvokerTest.java | 32 +- .../ScriptedDiagnosisGraphActions.java | 2 +- .../service/ToolTraceSummaryServiceTest.java | 209 ------- 46 files changed, 1223 insertions(+), 1211 deletions(-) create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/acceptance.md create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/brief.md create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/decisions.md create mode 100644 devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/evidence.md rename mvp/issues/{active => archived}/ISS-011-chat-diagnosis-stategraph-orchestration.md (93%) create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.archive-ready create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.committed create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/design.md create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/proposal.md create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/mvp-demo-trace-acceptance/spec.md create mode 100644 openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/tasks.md create mode 100644 openspec/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md delete mode 100644 src/main/java/com/superbiz/agent/hook/VerifierInputHook.java delete mode 100644 src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java delete mode 100644 src/main/java/com/superbiz/agent/util/VerifierContextHolder.java create mode 100644 src/test/java/com/superbiz/agent/graph/diagnosis/InterviewDemoScriptContractTest.java delete mode 100644 src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java diff --git a/devflow/index.md b/devflow/index.md index dd58c21..e0694cb 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -4,6 +4,7 @@ | 日期 | slug | 说明 | 领域 | 关键词 | 关联 OpenSpec | 状态 | |---|---|---|---|---|---|---| +| 2026-07-17 | chat-diagnosis-stategraph-cleanup-docs | 清理旧诊断编排闭包,对齐当前文档与 demo contract,并完成 ISS-011 最终 live、日志和数据库验收。 | Chat diagnosis orchestration/cleanup | legacy closure, current docs, orchestration trace, Maven E2E, MySQL ownership | openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs | archived | | 2026-07-17 | chat-diagnosis-stategraph-test-suite | 建立 Workflow、Node Contract、Chat Integration 三层权威测试体系并退役旧 Hook implementation tests。 | Chat diagnosis orchestration/testing | workflow test, node contract, Chat integration, coverage matrix, Hook test retirement | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-test-suite | archived | | 2026-07-17 | chat-diagnosis-stategraph-chatservice-cutover | 将复杂 Chat 单轨切换到 Diagnosis StateGraph,并增加 Run 级 orchestration trace 和 verified-only Verifier 输入。 | Chat diagnosis orchestration/production cutover | ChatService, CompiledGraph stream, runId metadata, orchestration trace, verified-only prompt, V012 | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-chatservice-cutover | archived | | 2026-07-17 | chat-diagnosis-stategraph-real-nodes | 接入真实 Agent/Java Nodes、显式 Gatekeeper、可信输入投影、关键证据补查与安全 Fallback,暂不切换生产入口。 | Chat diagnosis orchestration/nodes | ReactAgent adapter, Gatekeeper node, verified input, evidence retry, safe fallback, CompiledGraph | openspec/changes/archive/2026-07-17-chat-diagnosis-stategraph-real-nodes | archived | diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/acceptance.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/acceptance.md new file mode 100644 index 0000000..9c70876 --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/acceptance.md @@ -0,0 +1,49 @@ +# Acceptance + +## 静态验证 + +- 旧闭包路径不存在,executable legacy refs=0。 +- current-doc stale architecture refs=0;历史兼容引用有明确限定。 +- `git diff --check` 通过;无新增 migration/schema 变更;临时调试标记=0。 +- 当前 change strict 和 16 个 main specs strict 全部通过。 + +## 脚本验证 + +- `mvn -q -DskipTests test-compile`:通过。 +- 39-suite authoritative/focused Maven command:157 tests,0 failure/error/skipped。 +- 归档前 `mvn clean` + test compilation + 43-suite deterministic command:189 tests,0 failure/error/skipped。 +- `DiagnosisTraceEvaluatorTest` + `DiagnosisEvalBaselineDiffTest`:12/12 baseline,same diff=0。 +- PowerShell parser + `InterviewDemoScriptContractTest`:通过。 +- `run-interview-demo-check.ps1 -SessionId iss-011-stage5-20260720015557 -OutputDir target/iss-011-stage5-output-current`:exit 0。 +- `scripts/query_mysql.py` exact queries:V012、Run JSON、AgentStep/ToolInvocation ownership 全部通过。 +- `openspec validate --all --strict --no-interactive`:17/17 passed;`git diff --check`、current-doc、schema、debug source、port/temp scope checks 通过。 + +## Live E2E + +| 项目 | 结果 | +|---|---| +| Maven profile | `mvp-demo` | +| sessionId | `iss-011-stage5-20260720015557` | +| runId | `run-808ac38f-3ad0-4462-a6d0-ed50d8686473` | +| Run | `CHAT/SUCCESS` | +| answer / metrics | 109 chars / 75964ms / 111802 tokens / 8 steps / 12 tools | +| Graph | `stategraph-v1`, `fallback`, `fallback_completed`, degraded=true, 3 transitions, retry=0 | +| evaluation / feedback | non-empty / useful | +| new ERROR | 0 | +| DB ownership | wrong owner=0,wrong-session rows=0 | +| process cleanup | owned PIDs stopped,9900 released | + +## 浏览器/人工验证 + +- 不适用。本阶段验收入口是 API/PowerShell executable contract,无 UI 改动。 + +## 未验证 + +- 无 OpenSpec 必需项未验证。 + +## 归档状态 + +- ISS-011 已归档至 `mvp/issues/archived/ISS-011-chat-diagnosis-stategraph-orchestration.md`。 +- OpenSpec 归档路径:`openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs`。 +- OpenSpec CLI 已同步主 specs:新增 `chat-diagnosis-stategraph-cleanup-docs`,更新 `mvp-demo-trace-acceptance` 3 项 requirement。 +- 不 push。 diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/brief.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/brief.md new file mode 100644 index 0000000..7633cf3 --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/brief.md @@ -0,0 +1,32 @@ +# Chat Diagnosis StateGraph Cleanup And Final Acceptance + +## 背景 + +ISS-011 阶段 0-4 已完成 StateGraph 设计冻结、路由骨架、真实 Nodes、ChatService 单轨切换和三层权威测试。阶段 5 负责删除旧 Hook/ThreadLocal/full-trace service 闭包、对齐当前文档与 demo executable contract,并以唯一 Run 完成最终 Maven、日志和数据库验收。 + +## 目标 + +- 只保留 bounded StateGraph 复杂 Chat 编排和 verified-only Verifier 输入。 +- 让 current architecture/eval/demo 文档与 Run-owned orchestration trace 一致。 +- 先通过确定性回归,再用 Maven `mvp-demo` 证明 exact Chat/Trace/feedback、日志和数据库 ownership。 +- 所有门禁通过后关闭 ISS-011,并归档阶段 5 OpenSpec。 + +## 范围 + +- 删除 `VerifierInputHook`、`VerifierContextHolder`、`ToolTraceSummaryService` 及其 focused test。 +- 更新 current architecture/eval/demo 文档和 interview demo check。 +- 修复 live 暴露的 Graph event classloader 边界与 nested ReactAgent resume config 问题。 +- 完成 Graph/Chat/Trace/Eval 回归、Maven live E2E、日志/MySQL 核验、进程清理和 Issue 生命周期收口。 + +## 非目标 + +- 不修改公开 Chat/feedback API、Executor/Verifier/Composer 业务协议或数据库 schema。 +- 不重写 archived issues、历史 design notes 和 legacy fixtures。 +- 不删除旧 Trace/fixture 对 `tool_trace_summary` 的只读兼容。 +- 不 push,不删除失败尝试的审计数据。 + +## 元数据 + +- 分档:complex +- OpenSpec:`chat-diagnosis-stategraph-cleanup-docs` +- 接口影响:L2 内部类型/状态表示修复;外部 API/DTO/schema 不变 diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/decisions.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/decisions.md new file mode 100644 index 0000000..1973214 --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/decisions.md @@ -0,0 +1,159 @@ +# Chat Diagnosis StateGraph Cleanup, Final Acceptance And Documentation Decisions + +## Entry Summary + +- 问题:ISS-011 运行时已切换且测试体系已收敛,但旧 Hook/ThreadLocal/service死代码、当前架构文档和最终 live 证据尚未闭环。 +- 期望:阶段 5完成清理、文档、自动化回归、eval、Maven E2E、日志/DB 验收、Issue 归档和独立提交。 +- 分档:complex;接口影响 L2 内部删除 + 文档/demo 验收增强,外部 API/DB 协议不变。 +- Change:`chat-diagnosis-stategraph-cleanup-docs`。 +- 授权:用户已明确要求直接实现;本阶段按此前规则执行唯一最终 E2E。 + +## Context Sources + +- ISS-011 阶段 5、测试策略、协议影响、验收标准和冻结决策。 +- 阶段 0–4 OpenSpec archives、devflow acceptance 与提交 `581daff`、`42ba204`、`1460dd1`、`99e490f`、`208a231`。 +- 全仓 `VerifierInputHook`/`VerifierContextHolder`/`ToolTraceSummaryService` 定义与引用搜索。 +- `mvp/architecture/README.md` 列出的 current docs、`mvp/eval/README.md`、`mvp/demo/` scripts/checklist。 +- `mvp-demo` profile、payment-timeout request、`scripts/query_mysql.py` 和 logs/ 现有布局。 + +## Question Pool + +| # | 维度 | 问题 | 模式 | 状态 | +|---|---|---|---|---| +| Q1 | 清理 | 旧 Hook/ThreadLocal/trace summary service 是否还有生产消费者? | evidence-driven | 已解决 | +| Q2 | 文档 | 哪些旧引用应更新,哪些历史材料应保留? | evidence-driven | 已解决 | +| Q3 | E2E | 最终 live 场景如何绑定唯一 session/run 并证明 Graph 路径? | evidence-driven | 已解决 | +| Q4 | 日志/DB | 如何避免用旧日志/latest DB 记录冒充当前证据? | evidence-driven | 已解决 | +| Q5 | 验收 | 何时允许启动 Maven、是否需要日志和 DB 查询? | user-interview(用户最新规则) | 已确认 | +| Q6 | 关闭 | 何时把 ISS-011 从 active 移到 archived? | evidence-driven | 已解决 | + +## Evidence-driven Findings + +- Q1:旧闭包只有 `VerifierInputHook -> VerifierContextHolder + ToolTraceSummaryService`,以及 `ToolTraceSummaryServiceTest`;ChatService/Graph/Trace/Eval 均无引用,可整体删除。 +- 实现前规格校正:`ChatVerifierPromptContractTest` 和 `DiagnosisGraphTestSuiteStructureTest` 必须保留旧类型名称的负向字符串断言;这不构成 executable reference。OpenSpec 已收紧为无定义/import/实例化/type-use,允许负向 guard literal。 +- Q2:current architecture index 仍列 `agent-orchestration.md` 等为当前真理源,因此必须更新;`mvp/issues/design-notes`、archived issues、历史 eval fixtures 保留时间点/兼容语义,不做大规模重写。 +- Q3:`run-interview-demo-check.ps1` 已用 Chat response runId 查询 exact Trace/feedback,最适合扩展 `run.orchestrationTrace` fail-fast 和 summary,不另建重复脚本。 +- Q4:E2E 使用唯一 timestamp sessionId;日志记录启动前 byte/time 边界并按 session/run 搜索;DB 所有核心查询带 exact sessionId/runId,另查询错误 ownership count。 +- Q6:只有实现、回归、eval、live E2E、日志、DB 和 OpenSpec门禁全部通过后,Issue checkbox 才可完成并移动到 archived。 + +## User-interview Confirmation + +| 问题 | 用户原话 | 状态 | OpenSpec 回写 | +|---|---|---|---| +| Q5 最终验收节奏 | “端到端只在最后阶段全部完成后才验证……日志在log文件夹,项目库有查询数据库的py工具” | 已确认 | proposal | + +## Grill-with-docs Result + +- Session/Run/Trace 术语保持不变;新增强调 `orchestration_trace` 是 Run 路由摘要,不属于 self-evaluation 或日志。 +- StateGraph、Workflow/Node Contract/Chat Integration 属于实现/测试架构术语,不修改业务 glossary。 +- 当前文档必须使用 explicit Gatekeeper Node、verified-only Verifier 和 bounded evidence retry;历史设计笔记仍可描述当时 Hook 架构。 +- 删除旧闭包是阶段 0 已冻结单轨迁移的自然收尾,不形成新的难逆转权衡,无需 ADR。 + +## Discover Status + +- `devflow/index.md`:命中阶段 0–4 archives。 +- 接口影响:L2 内部类型删除;外部 API/DTO/DB/Prompt/状态语义无变化。 +- E2E 入口/脚本/日志/DB 工具已定位;真实执行留到 Apply 最后。 +- 未解决问题:0。 +- Draft 产物:proposal + decisions;尚未生成 design/spec/tasks,尚未删除代码或启动应用。 + +## Architecture Audit + +### Module and evidence map + +`ChatController -> ChatService -> ChatDiagnosisGraphRuntime -> DiagnosisRealGraphActionsFactory -> explicit Nodes -> DiagnosisGraphResultMapper -> DiagnosisRun/Trace` 是唯一当前 Chat链。旧 `VerifierInputHook -> ToolTraceSummaryService/VerifierContextHolder` 已从主链断开,删除不改变输入/输出或持久化。阶段 5新增的 demo script assertion只消费 exact Trace `run.orchestrationTrace`,DB/log检查是验收消费者,不成为运行时业务依赖。 + +| 模块 | 所有权 | 阶段 5动作 | +|---|---|---| +| Graph/Chat runtime | 路由、Node、Run 生命周期 | 不改行为,仅回归 | +| Legacy Hook closure | 旧 Sequential Verifier payload | 整体删除 | +| Current architecture docs | 当前实现真理源 | 更新 StateGraph/verified-only/trace | +| Historical docs/fixtures | 时间点/兼容记录 | 保留,不冒充当前实现 | +| Demo check | live Chat/Trace/feedback executable contract | 增加 exact orchestration trace fail-fast | +| logs/MySQL | live运行证据 | 只读本次 session/run | +| ISS/OpenSpec/devflow | 生命周期与交接 | 所有门禁通过后归档 | + +### Lifecycle and failure ownership + +- 自动化门禁失败:不启动 live Maven,修复代码/测试/规格后重跑。 +- live startup失败:应用未 ready,不执行 demo/DB成功声明,先读启动输出和新日志诊断。 +- Chat/Trace/feedback失败:保留 exact response/run证据,ISS保持 active。 +- log ERROR:逐条分类;未解释 ERROR阻塞验收。 +- DB不一致:以 exact run为准,不能用 API成功掩盖 persistence偏差。 +- finally:无论成功失败都停止本轮进程并确认端口,不扩大到未知已有进程。 + +### Consumer and compatibility audit + +- 外部 API/DTO/DB consumer无迁移;demo summary仅加字段。 +- Trace UI/eval 对历史 `tool_trace_summary` 的读取保留,旧 fixture不批量迁移。 +- current docs消费者将看到新 StateGraph架构;历史链接仍可追溯 old Hook设计。 +- Issue move只改变文档位置/index,代码/运行时不依赖该路径。 + +### Cross-artifact alignment + +| 上游 → 下游 | 检查内容 | 状态 | +|---|---|---| +| brief/proposal → proposal | cleanup、current docs、demo、regression、live/log/DB、Issue closure | 已对齐 | +| proposal → design | 删除闭包、current/history边界、顺序、identity、日志/DB、cleanup | 已对齐 | +| design → specs/tasks | 负向 literal例外、E2E字段、exact evidence、进程清理、Issue gate | 已对齐 | +| specs → tasks | 每条 requirement有可执行 cleanup/docs/test/live/log/DB/closure slice | 已对齐 | + +### Audit result + +审计确认阶段 5不需要新运行时抽象或 DB migration;主要风险来自外部 live状态和证据归属,已通过 unique session/run、log boundary、exact DB queries和process ownership缓解。规格误把负向名称 literal 当 executable reference 的 gap 已修正。接口影响 L2,cross-artifact gap=0,无新 ADR。 + +## Commit Gate + +- schema:spec-driven;proposal/design/2 delta specs/tasks 全部 done,applyRequires=`tasks` 已满足。 +- OpenSpec:当前 change strict pass;16 个主 specs strict pass。 +- Cross-artifact:4/4 已对齐,gap=0;负向 guard literal例外已写入 proposal/design/spec/tasks。 +- Question pool:5 个 evidence-driven 已解决,1 个 user-interview 已由用户原话确认,无未决项。 +- Interface impact:L2 internal type removal + demo/docs enhancement;外部协议/DB无变化。 +- Preflight:`git diff --check` 通过;尚未删除代码、修改 current docs/script或启动应用。 +- 结论:Draft OpenSpec 达到可执行状态,创建 `.committed` 后进入 Apply。 + +## Apply Progress + +### Legacy closure removal + +- 已删除 `VerifierInputHook`、`VerifierContextHolder`、`ToolTraceSummaryService` 和 `ToolTraceSummaryServiceTest`,四个路径均不存在。 +- `rg` 对 `src/main`、`src/test` 的旧类型扫描仅命中 `ChatVerifierPromptContractTest` 和 `DiagnosisGraphTestSuiteStructureTest` 中的负向守卫字符串;无定义、import、实例化、继承或类型依赖。 +- 删除后 focused 回归覆盖 Executor parser、Gatekeeper service/node、VerifiedInput、Verifier、Composer、Fallback、Workflow、Node Contract、Chat integration、Trace、result mapper 和结构契约:14 suites / 82 tests,0 failure、0 error、0 skipped。 +- `mvn -q -DskipTests test-compile` 通过;Graph/shared protocol 真理源保留,Spring 当前链路所需类型可完整编译。 + +### Current docs and demo contract + +- architecture index、编排、session/trace、current MVP、evidence pipeline、quality gates、feedback、retrieval 和 eval 文档已切换为 bounded StateGraph、显式 Gatekeeper/Verified Input、verified-only Verifier、有限重试/Fallback 和 Run-owned `orchestration_trace`。 +- current-doc scan 对 `SequentialAgent`、旧 Hook/ThreadLocal/service 及旧测试类名为 0 命中;`tool_trace_summary` 仅剩 4 处,均明确标注为旧 Run/fixture 只读兼容,不是当前 Verifier 输入。 +- interview demo check 绑定 Chat 返回的 exact runId,校验 Chat/Trace ownership、Run CHAT/SUCCESS、Agent/tool/self-evaluation、Graph trace 六个字段和 feedback success;summary 新增 orchestration version、final node、termination reason、degraded、transition count 和 evidence retry count。 +- PowerShell parser 语法检查通过;`InterviewDemoScriptContractTest` 2 tests 通过,覆盖 exact runId URL/response、orchestration fail-fast 和 summary 字段。 + +### Final deterministic gates + +- authoritative/focused regression:39 suites / 157 tests,0 failure、0 error、0 skipped;覆盖三层 Graph、全部 Graph Node/router/trace builder、Chat/Trace/Gatekeeper/Composer、Controller、Repository、schema、feedback/tool recorder 和 demo contract。 +- fixed diagnosis eval:12/12 passed,verdict distribution 为 PASS=5、LOW_CONFID=6、REJECT=1;same-baseline diff 无 regression、0 items。 +- `mvn -q -DskipTests test-compile` 通过;当前 change strict 通过,16 个主 specs strict 全部通过。 +- `git diff --check`、legacy executable refs、current-doc stale refs 和 unexpected schema change 检查全部通过。 +- focused 回归日志中的 Graph ERROR/exception stack trace 来自 `ChatServiceGraphIntegrationTest` 对 FAILED/no-answer/unhandled failure 的显式契约用例,Maven exit 0,不是未解释的 live ERROR。 + +### Live failure diagnosis and correction + +- 首次 live identity:sessionId=`iss-011-stage5-20260717140450`,runId=`run-6db680f8-f764-49d1-995f-0e55a4b05a06`。demo contract 在 exact Trace `run.orchestrationTrace=null` 处 fail-fast,未提交 feedback;Run 为 `CHAT/FAILED`,步骤/工具均为 0。 +- 新日志根因:`DiagnosisOrchestrationTraceBuilder` 收到类名相同但 classloader identity 不同的 `OrchestrationEvent`,`instanceof` 失败并抛出 `orchestration events contain unsupported value`。这是 Spring Boot DevTools live classloader 才暴露的 Graph state 表示缺陷,单元 JVM 未复现。 +- 冲突分类:代码偏离/运行时兼容 bug,OpenSpec 对 non-empty orchestration trace 和 live Maven startup 的要求正确,不修改验收口径。 +- RED:新增 portable event map builder 回归,修复前 1 test error;GREEN:Graph state 的 production/test actions 改存 classloader-neutral Map,builder兼容 local record/Map,Node/Workflow assertions 改读 Map。 +- 修复后先运行 5-suite Graph/Node/Runtime/Chat integration focused gate,再运行完整 39 suites / 157 tests,全部 0 failure/error/skipped;首次 Maven 进程链已按 ownership 停止,9900 已释放。 + +- 第二个 live blocker 为外层 Graph `RunnableConfig` 的 resume metadata 被原样传给内层 ReactAgent,触发 `Resume request without a configured checkpoint saver`。回归先证明 nested config 与 outer config 同一且含 `HUMAN_FEEDBACK`,再改为保留 sessionId/runId、剔除 resume/state-update/checkpoint 控制信息的独立配置;5 suites / 50 tests 和随后完整 39 suites / 157 tests 通过,临时 `[DEBUG-ISS011-NODE]` 探针已删除且源码扫描为 0。 +- 2026-07-17 的一次长请求在工具执行期间遭遇外部 MySQL 瞬时 `Connection is closed`,留下精确 `RUNNING` 失败尝试;仓库查询工具随后证明数据库恢复且 server `wait_timeout=28800`。该失败未被当作验收通过,失败 Run 保留为真实审计记录。 + +### Accepted live E2E + +- 启动:`mvn spring-boot:run "-Dspring-boot.run.profiles=mvp-demo"`;启动前 9900 空闲,隐藏进程链为 cmd `25080` -> Maven Java `17860` -> app Java `10732`,readiness 后执行固定 payment-timeout demo。 +- identity:sessionId=`iss-011-stage5-20260720015557`,runId=`run-808ac38f-3ad0-4462-a6d0-ed50d8686473`;Chat/Trace exact identity 一致,answer 长度 109,feedback request success。 +- Trace:Run=`CHAT/SUCCESS`,8 AgentSteps、12 ToolInvocations,self-evaluation 非空;`stategraph-v1`,final node=`fallback`,termination=`fallback_completed`,degraded=`true`,3 transitions,evidence retry count=0。 +- Fallback 原因是 Gatekeeper LOW_CONFID 且本轮模型输出缺少 `source_invocation_id`;这是按冻结契约执行的安全降级,answer 非空且未绕过 Gatekeeper,日志/DB 均可审计。 +- 最终日志发生跨日 rollover:7 月 20 日 active `application.log`/`chat.log` 全部属于本轮;`application-error.log` 最后写入仍为 7 月 17 日。本轮 `rg " ERROR "` 对 application/chat 为 0,新增 error-file bytes 为 0;session/run、Graph 75964ms、evaluation、exact Trace 和 useful feedback 均有关联日志。 +- MySQL:V012 `orchestration_trace` 为 nullable JSON;exact Run answer=109、duration=75964、token=111802、steps=8、tools=12、evaluation len=8822、trace len=438、feedback=useful;JSON 路由与 Trace 完全一致。 +- ownership:AgentStep 8、ToolInvocation 12,各自 distinct session/run=1、wrong owner=0;唯一 session 下 wrong run/step/tool 均为 0。 +- cleanup:只停止 PID `10732/17860/25080`,最终 9900 已释放,无剩余 owned process。 diff --git a/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/evidence.md b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/evidence.md new file mode 100644 index 0000000..87228a0 --- /dev/null +++ b/devflow/projects/2026-07-17-chat-diagnosis-stategraph-cleanup-docs/evidence.md @@ -0,0 +1,38 @@ +# Evidence + +## Source And Dependency Evidence + +- 旧闭包四个文件已删除;`src/main`/`src/test` 旧类型扫描仅剩两个测试中的负向名称守卫,无 definition/import/instantiation/type dependency。 +- 当前真理源保留 `ExecutorEvidenceParser`、`ExecutorGatekeeperService`、`GatekeeperNode`、`VerifiedInputNode`、`VerifierNodeAdapter` 和 `DiagnosisGraphResultMapper`。 +- current docs 不再描述 SequentialAgent、Hook Gatekeeper 或 full-trace Verifier;4 处 `tool_trace_summary` 均明确为历史只读兼容。 + +## Deterministic Evidence + +- 删除后 focused:14 suites / 82 tests,0 failure/error/skipped。 +- 最终 authoritative/focused:39 suites / 157 tests,0 failure/error/skipped。 +- 归档前 `mvn clean` 后重建验证:43 suites / 189 tests,0 failure/error/skipped;额外覆盖 4 个无需外部服务的现存测试类。 +- diagnosis eval:12/12 passed;PASS=5、LOW_CONFID=6、REJECT=1;same-baseline diff=0。 +- `mvn -q -DskipTests test-compile`、PowerShell parser、OpenSpec current strict、16 main specs strict、`git diff --check`、legacy/current-doc/schema scans 均通过。 + +## Diagnose Evidence + +- DevTools live classloader 使 record `instanceof` 边界失效;portable event map regression 先 RED,Node state 改用 Map 且 builder 兼容 record/Map 后 GREEN。 +- outer Graph resume metadata 污染 nested ReactAgent;nested config isolation regression 先 RED,保留 session/run metadata并剔除 resume/state-update/checkpoint 控制信息后 GREEN。 +- 两次修复后均重跑 focused 和完整 deterministic gate;临时调试探针为 0。 + +## Accepted Live Evidence + +- sessionId:`iss-011-stage5-20260720015557` +- runId:`run-808ac38f-3ad0-4462-a6d0-ed50d8686473` +- Maven profile:`mvp-demo`;Chat/Trace/feedback script exit 0。 +- Run:CHAT/SUCCESS;answer=109 chars;8 steps;12 tools;self-evaluation 非空;feedback=useful。 +- Graph:stategraph-v1;planner -> executor -> gatekeeper -> fallback;termination=fallback_completed;degraded=true;evidence retries=0。 +- Logs:本轮 active application/chat 中 ERROR=0,error appender 无新写入;session/run、evaluation、Trace、feedback 可关联。 +- DB:V012 JSON column 存在;Run fields/JSON 与 API 一致;step/tool wrong owner=0;unique-session wrong rows=0。 +- Cleanup:owned process chain 已停止,9900 已释放。 +- Archive preflight:临时 `target/iss-011-stage5*` 目录为 0,OpenSpec strict 17/17,current-doc stale=0,schema diff=0,`git diff --check` 通过。 + +## Known Limits + +- 本次真实模型遗漏 `source_invocation_id`,Gatekeeper 按契约降为 LOW_CONFID 并进入安全 Fallback;这是成功且可审计的 degraded Run,不是完整 Composer 正常路径。 +- 2026-07-17 外部 MySQL 瞬时断连留下一个 RUNNING 失败尝试;它未计入验收且保留审计,不影响 2026-07-20 exact accepted Run。 diff --git a/mvp/architecture/README.md b/mvp/architecture/README.md index 568e2a8..0de1bd0 100644 --- a/mvp/architecture/README.md +++ b/mvp/architecture/README.md @@ -1,6 +1,6 @@ # MVP 架构文档 -**更新日期**:2026-07-10 +**更新日期**:2026-07-17 这里是 MVP 当前架构的唯一入口。旧版设计、早期拆解和已经被新实现替代的方案已归档到: @@ -14,9 +14,9 @@ |---|---| | [current-mvp-architecture.md](current-mvp-architecture.md) | 当前可运行 MVP 的总体架构、链路、持久化和质量门禁 | | [interview-one-pager.md](interview-one-pager.md) | 面试一页式架构讲解,包含总图、亮点、取舍和追问回答 | -| [agent-orchestration.md](agent-orchestration.md) | Agent 编排细节,覆盖 Chat SequentialAgent、AIOps SupervisorAgent、工具边界 | +| [agent-orchestration.md](agent-orchestration.md) | Agent 编排细节,覆盖 Chat bounded StateGraph、AIOps SupervisorAgent、工具边界 | | [executor-evidence-pipeline-refactor.md](executor-evidence-pipeline-refactor.md) | Chat 证据链路当前数据契约,覆盖 Executor V2、Gatekeeper、Verifier、Composer、`evidence_refs` | -| [harness-quality-gates.md](harness-quality-gates.md) | Prompt、Hook、Trace、Gatekeeper、Verifier、Composer、评测基线组成的质量门禁 | +| [harness-quality-gates.md](harness-quality-gates.md) | Prompt、StateGraph、Trace、Gatekeeper、Verifier、Composer、评测基线组成的质量门禁 | | [rag-architecture.md](rag-architecture.md) | RAG/知识检索新架构,覆盖 L0 hint、VectorStore 主路径、SDK fallback、证据追踪 | | [modular-rag-pipeline.md](modular-rag-pipeline.md) | `lookup_knowledge` 模块化 RAG 落地架构,覆盖 pipeline、fallback、evidence-first contract、trace | | [rag-eval-closure.md](rag-eval-closure.md) | RAG 评测闭环,覆盖 offline baseline、baseline diff、diagnosis eval 和 live acceptance | @@ -29,7 +29,7 @@ ## 当前架构一句话 -SuperBizAgent MVP 是一个面向故障诊断的可追踪 Agent 系统:Chat 和 AIOps 入口统一进入 Agent 编排,Executor 通过显式工具收集日志、指标和知识库证据,Chat 链路由 Gatekeeper 做引用真实性校验、Verifier 做可推导性判断、Composer 生成最终表达;多轮会话元数据落到 `chat_session`,每次诊断运行落到 `diagnosis_run`,步骤和工具明细通过 `agent_step.run_id`、`tool_invocation.run_id` 关联,最终通过 Trace API 和评测脚本证明诊断链路可解释、可回放、可对比。 +SuperBizAgent MVP 是一个面向故障诊断的可追踪 Agent 系统:复杂 Chat 由有界 StateGraph 显式编排 Planner、Executor、Gatekeeper、Verified Input、Verifier、Composer 与安全 Fallback,Executor 通过工具收集日志、指标和知识库证据;多轮会话元数据落到 `chat_session`,每次诊断运行落到 `diagnosis_run`,步骤和工具明细通过 `run_id` 关联,Graph 路由摘要独立保存为 `orchestration_trace`,最终通过精确 Run Trace API 和评测脚本证明诊断链路可解释、可回放、可对比。 ## 阅读顺序 diff --git a/mvp/architecture/agent-orchestration.md b/mvp/architecture/agent-orchestration.md index 602e05a..95c5321 100644 --- a/mvp/architecture/agent-orchestration.md +++ b/mvp/architecture/agent-orchestration.md @@ -1,6 +1,6 @@ # Agent 编排架构 -**更新日期**:2026-07-08 +**更新日期**:2026-07-17 **状态**:当前可运行架构 **参考历史文档**:`archive/2026-07-05-legacy/agent-architecture.md` @@ -8,7 +8,7 @@ 旧版 Agent 架构把系统描述为 Supervisor、Planner、SubAgent、Verifier 的团队协作。当前 MVP 保留这个核心思想,但实现更收敛: -- Chat 链路使用固定顺序工作流:`Planner -> Executor -> Gatekeeper -> Verifier -> Composer`。 +- Chat 复杂诊断使用有递归上限的显式 StateGraph;正常路径是 `Planner -> Executor -> Gatekeeper -> Verified Input -> Verifier -> Composer`,条件边负责有限技术重试、一次补证据和安全 Fallback。 - AIOps 链路使用 `SupervisorAgent` 调度 `Planner + Executor`,最终由规则评估器做轻量验证。 - 当前没有拆分 ExternalApiSubAgent、InternalErrorSubAgent、DatabaseSubAgent;这些作为后续演进方向保留。 - 证据工具不直接散落在各个 Agent 里,而是通过 Spring AI ToolCallback / `@Tool` 统一暴露。 @@ -19,15 +19,19 @@ flowchart TB subgraph Chat["Chat diagnosis"] ChatIn["POST /api/chat"] --> ChatService["ChatService"] - ChatService --> ChatPlanner["chat_planner"] + ChatService --> ChatGraph["ChatDiagnosisGraphRuntime / StateGraph"] + ChatGraph --> ChatPlanner["Planner Node"] ChatPlanner --> ChatExecutor["chat_executor"] ChatExecutor --> ChatTools["evidence tools"] ChatTools --> ChatExecutor - ChatExecutor --> ChatGatekeeper["ExecutorGatekeeperService"] - ChatGatekeeper --> ChatVerifier["chat_verifier"] + ChatExecutor --> ChatGatekeeper["Gatekeeper Node"] + ChatGatekeeper --> VerifiedInput["Verified Input Node"] + VerifiedInput --> ChatVerifier["Verifier Node"] ChatVerifier --> ChatDecision{"PASS / LOW_CONFID / REJECT"} - ChatDecision --> ChatComposer["chat_composer"] + ChatDecision --> ChatComposer["Composer Node"] + ChatDecision --> ChatFallback["Fallback Node"] ChatComposer --> ChatAnswer["final answer"] + ChatFallback --> ChatAnswer end subgraph AiOps["AIOps diagnosis"] @@ -54,6 +58,7 @@ flowchart TB ChatPlanner --> Step ChatExecutor --> Step ChatGatekeeper --> SelfEval + ChatGraph --> Run ChatVerifier --> Step ChatTools --> Invocation ChatDecision --> SelfEval @@ -69,19 +74,13 @@ flowchart TB ## 3. Chat 编排 -Chat 复杂诊断采用 `SequentialAgent`,顺序固定: +Chat 复杂诊断采用 `ChatDiagnosisGraphRuntime` 编译的 bounded StateGraph。它有一条正常路径和显式条件边,不再依赖固定顺序 Agent 或 Verifier Hook: ```text -chat_planner - -> chat_executor - -> lookup_knowledge / query_logs / query_metrics / date_time - -> outputs executor_evidence_v2 - -> VerifierInputHook / ExecutorGatekeeperService - -> validates source_invocation_id / raw_path / evidence_excerpt - -> chat_verifier - -> judges whether verified evidence can derive claims - -> chat_composer - -> writes final user-facing answer +START -> PLANNER -> EXECUTOR -> GATEKEEPER -> VERIFIED_INPUT -> VERIFIER -> COMPOSER -> END + | | | | | + + retry + fallback + fallback + retry + retry/fallback + + EVIDENCE_RETRY -> PLANNER (最多一次) ``` 关键行为: @@ -90,16 +89,18 @@ chat_planner |---|---|---| | `chat_planner` | 拆解问题,注入知识域地图和对话历史,给出排查方向 | `planner_plan` | | `chat_executor` | 按计划调用证据工具,抽取带 `source_invocation_id + raw_path + evidence_excerpt` 的微观事实 | `executor_evidence_v2` | -| `ExecutorGatekeeperService` | 在 Verifier 前做代码级引用验真,拒绝伪造 ID、错配 raw_path、错配 excerpt | `gatekeeper_result` | -| `chat_verifier` | 只判断已验真 evidence excerpt 是否能推出 claim,不做新检索 | `verifier_output` | +| `GatekeeperNode` / `ExecutorGatekeeperService` | 按当前 `runId` 做代码级引用验真,拒绝伪造 ID、错配 raw_path、错配 excerpt | `gatekeeper_result` | +| `VerifiedInputNode` | 只投影 Gatekeeper 通过的 claims 与 matched evidence,隔离完整工具 Trace | `verified_executor_output`、`verified_evidence` | +| `chat_verifier` | 只判断已验真的 evidence excerpt 是否能推出 claim,不做新检索、不读取完整工具 Trace | `verifier_output` | | `chat_composer` | 只表达 Verifier 允许输出的 claims、缺口和建议,生成最终用户答复 | `composer_output` | +| `FallbackNode` | 在不可恢复失败或路由上限触发时生成非空安全答复 | `final_answer`、degraded trace | -Chat 链路最多支持两轮验证: +Chat Graph 支持有限技术重试,并只允许一次 evidence retry;所有分支最终进入 Composer 或 Fallback: ```mermaid sequenceDiagram autonumber - participant C as ChatService + participant C as ChatService / StateGraph participant P as chat_planner participant E as chat_executor participant T as tools @@ -114,18 +115,22 @@ sequenceDiagram E->>T: 调用证据工具 T-->>E: 证据结果 E-->>C: executor_evidence_v2 - C->>G: executor_structured_output + tool_invocation.evidence_refs + C->>G: executor_output + run-owned tool_invocation.evidence_refs G-->>C: gatekeeper_result - C->>V: executor_structured_output + gatekeeper_result + tool_trace_summary + C->>C: VerifiedInputNode projects passed claims/evidence + C->>V: verified_executor_output + verified_evidence + gatekeeper_audit V-->>C: PASS / LOW_CONFID / REJECT C->>R: 写入 verifier_evaluation - alt LOW_CONFID 且允许补证据 + alt LOW_CONFID 且允许一次补证据 C->>P: retry_context: 仅补缺失证据 - else PASS 或 REJECT + else PASS / LOW_CONFID 可输出 C->>M: allowed_claims + missing_info + recommended_actions M-->>C: composer_output C->>R: 保存 Composer 最终 answer + else 不可恢复失败 + C->>R: Fallback 安全答复 end + C->>R: 保存 orchestration_trace(version/transitions/final_node/termination_reason/degraded/evidence_retry_count) ``` 决策语义: diff --git a/mvp/architecture/current-mvp-architecture.md b/mvp/architecture/current-mvp-architecture.md index a4bebb8..cdb1c96 100644 --- a/mvp/architecture/current-mvp-architecture.md +++ b/mvp/architecture/current-mvp-architecture.md @@ -1,6 +1,6 @@ # 当前 MVP 架构 -**更新日期**:2026-07-08 +**更新日期**:2026-07-17 **状态**:当前可运行架构 **适用范围**:Demo、面试讲解、后续迭代规划 @@ -36,11 +36,14 @@ flowchart TB subgraph Agent["Agent Orchestration"] Supervisor["Supervisor"] + StateGraph["Chat Diagnosis StateGraph"] Planner["Planner"] Executor["Executor"] Gatekeeper["Gatekeeper"] + VerifiedInput["Verified Input"] Verifier["Verifier"] Composer["Composer"] + Fallback["Fallback"] end subgraph Tools["Evidence Tools"] @@ -74,7 +77,14 @@ flowchart TB end API --> App - ChatService --> Agent + ChatService --> StateGraph + StateGraph --> Planner + StateGraph --> Executor + StateGraph --> Gatekeeper + StateGraph --> VerifiedInput + StateGraph --> Verifier + StateGraph --> Composer + StateGraph --> Fallback AiOpsService --> Agent SkillRegistry --> PlannerSkillHook PlannerSkillHook --> Planner @@ -374,11 +384,12 @@ Trace API 聚合: - Agent step 序列。 - 工具调用和检索细节。 - Chat Gatekeeper / Verifier / Composer 结果。 +- Chat `run.orchestrationTrace` 路由摘要,独立于 self-evaluation 和步骤/工具明细。 - AIOps rule evaluation 结果。 Trace 是本项目区别于普通问答系统的关键:答案不是孤立文本,而是可以追溯到 Agent 决策、工具调用和证据来源。 -Prompt、Hook、Gatekeeper、Verifier、Composer 和评测门禁的完整说明见 [harness-quality-gates.md](harness-quality-gates.md),用户反馈与 `self_evaluation` 闭环见 [feedback-architecture.md](feedback-architecture.md)。 +Prompt、StateGraph、Gatekeeper、Verifier、Composer 和评测门禁的完整说明见 [harness-quality-gates.md](harness-quality-gates.md),用户反馈与 `self_evaluation` 闭环见 [feedback-architecture.md](feedback-architecture.md)。 ## 8. 质量门禁 @@ -386,9 +397,11 @@ Prompt、Hook、Gatekeeper、Verifier、Composer 和评测门禁的完整说明 | 门禁 | 位置 | 作用 | |---|---|---| -| Executor Gatekeeper | `VerifierInputHook` / `ExecutorGatekeeperService` | 校验 Executor 引用的 invocation、`raw_path`、`evidence_excerpt` 是否真实 | -| Chat Verifier | `ChatService` | 判断已验真证据是否能推出 Executor claims | -| Chat Composer | `ChatService` | 只表达 Verifier 允许输出的内容,避免把 no-evidence 说成已排除 | +| Executor Gatekeeper | `GatekeeperNode` / `ExecutorGatekeeperService` | 校验当前 Run 的 invocation、`raw_path`、`evidence_excerpt` 是否真实 | +| Verified Input | `VerifiedInputNode` | 仅投影 Gatekeeper 通过的 claims/evidence,阻断完整工具 Trace 进入 Verifier | +| Chat Verifier | `VerifierNodeAdapter` | 判断已验真证据是否能推出 Executor claims | +| Chat Composer / Fallback | `ComposerNodeAdapter` / `FallbackNode` | 输出受控答复;异常分支也必须安全终止 | +| Graph routing | `DiagnosisGraphWorkflowTest` / `run.orchestrationTrace` | 验证条件边、有限重试、最终节点和终止原因 | | AIOps Rule Evaluation | `AiOpsRuleEvaluationService` | 校验告警诊断是否聚焦 payload 并使用证据 | | Diagnosis Eval Baseline | `mvp/eval/` | 固化诊断 trace 和报告行为 | | RAG Retrieval Baseline | `eval/rag-retrieval/` | 固化检索召回行为,避免 RAG 重构回退 | @@ -399,6 +412,7 @@ Prompt、Hook、Gatekeeper、Verifier、Composer 和评测门禁的完整说明 已经完成: - Chat 和 AIOps 两条入口链路。 +- Chat 复杂诊断已单轨切换到 bounded StateGraph,并持久化 Run-owned `orchestration_trace`。 - 显式 `lookup_knowledge` Agent Tool。 - L0 从最终决策降级为 domain/entity hint。 - `VectorSearchService` 作为稳定检索门面。 @@ -430,6 +444,7 @@ Prompt、Hook、Gatekeeper、Verifier、Composer 和评测门禁的完整说明 | 能力 | 代码 | |---|---| | Chat 入口与编排 | `ChatController`, `ChatService` | +| Chat StateGraph | `ChatDiagnosisGraphRuntime`, `DiagnosisGraphFactory`, `DiagnosisRealGraphActionsFactory` | | AIOps 入口与编排 | `ChatController.aiOps`, `AiOpsService` | | AIOps 规则验证 | `AiOpsRuleEvaluationService` | | 知识库工具 | `LookupKnowledgeTool` | @@ -440,5 +455,5 @@ Prompt、Hook、Gatekeeper、Verifier、Composer 和评测门禁的完整说明 | Spring AI VectorStore 配置辅助 | `SpringAiVectorStoreSidecarService` | | Trace 聚合 | `DiagnosisTraceService` | | 工具调用记录 | `ToolInvocationRecorder` | -| Executor 引用验真 | `ExecutorGatekeeperService`, `VerifierInputHook` | +| Executor 引用验真与投影 | `GatekeeperNode`, `ExecutorGatekeeperService`, `VerifiedInputNode` | | self_evaluation 合并 | `SelfEvaluationMergeService` | diff --git a/mvp/architecture/executor-evidence-pipeline-refactor.md b/mvp/architecture/executor-evidence-pipeline-refactor.md index 792e0b4..3b7cb77 100644 --- a/mvp/architecture/executor-evidence-pipeline-refactor.md +++ b/mvp/architecture/executor-evidence-pipeline-refactor.md @@ -1,7 +1,7 @@ # Chat Evidence Pipeline Contracts **状态**:当前实现 -**更新日期**:2026-07-08 +**更新日期**:2026-07-17 **范围**:Chat 复杂诊断链路中的 Planner、Executor、Gatekeeper、Verifier、Composer 数据契约 当前 Chat 复杂诊断链路是: @@ -9,7 +9,8 @@ ```text chat_planner -> chat_executor - -> VerifierInputHook / ExecutorGatekeeperService + -> GatekeeperNode / ExecutorGatekeeperService + -> VerifiedInputNode -> chat_verifier -> chat_composer -> final answer @@ -219,13 +220,13 @@ Executor 必须遵守: ## 4. Gatekeeper -Gatekeeper 位于 Verifier 前,由 `VerifierInputHook` 触发,负责代码级引用真实性校验。 +Gatekeeper 是 StateGraph 中的显式 Node,调用 `ExecutorGatekeeperService` 对当前 Run 的证据引用做代码级真实性校验。 ### 4.1 输入 -- `sessionId` +- `sessionId + runId`(来自 `RunnableConfig`,工具查询以 `runId` 为边界) - `executor_structured_output` -- 当前 session 的 `tool_invocation` +- 当前 run 的 `tool_invocation` ### 4.2 输出 @@ -304,22 +305,20 @@ Gatekeeper 位于 Verifier 前,由 `VerifierInputHook` 触发,负责代码 ## 5. Verifier -Verifier 输入由 `VerifierInputHook` 构造: +`VerifiedInputNode` 只保留 Gatekeeper 检查通过的 claim/binding,并为 Verifier 构造最小输入: ```json { - "original_query": "用户原始问题", - "executor_final_answer": "{...executor raw text for debug/fallback only...}", - "executor_structured_output": { + "diagnosis_context": { + "query": "用户原始问题" + }, + "verified_executor_output": { "answer_version": "executor_evidence_v2", "claims": [] }, - "executor_output_parse_status": { - "status": "valid", - "detail": "parsed executor evidence contract" - }, - "tool_trace_summary": [], - "gatekeeper_result": {}, + "verified_evidence": [], + "gatekeeper_audit": {}, + "verdict_ceiling": "PASS", "retry_context": null } ``` @@ -330,7 +329,7 @@ Verifier 职责: - 不读 skill。 - 不逐字核验 excerpt 真伪;这由 Gatekeeper 完成。 - 只判断 `claim_text` 是否能由已核验的 `evidence_excerpt` 推出。 -- 结构化输出有效时,不得从 `executor_final_answer` 抽取额外确认事实。 +- 不读取 Executor 原始答复或完整工具 Trace,只读取 verified projection。 - 对 `gatekeeper_result.severity=reject` 不得输出 `PASS`。 - 对 `gatekeeper_result.severity=low_confid` 不得输出 `PASS`。 @@ -410,7 +409,7 @@ Composer 位于 Verifier 之后,输入是 ChatService 过滤后的允许表达 "rule_set_version": "gatekeeper-rules-v1" }, "composer_output": {}, - "tool_trace_summary": [] + "verified_evidence": [] } } ``` @@ -422,6 +421,9 @@ Trace API 可用于回放: - Gatekeeper 是否通过、是否自动回填。 - Verifier 如何判断可推导性。 - Composer 最终如何表达给用户。 +- `run.orchestrationTrace` 如何经过条件边、有限重试并终止。 + +历史 Run/fixture 的 `verifier_evaluation.tool_trace_summary` 仍可被 Trace UI 或离线评测只读解析,但它是旧链路兼容字段,不是当前 Verifier 输入,也不再由生产链路生成。 --- diff --git a/mvp/architecture/feedback-architecture.md b/mvp/architecture/feedback-architecture.md index 6353ea2..48296eb 100644 --- a/mvp/architecture/feedback-architecture.md +++ b/mvp/architecture/feedback-architecture.md @@ -1,6 +1,6 @@ # 反馈与自评估架构 -**更新日期**:2026-07-10 +**更新日期**:2026-07-17 **状态**:当前可运行架构 **参考历史文档**:`archive/2026-07-05-legacy/confidence-feedback.md` @@ -27,9 +27,8 @@ flowchart TD Invocation["tool_invocation"] --> RuleEval["EvaluationService: rule_evaluation"] Invocation --> EvidenceRefs["evidence_refs"] EvidenceRefs --> Gatekeeper["ExecutorGatekeeperService"] - Invocation --> TraceSummary["ToolTraceSummaryService"] - Gatekeeper --> Verifier["chat_verifier"] - TraceSummary --> Verifier + Gatekeeper --> Projection["VerifiedInputNode"] + Projection --> Verifier["chat_verifier"] Verifier --> VerifierEval["verifier_evaluation"] Verifier --> Composer["chat_composer"] Composer --> VerifierEval @@ -81,7 +80,7 @@ flowchart TD "executor_structured_output": {}, "gatekeeper_result": {}, "composer_output": {}, - "tool_trace_summary": [] + "verified_evidence": [] }, "aiops_rule_evaluation": { "verdict": "...", @@ -136,14 +135,13 @@ Chat 自评估分三步: ```mermaid flowchart LR ExecutorOutput["executor_evidence_v2"] --> Gatekeeper["ExecutorGatekeeperService"] - Invocation["tool_invocation"] --> Summary["ToolTraceSummaryService"] - Invocation --> EvidenceRefs["retrieval_details.evidence_refs"] + Invocation["tool_invocation"] --> EvidenceRefs["retrieval_details.evidence_refs"] EvidenceRefs --> Gatekeeper Gatekeeper --> GateResult["gatekeeper_result"] - Summary --> Evidence["tool_trace_summary"] - GateResult --> Verifier["chat_verifier"] - ExecutorOutput --> Verifier - Evidence --> Verifier + GateResult --> Projection["VerifiedInputNode"] + ExecutorOutput --> Projection + Projection --> VerifiedOutput["verified_executor_output + verified_evidence"] + VerifiedOutput --> Verifier["chat_verifier"] Verifier --> Output["verifier_output JSON"] Output --> Composer["chat_composer"] Composer --> ComposerOutput["composer_output"] @@ -163,9 +161,9 @@ Verifier 输出: | `facts_checked` | 逐条事实校验 | | `rationale` | 判定原因 | | `executor_structured_output` | Executor 输出的结构化 claims 与证据绑定 | +| `verified_evidence` | Gatekeeper 通过并投影给 Verifier 的最小 matched evidence | | `gatekeeper_result` | 引用真实性校验结果 | | `composer_output` | 最终表达的解析状态和摘要 | -| `tool_trace_summary` | 本次校验使用的工具调用导航索引 | ChatService 根据 verdict 决定: @@ -177,6 +175,7 @@ ChatService 根据 verdict 决定: - `executor_final_answer` 只作为 debug/fallback 上下文;结构化输出有效时,Verifier 不得从中抽取额外确认事实。 - `$.no_evidence` 只能表达“当前查询未检索到匹配证据”,不能表达“已排除/确认没有”。 +- `run.orchestrationTrace` 是独立的 StateGraph 路由摘要,不属于 `self_evaluation`;历史 `tool_trace_summary` 仅用于旧 Run/fixture 只读兼容,不是当前 Verifier 输入。 ## 6. AIOps 规则自评估 diff --git a/mvp/architecture/harness-quality-gates.md b/mvp/architecture/harness-quality-gates.md index b91d4e3..1e7c586 100644 --- a/mvp/architecture/harness-quality-gates.md +++ b/mvp/architecture/harness-quality-gates.md @@ -1,6 +1,6 @@ # Harness 与质量门禁架构 -**更新日期**:2026-07-08 +**更新日期**:2026-07-17 **状态**:当前可运行架构 + 后续门禁规划 **参考历史文档**:`archive/2026-07-05-legacy/agent-architecture.md` @@ -20,6 +20,7 @@ Agent 系统的核心风险不是“没有答案”,而是: Prompt contract + Tool boundary + Agent hooks + + StateGraph routing contract + Trace persistence + Gatekeeper deterministic validation + Verifier / rule evaluation @@ -31,7 +32,7 @@ Prompt contract ```mermaid flowchart TB Input["User / AIOps input"] --> Prompt["Prompt contract"] - Prompt --> Agent["Planner / Executor / Verifier / Composer"] + Prompt --> Agent["Diagnosis StateGraph Nodes"] Agent --> Tools["Evidence tools"] Tools --> Invocation["tool_invocation"] Agent --> StepHook["AgentLoggingHook"] @@ -41,15 +42,16 @@ flowchart TB Invocation --> EvidenceRefs["retrieval_details.evidence_refs"] EvidenceRefs --> Gatekeeper["ExecutorGatekeeperService"] Agent --> Gatekeeper - Invocation --> TraceSummary["ToolTraceSummaryService"] - Gatekeeper --> Verifier["chat_verifier"] - TraceSummary --> Verifier + Gatekeeper --> Projection["VerifiedInputNode"] + Projection --> Verifier["chat_verifier"] Verifier --> SelfEval["self_evaluation.verifier_evaluation"] Invocation --> AiOpsRule["AiOpsRuleEvaluationService"] AiOpsRule --> AiOpsEval["self_evaluation.aiops_rule_evaluation"] Run --> TraceAPI["DiagnosisTraceService"] + Agent --> Routing["diagnosis_run.orchestration_trace"] + Routing --> TraceAPI Step --> TraceAPI Invocation --> TraceAPI SelfEval --> TraceAPI @@ -161,7 +163,7 @@ error_message ## 6. Gatekeeper 与 Verifier 门禁 -Chat Verifier 前置一层 Gatekeeper。Gatekeeper 不调用 LLM,只用代码检查 Executor 输出的证据引用是否真实存在。 +Chat StateGraph 在 Verifier 前显式执行 Gatekeeper 和 Verified Input。Gatekeeper 不调用 LLM,只用代码检查 Executor 输出的证据引用是否真实存在;Verified Input 只投影通过的 binding。 ```mermaid flowchart LR @@ -169,11 +171,12 @@ flowchart LR ExecutorOutput["executor_evidence_v2"] --> Gatekeeper["ExecutorGatekeeperService"] EvidenceRefs --> Gatekeeper Gatekeeper --> GateResult["gatekeeper_result"] - Invocation --> Summary["ToolTraceSummaryService"] - Summary --> EvidenceIndex["tool_trace_summary"] + GateResult --> Projection["VerifiedInputNode"] + Projection --> VerifiedClaims["verified_executor_output"] + Projection --> VerifiedEvidence["verified_evidence"] GateResult --> Verifier["chat_verifier"] - ExecutorOutput --> Verifier - EvidenceIndex --> Verifier + VerifiedClaims --> Verifier + VerifiedEvidence --> Verifier Verifier --> Verdict{"verdict"} Verdict -->|PASS| Composer["chat_composer"] Composer --> Pass["输出最终答复"] @@ -215,7 +218,7 @@ Verifier 不再逐字核验 excerpt 真伪;这由 Gatekeeper 完成。Verifier diagnosis_run.self_evaluation.verifier_evaluation ``` -其中同时持久化 `executor_structured_output`、`gatekeeper_result`、`tool_trace_summary`、`prompt_audit` 和 `composer_output`,用于 Trace 回放。 +其中持久化 verified `executor_structured_output`、`verified_evidence`、`gatekeeper_result`、`prompt_audit` 和 `composer_output`,用于 Trace 回放。Graph 路由另存 `diagnosis_run.orchestration_trace`;历史 `tool_trace_summary` 只作为旧 Run/fixture 的读取兼容字段,不属于当前 Verifier 输入。 ## 7. AIOps 规则门禁 diff --git a/mvp/architecture/retrieval-observability.md b/mvp/architecture/retrieval-observability.md index b6f00a6..360b379 100644 --- a/mvp/architecture/retrieval-observability.md +++ b/mvp/architecture/retrieval-observability.md @@ -142,7 +142,7 @@ post-retrieval 层再把检索候选归一为: - 给 Agent 输出 completeness hint。 - 写入 `tool_invocation.relevance_level`。 - 给 Gatekeeper 提供 `evidence_refs` 引用验真源。 -- 给 Verifier 构造 `tool_trace_summary` 审计导航。 +- 由 Gatekeeper 核验后,经 `VerifiedInputNode` 给 Verifier 构造最小 `verified_evidence` 投影。 - 供 EvaluationService 计算 evidence score。 ## 6. 文档切片和 metadata @@ -180,11 +180,13 @@ flowchart LR Recorder --> Invocation["tool_invocation"] Invocation --> Trace["DiagnosisTraceService"] Invocation --> Gatekeeper["ExecutorGatekeeperService"] - Invocation --> Summary["ToolTraceSummaryService"] - Summary --> Verifier["chat_verifier"] + Gatekeeper --> Projection["VerifiedInputNode"] + Projection --> Verifier["chat_verifier"] Invocation --> Eval["EvaluationService / RAG eval"] ``` +旧 Trace/fixture 中的 `tool_trace_summary` 只保留读取兼容;当前 StateGraph 不再生成它,也不会把完整工具调用摘要输入 Verifier。 + `tool_invocation` 中与检索相关的字段: ```text diff --git a/mvp/architecture/session-trace-lifecycle.md b/mvp/architecture/session-trace-lifecycle.md index cbb61c4..12c1dd1 100644 --- a/mvp/architecture/session-trace-lifecycle.md +++ b/mvp/architecture/session-trace-lifecycle.md @@ -1,6 +1,6 @@ # 会话与 Trace 生命周期 -**更新日期**:2026-07-10 +**更新日期**:2026-07-17 **状态**:当前可运行架构 **参考历史文档**:`archive/2026-07-05-legacy/session-management.md` @@ -29,7 +29,7 @@ flowchart TD Session --> Run["create diagnosis_run(runId)"] Run --> Running["run.status = RUNNING"] - Running --> Agent["Agent workflow"] + Running --> Agent["Chat StateGraph / AIOps workflow"] Agent --> Context["execution context(sessionId, runId)"] Context --> StepHook["AgentLoggingHook"] StepHook --> Step["agent_step(session_id, run_id)"] @@ -37,7 +37,8 @@ flowchart TD Tool --> Invocation["tool_invocation(session_id, run_id)"] Invocation --> Gatekeeper["Gatekeeper evidence validation"] - Agent --> Final{"workflow result"} + Agent --> GraphTrace["Chat: save orchestration_trace"] + GraphTrace --> Final{"workflow result"} Final -->|success| Success["run.status = SUCCESS, answer saved"] Final -->|failed| Failed["run.status = FAILED"] @@ -81,6 +82,7 @@ stateDiagram-v2 | `status` | `diagnosis_run` | 单次运行执行状态 | | `answer` | `diagnosis_run` | 本次运行最终报告或答复 | | `self_evaluation` | `diagnosis_run` | 本次运行系统自评估 JSON | +| `orchestration_trace` | `diagnosis_run` | Chat StateGraph 路由摘要;包含 version、transitions、final node、termination reason、degraded 和 evidence retry count | | `feedback` | `diagnosis_run` | 本次运行用户反馈 | `feedback` 不修改 `status`。一个执行成功但用户标记 `not_useful` 的 run,仍然应该是 `SUCCESS + feedback=not_useful`。 @@ -115,7 +117,7 @@ ToolInvocationRecorder -> retrieval_details / evidence_refs ``` -Verifier、Gatekeeper 和 EvaluationService 应按 `run_id` 读取工具调用,避免同一 session 的其他 run 参与评分或证据校验。 +Gatekeeper 和 EvaluationService 按 `run_id` 读取工具调用,避免同一 session 的其他 run 参与评分或证据校验。Verifier 只读取 `VerifiedInputNode` 生成的 verified projection,不直接读取完整工具调用列表。 ## 7. Trace API 聚合 @@ -128,6 +130,7 @@ GET /api/diagnosis/{sessionId}/trace?runId=run-... ```text diagnosis_run by sessionId + runId + + run.orchestrationTrace parsed from diagnosis_run.orchestration_trace + chat_session metadata when available + agent_step where run_id = runId, ordered by the Trace API + tool_invocation where run_id = runId order by id @@ -136,16 +139,20 @@ diagnosis_run by sessionId + runId 当 `runId` 缺失时,Trace API 为兼容旧客户端解析最新 run,并在响应中返回 resolved `runId`。当 `runId` 属于其他 `sessionId` 时,API 必须拒绝,不能泄漏其他会话的 Trace。 +`run.orchestrationTrace` 只属于精确 Run 投影,不复制到顶层或 `session`。它解释 Graph 路由;`selfEvaluation` 解释证据/答案质量;`steps` 和 `toolInvocations` 保存详细执行证据,三者职责互不替代。历史 Run 的该字段可以为空。 + ## 8. Chat 与 AIOps 差异 | 维度 | Chat | AIOps | |---|---|---| | `agent_flow` | `CHAT` | `AI_OPS` | -| 编排方式 | `SequentialAgent`: Planner -> Executor -> Gatekeeper -> Verifier -> Composer | `SupervisorAgent`: Planner + Executor | +| 编排方式 | bounded `StateGraph`: Planner / Executor / Gatekeeper / Verified Input / Verifier / Composer / Fallback | `SupervisorAgent`: Planner + Executor | | 自评估 | `rule_evaluation` + `verifier_evaluation` | `aiops_rule_evaluation` | | 答案字段 | Chat 最终答复 | 告警分析报告 | | runId 暴露 | `/api/chat` JSON response | `/api/ai_ops` SSE metadata message | +Chat StateGraph 的权威自动化验收分三层:`DiagnosisGraphWorkflowTest` 验证路由,`DiagnosisGraphNodeContractTest` 验证真实 Node 输入输出,`ChatServiceGraphIntegrationTest` 验证 Run 生命周期、Trace 持久化和对外集成。 + ## 9. 清理与边界 - Redis 会话历史用于多轮上下文,不是长期审计记录。 diff --git a/mvp/demo/README.md b/mvp/demo/README.md index f4757a8..1177ef7 100644 --- a/mvp/demo/README.md +++ b/mvp/demo/README.md @@ -8,7 +8,7 @@ - `interview-walkthrough.md`:面试讲解话术。 - `evidence-pipeline-scenarios.md`:PASS / LOW_CONFID / REJECT / no-evidence 场景矩阵。 - `trace-inspection-checklist.md`:Trace 字段检查清单。 -- `scripts/run-interview-demo-check.ps1`:面试预检脚本,包含服务可达性、Chat、Trace、反馈和 summary 输出。 +- `scripts/run-interview-demo-check.ps1`:面试预检脚本,绑定 exact runId,强制校验 Run orchestration trace,并输出 Chat、Trace、反馈和 summary。 - `scripts/run-payment-timeout-demo.ps1`:本地可执行 Demo 脚本。 - `interview-q-and-a.md`:面试追问回答,覆盖 Agent 工程取舍、审计和评测。 - `requests/payment-timeout-chat.json`:固定 Chat 请求 payload。 @@ -51,6 +51,8 @@ mvp/demo/output/feedback-response.json mvp/demo/output/interview-demo-summary.json ``` +自动化验收应传入唯一 `-SessionId`,并用 `-OutputDir target/...` 避免覆盖仓库样例。脚本从 Chat 响应取得 exact `runId`,缺少 `data.run.orchestrationTrace` 或 version/final node/termination reason/transitions/degraded/evidence retry count 时会立即失败。summary 额外包含 `orchestrationVersion`、`finalNode`、`terminationReason`、`degraded`、`transitionCount` 和 `evidenceRetryCount`。 + 手动请求: ```powershell @@ -100,6 +102,10 @@ Invoke-RestMethod ` - `data.runId` 等于 `$runId` - `data.session.sessionId` 等于 Chat session id - `data.run.runId` 等于 `$runId` +- `data.run.orchestrationTrace.version` 非空 +- `data.run.orchestrationTrace.final_node` 和 `termination_reason` 非空 +- `data.run.orchestrationTrace.transitions` 是本次 Graph 的条件边记录 +- `data.run.orchestrationTrace.degraded` 和 `evidence_retry_count` 记录安全降级与补证据次数 - `data.steps` 包含 planner / executor / verifier 等步骤 - `data.toolInvocations` 包含 `lookup_knowledge`、`query_logs`、`query_metrics` 等证据工具 - `data.session.selfEvaluation` 包含 verifier 或 rule evaluation @@ -174,9 +180,10 @@ Chat 主线: ```text 一个 session id + 一个 run id -> 用户问题 --> 多 Agent 执行 +-> bounded StateGraph(Planner / Executor / Gatekeeper / Verified Input / Verifier / Composer / Fallback) -> 证据工具 -> Verifier / self_evaluation +-> run.orchestrationTrace 路由摘要 -> 最终答案 -> 用户反馈 -> Trace API 回放 diff --git a/mvp/demo/scripts/run-interview-demo-check.ps1 b/mvp/demo/scripts/run-interview-demo-check.ps1 index be63919..cf952c5 100644 --- a/mvp/demo/scripts/run-interview-demo-check.ps1 +++ b/mvp/demo/scripts/run-interview-demo-check.ps1 @@ -79,6 +79,15 @@ $chatPath = Join-Path $OutputDir "chat-response.json" $chat | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $chatPath $runId = $chat.data.runId +if ($chat.data.success -ne $true) { + throw "Chat response was not successful." +} +if ([string]::IsNullOrWhiteSpace([string]$chat.data.answer)) { + throw "Chat response did not include a non-empty answer." +} +if ($chat.data.sessionId -ne $SessionId) { + throw "Chat response sessionId '$($chat.data.sessionId)' did not match requested sessionId '$SessionId'." +} if (-not $runId) { throw "Chat response did not include runId; exact trace verification cannot continue." } @@ -92,6 +101,51 @@ $trace = Invoke-RestMethod @traceRequest $tracePath = Join-Path $OutputDir "trace-response.json" $trace | ConvertTo-Json -Depth 80 | Set-Content -Encoding UTF8 -Path $tracePath +$traceData = Get-TraceData -TraceResponse $trace +if ($null -eq $traceData -or $null -eq $traceData.run) { + throw "Exact Trace response did not include data.run." +} +if ($traceData.runId -ne $runId -or $traceData.run.runId -ne $runId) { + throw "Exact Trace runId did not match Chat runId '$runId'." +} +if ($traceData.run.sessionId -ne $SessionId) { + throw "Exact Trace run did not belong to requested sessionId '$SessionId'." +} + +$orchestrationTrace = $traceData.run.orchestrationTrace +if ($null -eq $orchestrationTrace) { + throw "Exact Trace data.run.orchestrationTrace is missing." +} +foreach ($field in @("version", "final_node", "termination_reason")) { + if (-not ($orchestrationTrace.PSObject.Properties.Name -contains $field) -or + [string]::IsNullOrWhiteSpace([string]$orchestrationTrace.$field)) { + throw "Exact Trace data.run.orchestrationTrace.$field is missing." + } +} +foreach ($field in @("transitions", "degraded", "evidence_retry_count")) { + if (-not ($orchestrationTrace.PSObject.Properties.Name -contains $field)) { + throw "Exact Trace data.run.orchestrationTrace.$field is missing." + } +} +if ($null -eq $orchestrationTrace.transitions) { + throw "Exact Trace data.run.orchestrationTrace.transitions must be an array." +} +if ([int]$orchestrationTrace.evidence_retry_count -lt 0) { + throw "Exact Trace data.run.orchestrationTrace.evidence_retry_count must not be negative." +} +if ($traceData.run.status -ne "SUCCESS" -or $traceData.run.agentFlow -ne "CHAT") { + throw "Exact Trace run must be CHAT/SUCCESS." +} +if ([string]::IsNullOrWhiteSpace([string]$traceData.run.answer)) { + throw "Exact Trace run did not include a non-empty answer." +} +if (@($traceData.steps).Count -eq 0 -or @($traceData.toolInvocations).Count -eq 0) { + throw "Exact Trace did not include both Agent steps and tool invocation evidence." +} +if ($null -eq $traceData.run.selfEvaluation) { + throw "Exact Trace run did not include selfEvaluation." +} + $feedbackBody = @{ sessionId = $SessionId runId = $runId @@ -105,11 +159,13 @@ $feedbackRequest = @{ Body = $feedbackBody } $feedback = Invoke-RestMethod @feedbackRequest +if ($feedback.success -ne $true) { + throw "Feedback request was not successful for runId '$runId'." +} $feedbackPath = Join-Path $OutputDir "feedback-response.json" $feedback | ConvertTo-Json -Depth 30 | Set-Content -Encoding UTF8 -Path $feedbackPath -$traceData = Get-TraceData -TraceResponse $trace $selfEvaluation = Get-SelfEvaluation -TraceData $traceData $verifierEvaluation = $null if ($null -ne $selfEvaluation) { @@ -138,6 +194,7 @@ if ($null -ne $promptAudit) { $promptAuditVersion = $promptAudit.version } $toolNames = Get-ToolNames -TraceData $traceData +$transitionCount = @($orchestrationTrace.transitions).Count $summaryPath = Join-Path $OutputDir "interview-demo-summary.json" $summary = [ordered]@{ @@ -149,6 +206,12 @@ $summary = [ordered]@{ gatekeeperStatus = $gatekeeperStatus gatekeeperRuleSetVersion = $gatekeeperRuleSetVersion promptAuditVersion = $promptAuditVersion + orchestrationVersion = $orchestrationTrace.version + finalNode = $orchestrationTrace.final_node + terminationReason = $orchestrationTrace.termination_reason + degraded = [bool]$orchestrationTrace.degraded + transitionCount = $transitionCount + evidenceRetryCount = [int]$orchestrationTrace.evidence_retry_count toolNames = $toolNames paths = [ordered]@{ chat = $chatPath @@ -165,4 +228,6 @@ Write-Host "Interview demo preflight completed." Write-Host "Verdict: $($summary.verdict)" Write-Host "Gatekeeper rules: $($summary.gatekeeperRuleSetVersion)" Write-Host "Prompt audit: $($summary.promptAuditVersion)" +Write-Host "Graph final node: $($summary.finalNode)" +Write-Host "Graph termination: $($summary.terminationReason)" Write-Host "Summary: $summaryPath" diff --git a/mvp/demo/trace-inspection-checklist.md b/mvp/demo/trace-inspection-checklist.md index 5ebc134..f0ceb66 100644 --- a/mvp/demo/trace-inspection-checklist.md +++ b/mvp/demo/trace-inspection-checklist.md @@ -7,16 +7,30 @@ | JSON path | 检查点 | 面试讲点 | |---|---|---| | `data.runId` / `data.run.runId` | 是否等于 demo 响应中的 `runId` | `runId` 精确绑定这一次诊断运行 | -| `data.session.sessionId` | 是否等于 `mvp-demo-payment-timeout-001` | `sessionId` 保留多轮上下文,Trace 精确回放依赖 `runId` | -| `data.session.query` | 是否包含支付超时问题 | Trace 记录了原始用户意图 | -| `data.session.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace | -| `data.session.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 | -| `data.session.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 如果是 Chat V2 链路,是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 | -| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.version` | 如果是 Chat V2 链路,是否记录 Prompt 审计版本 | Prompt 变更可解释、可回归 | -| `data.session.selfEvaluation.verifier_evaluation.prompt_audit.prompts[*].version` | 是否记录 planner / executor / verifier / composer 版本 | 便于定位 Prompt 变更影响 | -| `data.session.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在当前 diagnosis run 上 | +| `data.run.sessionId` | 是否等于本次 Chat 请求的唯一 sessionId | `sessionId` 保留多轮上下文,Trace 精确回放依赖 `runId` | +| `data.run.query` | 是否包含支付超时问题 | Trace 记录了原始用户意图 | +| `data.run.answer` | 是否包含最终诊断答案 | 最终答案没有脱离 Trace | +| `data.run.selfEvaluation` | 是否包含 verifier 或 rule evaluation | 答案经过质量门,不只是模型原始输出 | +| `data.run.selfEvaluation.verifier_evaluation.gatekeeper_result.rule_set_version` | 是否记录 Gatekeeper 规则版本 | 安全规则可审计、可回归 | +| `data.run.selfEvaluation.verifier_evaluation.prompt_audit.version` | 是否记录 Prompt 审计版本 | Prompt 变更可解释、可回归 | +| `data.run.selfEvaluation.verifier_evaluation.prompt_audit.prompts[*].version` | 是否记录 planner / executor / verifier / composer 版本 | 便于定位 Prompt 变更影响 | +| `data.run.feedback` | 提交反馈后是否变为 `useful` | 用户反馈挂在当前 diagnosis run 上 | -## 2. Agent 步骤 +## 2. StateGraph 路由 + +| JSON path | 检查点 | 面试讲点 | +|---|---|---| +| `data.run.orchestrationTrace.version` | 是否存在当前 trace contract 版本 | 路由摘要可演进、可兼容 | +| `data.run.orchestrationTrace.transitions[*]` | 是否记录实际经过的 Node 和 route | Graph 条件边不是从日志推断 | +| `data.run.orchestrationTrace.final_node` | 最终是 Composer 还是 Fallback | 正常输出与安全降级明确区分 | +| `data.run.orchestrationTrace.termination_reason` | 是否给出终止原因 | 每次 Run 都有可解释终点 | +| `data.run.orchestrationTrace.degraded` | 是否发生安全降级 | fallback 是可审计行为 | +| `data.run.orchestrationTrace.evidence_retry_count` | 是否为 0 或 1 | 补证据循环有硬上限 | +| `interview-demo-summary.json.finalNode` 等摘要字段 | 是否与 exact Trace 一致 | summary 只消费 Run 路由真理源 | + +`orchestrationTrace` 负责路由;`selfEvaluation` 负责证据和答案质量;AgentStep/ToolInvocation 负责详细执行与工具证据。三者不能互相替代。 + +## 3. Agent 步骤 | JSON path | 检查点 | 面试讲点 | |---|---|---| @@ -25,7 +39,7 @@ | `data.steps[*].durationMs` | 是否有步骤耗时 | Trace 可用于耗时分析 | | `data.steps[*].tokenCount` | 如可用,是否记录 token | Trace 可用于模型成本分析 | -## 3. 工具证据 +## 4. 工具证据 | JSON path | 检查点 | 面试讲点 | |---|---|---| @@ -37,7 +51,7 @@ | `data.toolInvocations[*].retrievalDetails.evidence_refs` | 是否包含 `raw_path + text` | Gatekeeper 可以用代码核对 Executor 引用 | | `data.toolInvocations[*].relevanceLevel` | 是否有相关性等级 | 可解释检索结果强弱 | -## 4. Summary +## 5. Summary | JSON path | 检查点 | 面试讲点 | |---|---|---| @@ -46,7 +60,7 @@ | `data.summary.hasVerifierEvaluation` | 是否存在 Verifier 结果 | 最终答案经过质量门 | | `data.summary.hasFeedback` | 提交反馈后是否为 true | 人类反馈闭环完成 | -## 5. 好的结果长什么样 +## 6. 好的结果长什么样 ```text 同一个 session id + run id @@ -54,5 +68,6 @@ -> 持久化 agent steps -> 持久化 evidence tool calls -> verifier / self-evaluation +-> run.orchestrationTrace routing summary -> feedback attached to the same run ``` diff --git a/mvp/eval/README.md b/mvp/eval/README.md index 27849a3..899359b 100644 --- a/mvp/eval/README.md +++ b/mvp/eval/README.md @@ -4,13 +4,15 @@ This folder contains the fixed offline regression set for the MVP diagnosis Agen ## Background -The current diagnosis chain is: +The current complex Chat diagnosis chain is a bounded StateGraph: ```text -Planner -> Executor -> Gatekeeper -> Verifier -> Composer -> final answer +Planner -> Executor -> Gatekeeper -> Verified Input -> Verifier -> Composer -> final answer + | | + + bounded evidence retry + safe Fallback ``` -Stages 1-4 introduced Executor V2 structured output, deterministic Gatekeeper audit, Verifier `claim_checks`, and Composer final-answer rendering. Stage 5 makes those audit fields part of the offline regression harness so future prompt, tool, or chain changes can be checked without relying on a one-off demo. +Executor V2 structured output, deterministic Gatekeeper audit, verified-only Verifier input, `claim_checks`, Composer rendering, and StateGraph routing are covered by deterministic tests so future prompt, tool, or graph changes can be checked without relying on a one-off demo. ## Scope @@ -55,10 +57,10 @@ Run the focused evaluator test: mvn -q "-Dtest=DiagnosisTraceEvaluatorTest" test ``` -Run the broader phase-5 regression set: +Run the authoritative Graph layers plus the fixed evaluator checks: ```powershell -mvn "-Dtest=DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +mvn -q "-Dtest=DiagnosisGraphWorkflowTest,DiagnosisGraphNodeContractTest,ChatServiceGraphIntegrationTest,DiagnosisTraceEvaluatorTest,DiagnosisEvalBaselineDiffTest,ExecutorGatekeeperServiceTest" test ``` When fixtures or evaluator rules change, regenerate both baseline reports from the same case file and fixture directory, then update JSON and Markdown together. diff --git a/mvp/issues/README.md b/mvp/issues/README.md index 89c78d8..889c021 100644 --- a/mvp/issues/README.md +++ b/mvp/issues/README.md @@ -1,6 +1,6 @@ # MVP Issues 索引 -**更新日期**:2026-07-16 +**更新日期**:2026-07-20 **状态**:按活跃问题、设计笔记、RAG 问题集和已归档问题整理 ## 目录约定 @@ -16,7 +16,6 @@ | 名称 | 标题 | 严重程度 | 状态 | 文件 | |---|---|---|---|---| -| ISS-011 | Chat 诊断 StateGraph 编排改造 | 高 | 待实现 | [active/ISS-011-chat-diagnosis-stategraph-orchestration.md](active/ISS-011-chat-diagnosis-stategraph-orchestration.md) | | ISS-003 | MVP 设计与实现 Review 收敛 | 高 | 待规划 | [active/ISS-003-mvp-design-implementation-review.md](active/ISS-003-mvp-design-implementation-review.md) | | ISS-004 | Executor 域级检索水位控制 | 低 | 待规划 | [active/ISS-004-executor-domain-hard-limit.md](active/ISS-004-executor-domain-hard-limit.md) | | executor-evidence-attribution-hallucination | Executor 证据归因幻觉 | 高 | 待规划 | [active/executor-evidence-attribution-hallucination.md](active/executor-evidence-attribution-hallucination.md) | @@ -53,6 +52,7 @@ | 名称 | 标题 | 状态 | 文件 | |---|---|---|---| +| ISS-011 | Chat 诊断 StateGraph 编排改造 | 已归档 | [archived/ISS-011-chat-diagnosis-stategraph-orchestration.md](archived/ISS-011-chat-diagnosis-stategraph-orchestration.md) | | ISS-001 | Executor 重复召回同一文档 | 已修复 | [archived/ISS-001-duplicate-retrieval.md](archived/ISS-001-duplicate-retrieval.md) | | ISS-002 | Executor 无约束重复调用 lookup_knowledge | 已修复 | [archived/ISS-002-executor-unconstrained-lookup.md](archived/ISS-002-executor-unconstrained-lookup.md) | | ISS-005 | 证据链补齐与降级契约收敛 | 已归档 | [archived/ISS-005-evidence-trace-hardening.md](archived/ISS-005-evidence-trace-hardening.md) | diff --git a/mvp/issues/active/ISS-011-chat-diagnosis-stategraph-orchestration.md b/mvp/issues/archived/ISS-011-chat-diagnosis-stategraph-orchestration.md similarity index 93% rename from mvp/issues/active/ISS-011-chat-diagnosis-stategraph-orchestration.md rename to mvp/issues/archived/ISS-011-chat-diagnosis-stategraph-orchestration.md index c366d06..2d1e88c 100644 --- a/mvp/issues/active/ISS-011-chat-diagnosis-stategraph-orchestration.md +++ b/mvp/issues/archived/ISS-011-chat-diagnosis-stategraph-orchestration.md @@ -1,8 +1,9 @@ # ISS-011 Chat 诊断 StateGraph 编排改造 -**状态**:待实现 +**状态**:已归档 **严重程度**:高 **发现时间**:2026-07-16 +**完成时间**:2026-07-20 **来源**:OnCall / Agent 编排模拟面试、当前 Chat 复杂诊断调用链复核 **预计实施周期**:2–3 个工作日 @@ -871,69 +872,69 @@ Eval baseline ### 编排 -- [ ] 每次进入 Planner 阶段时,INVALID_OUTPUT / RETRYABLE_FAILED 最多触发一次技术重试。 -- [ ] Planner NON_RETRYABLE_FAILED 或当前阶段第二次技术失败直接进入 Fallback。 -- [ ] Planner 技术重试不增加 `evidence_retry_count`,补证据重新进入 Planner 时重置当前阶段的 `planner_retry_count`。 -- [ ] Executor FAILED / TOOL_BLOCKED 后不会执行 Gatekeeper 和 Verifier。 -- [ ] Executor INVALID_OUTPUT 不重试,不执行 Gatekeeper、Verifier 和模型 Composer。 -- [ ] TOOL_BLOCKED 只用于工具层明确阻断且不存在合法 Executor 输出的场景。 -- [ ] 工具空结果或工具失败后仍形成合法 Executor 输出时状态为 COMPLETED,并继续 Gatekeeper。 -- [ ] Executor 合法 no-evidence 会继续执行 Gatekeeper 和 Verifier。 -- [ ] Gatekeeper REJECT 直接进入 Fallback,不执行 Verifier。 -- [ ] Gatekeeper LOW_CONFID 且零条已验真 binding 时直接进入 Fallback。 -- [ ] Gatekeeper LOW_CONFID 且存在已验真 binding 时,Verifier 只接收通过校验的 binding。 -- [ ] Gatekeeper PASS 和可继续的 LOW_CONFID 都经过 Verifier Input Builder。 -- [ ] Verifier 只接收通过 binding 对应的 `verified_evidence`,不接收完整 `tool_trace_summary`。 -- [ ] 未被 Executor 引用或未通过 Gatekeeper 的工具结果不能进入 Verifier 输入。 -- [ ] Gatekeeper LOW_CONFID 路径的 `effective_verdict` 不得升级为 PASS。 -- [ ] Gatekeeper 原始 pass/fail + severity 正确标准化为 PASS / LOW_CONFID / REJECT,未知状态安全映射为 REJECT。 -- [ ] Verifier 执行状态与诊断 verdict 分离,任何失败状态不得出现在 model/effective verdict 中。 -- [ ] Composer 和 Graph 条件边只读取 `effective_verdict`。 -- [ ] Verifier INVALID_OUTPUT / RETRYABLE_FAILED 使用相同 verified input 最多技术重试一次,且不重新执行 Gatekeeper、Executor 或工具。 -- [ ] Composer INVALID_OUTPUT / RETRYABLE_FAILED 使用相同安全输入最多技术重试一次,且不重新执行 Verifier 或前序节点。 -- [ ] Verifier 第二次技术失败或 NON_RETRYABLE_FAILED 的 Fallback 不输出 Executor claim。 -- [ ] Composer 第二次技术失败或 NON_RETRYABLE_FAILED 使用确定性安全模板。 -- [ ] `verifier_retry_count`、`composer_retry_count` 和 `evidence_retry_count` 互相独立。 -- [ ] Verifier LOW_CONFID 最多触发一次 Planner 补证据。 -- [ ] LOW_CONFID 补证据循环受一次补查上限和 Graph recursion limit 限制。 -- [ ] Gatekeeper verdict ceiling 导致的 LOW_CONFID 不触发补证据。 -- [ ] 无法从 `facts_checked` 提取有效 `evidence_gaps` 时不触发补证据。 -- [ ] 第二轮 Planner 只输出增量计划,不扩大诊断范围或重复成功查询。 -- [ ] 第二轮 Executor 只执行增量查询,但输出完整 `executor_evidence_v2` 快照,而不是仅输出新增片段。 -- [ ] 第二轮完整快照包含需要保留的第一轮可信 claims,并由 Gatekeeper 对全部 binding 重新验真。 -- [ ] Java 编排层不对两轮 claim 文本进行语义合并。 -- [ ] Composer 技术重试耗尽或不可重试失败时使用固定模板结束。 +- [x] 每次进入 Planner 阶段时,INVALID_OUTPUT / RETRYABLE_FAILED 最多触发一次技术重试。 +- [x] Planner NON_RETRYABLE_FAILED 或当前阶段第二次技术失败直接进入 Fallback。 +- [x] Planner 技术重试不增加 `evidence_retry_count`,补证据重新进入 Planner 时重置当前阶段的 `planner_retry_count`。 +- [x] Executor FAILED / TOOL_BLOCKED 后不会执行 Gatekeeper 和 Verifier。 +- [x] Executor INVALID_OUTPUT 不重试,不执行 Gatekeeper、Verifier 和模型 Composer。 +- [x] TOOL_BLOCKED 只用于工具层明确阻断且不存在合法 Executor 输出的场景。 +- [x] 工具空结果或工具失败后仍形成合法 Executor 输出时状态为 COMPLETED,并继续 Gatekeeper。 +- [x] Executor 合法 no-evidence 会继续执行 Gatekeeper 和 Verifier。 +- [x] Gatekeeper REJECT 直接进入 Fallback,不执行 Verifier。 +- [x] Gatekeeper LOW_CONFID 且零条已验真 binding 时直接进入 Fallback。 +- [x] Gatekeeper LOW_CONFID 且存在已验真 binding 时,Verifier 只接收通过校验的 binding。 +- [x] Gatekeeper PASS 和可继续的 LOW_CONFID 都经过 Verifier Input Builder。 +- [x] Verifier 只接收通过 binding 对应的 `verified_evidence`,不接收完整 `tool_trace_summary`。 +- [x] 未被 Executor 引用或未通过 Gatekeeper 的工具结果不能进入 Verifier 输入。 +- [x] Gatekeeper LOW_CONFID 路径的 `effective_verdict` 不得升级为 PASS。 +- [x] Gatekeeper 原始 pass/fail + severity 正确标准化为 PASS / LOW_CONFID / REJECT,未知状态安全映射为 REJECT。 +- [x] Verifier 执行状态与诊断 verdict 分离,任何失败状态不得出现在 model/effective verdict 中。 +- [x] Composer 和 Graph 条件边只读取 `effective_verdict`。 +- [x] Verifier INVALID_OUTPUT / RETRYABLE_FAILED 使用相同 verified input 最多技术重试一次,且不重新执行 Gatekeeper、Executor 或工具。 +- [x] Composer INVALID_OUTPUT / RETRYABLE_FAILED 使用相同安全输入最多技术重试一次,且不重新执行 Verifier 或前序节点。 +- [x] Verifier 第二次技术失败或 NON_RETRYABLE_FAILED 的 Fallback 不输出 Executor claim。 +- [x] Composer 第二次技术失败或 NON_RETRYABLE_FAILED 使用确定性安全模板。 +- [x] `verifier_retry_count`、`composer_retry_count` 和 `evidence_retry_count` 互相独立。 +- [x] Verifier LOW_CONFID 最多触发一次 Planner 补证据。 +- [x] LOW_CONFID 补证据循环受一次补查上限和 Graph recursion limit 限制。 +- [x] Gatekeeper verdict ceiling 导致的 LOW_CONFID 不触发补证据。 +- [x] 无法从 `facts_checked` 提取有效 `evidence_gaps` 时不触发补证据。 +- [x] 第二轮 Planner 只输出增量计划,不扩大诊断范围或重复成功查询。 +- [x] 第二轮 Executor 只执行增量查询,但输出完整 `executor_evidence_v2` 快照,而不是仅输出新增片段。 +- [x] 第二轮完整快照包含需要保留的第一轮可信 claims,并由 Gatekeeper 对全部 binding 重新验真。 +- [x] Java 编排层不对两轮 claim 文本进行语义合并。 +- [x] Composer 技术重试耗尽或不可重试失败时使用固定模板结束。 ### 证据和安全 -- [ ] Gatekeeper 规则语义不放宽。 -- [ ] Verifier 只消费已验真证据。 -- [ ] no-evidence 不得表达为已排除或问题不存在。 -- [ ] REJECT 降级不泄漏 Executor 原始答案和未验证根因。 -- [ ] Executor INVALID_OUTPUT、Gatekeeper REJECT 和零条可信 binding 的固定 Fallback 不输出任何 Executor claim。 -- [ ] 前置验证失败 Fallback 只展示校验状态、工具执行概况、诊断限制和人工复核建议。 +- [x] Gatekeeper 规则语义不放宽。 +- [x] Verifier 只消费已验真证据。 +- [x] no-evidence 不得表达为已排除或问题不存在。 +- [x] REJECT 降级不泄漏 Executor 原始答案和未验证根因。 +- [x] Executor INVALID_OUTPUT、Gatekeeper REJECT 和零条可信 binding 的固定 Fallback 不输出任何 Executor claim。 +- [x] 前置验证失败 Fallback 只展示校验状态、工具执行概况、诊断限制和人工复核建议。 ### 数据与审计 -- [ ] Graph 使用 runId 作为 threadId。 -- [ ] Agent step、tool invocation 和 self_evaluation 仍绑定正确 runId。 -- [ ] `orchestration_trace` 只写入当前 diagnosis run,不污染其他 run 或 session 级数据。 -- [ ] `orchestration_trace` 不包含 Prompt、模型思考、工具原文和 Graph State 快照。 -- [ ] `orchestration_trace.transitions` 由有界 `orchestration_events` 生成,与实际节点执行顺序一致。 -- [ ] 可处理异常发生时,已经产生的 orchestration events 能够 best-effort 写入当前 run。 -- [ ] Trace 能展示实际节点路径、重试原因和终止原因。 -- [ ] 每个新 StateGraph Chat run 的 `run.orchestrationTrace` 非空,且顶层和兼容 `session` 投影不重复该字段。 -- [ ] Run 最终状态、答案、耗时、Token 和工具调用数正确回填。 -- [ ] 所有成功生成安全响应的终止路径将 Run 标记为 SUCCESS,并通过 verdict 或 `orchestrationTrace.degraded` 表达质量。 -- [ ] 只有未处理异常、持久化失败或无法生成安全响应时将 Run 标记为 FAILED。 +- [x] Graph 使用 runId 作为 threadId。 +- [x] Agent step、tool invocation 和 self_evaluation 仍绑定正确 runId。 +- [x] `orchestration_trace` 只写入当前 diagnosis run,不污染其他 run 或 session 级数据。 +- [x] `orchestration_trace` 不包含 Prompt、模型思考、工具原文和 Graph State 快照。 +- [x] `orchestration_trace.transitions` 由有界 `orchestration_events` 生成,与实际节点执行顺序一致。 +- [x] 可处理异常发生时,已经产生的 orchestration events 能够 best-effort 写入当前 run。 +- [x] Trace 能展示实际节点路径、重试原因和终止原因。 +- [x] 每个新 StateGraph Chat run 的 `run.orchestrationTrace` 非空,且顶层和兼容 `session` 投影不重复该字段。 +- [x] Run 最终状态、答案、耗时、Token 和工具调用数正确回填。 +- [x] 所有成功生成安全响应的终止路径将 Run 标记为 SUCCESS,并通过 verdict 或 `orchestrationTrace.degraded` 表达质量。 +- [x] 只有未处理异常、持久化失败或无法生成安全响应时将 Run 标记为 FAILED。 ### 工程质量 -- [ ] 新 Graph 测试覆盖所有分支。 -- [ ] `ChatServiceSequentialAgentTest` 已由新测试替换。 -- [ ] 不保留长期重复的 Sequential 和 Graph 两套实现。 -- [ ] 数据库 schema 仅新增 `diagnosis_run.orchestration_trace` nullable JSON 字段。 -- [ ] `/api/chat` 和证据协议不变;Trace API 仅在 `run` 对象新增必有的 `orchestrationTrace` 字段。 +- [x] 新 Graph 测试覆盖所有分支。 +- [x] `ChatServiceSequentialAgentTest` 已由新测试替换。 +- [x] 不保留长期重复的 Sequential 和 Graph 两套实现。 +- [x] 数据库 schema 仅新增 `diagnosis_run.orchestration_trace` nullable JSON 字段。 +- [x] `/api/chat` 和证据协议不变;Trace API 仅在 `run` 对象新增必有的 `orchestrationTrace` 字段。 --- diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.archive-ready b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.archive-ready new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.archive-ready @@ -0,0 +1 @@ + diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.committed b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.committed new file mode 100644 index 0000000..c852650 --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/.committed @@ -0,0 +1,3 @@ +committed_at: 2026-07-17 +checkpoint: Commit +authorization: user-requested-direct-implementation diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/design.md b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/design.md new file mode 100644 index 0000000..0e1fbd2 --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/design.md @@ -0,0 +1,127 @@ +## Context + +ISS-011 的运行时和测试迁移已经完成:复杂 Chat 通过真实 Diagnosis StateGraph,Graph Nodes 使用显式 RunnableConfig/verified state,Run 保存 orchestration trace,三层权威 tests 已建立。剩余工作跨越代码清理、current docs、demo script、eval、live process、日志、数据库和 Issue 生命周期,必须按“静态/自动化先完成,live E2E 最后执行”的顺序收口。 + +全仓引用证明旧闭包为 `VerifierInputHook -> VerifierContextHolder + ToolTraceSummaryService`,外加 `ToolTraceSummaryServiceTest`。新 Graph 不依赖该闭包。current architecture docs 仍描述 Sequential/Hook/full trace;历史 issues/design-notes/fixtures 则应保留当时语义或历史兼容数据。 + +## Goals / Non-Goals + +**Goals:** + +- 删除旧闭包且保持 Graph parser/Gatekeeper/projection/evaluation 行为。 +- 把 architecture index 声明的 current docs、active eval/demo docs 与真实 StateGraph/Run trace 对齐。 +- 让 interview demo check 将 exact `run.orchestrationTrace` 作为强制验收字段和 summary 输出。 +- 运行确定性 Graph/test/eval gates,再启动 Maven 完成唯一最终 live E2E。 +- 对本次 E2E 的日志和数据库做 exact session/run 证据核验并清理进程。 +- 全部通过后关闭/归档 ISS-011 和阶段 5 OpenSpec。 + +**Non-Goals:** + +- 不修改 Graph 路由、Prompt、公开 API、DTO 或 DB schema。 +- 不重写历史 archived/design-note 文档和 legacy eval fixtures。 +- 不删除 UI/eval 对历史 `tool_trace_summary` 的兼容读取。 +- 不治理仓库凭据、外部基础设施或其他 active issues。 + +## Decisions + +### 1. 删除完整旧闭包,不留下空壳 Hook + +删除 `VerifierInputHook.java`、`VerifierContextHolder.java`、`ToolTraceSummaryService.java` 和 `ToolTraceSummaryServiceTest.java`。保留空类或 deprecated wrapper 会继续让维护者误认为存在第二条 Verifier 输入路径,也会让 Spring component 扫描注册无消费者 service。 + +删除前后以全仓 definition/import/instantiation/type-use 搜索、test compilation、Graph suite、Gatekeeper/protocol tests 证明闭包;负向契约 test 可保留名称字符串,任何可执行类型依赖都视为 cleanup blocker。 + +### 2. current docs 改为显式 StateGraph,历史 docs 保留 + +必须更新: + +- `mvp/architecture/README.md` +- `agent-orchestration.md` +- `session-trace-lifecycle.md` +- `current-mvp-architecture.md` +- `executor-evidence-pipeline-refactor.md` +- `harness-quality-gates.md` +- `feedback-architecture.md` +- `retrieval-observability.md` +- `mvp/eval/README.md` +- `mvp/demo/README.md` / trace checklist + +历史 archived issues/design-notes 和 legacy fixtures不批量替换:它们记录演进阶段或兼容旧 Trace。current docs 若提到 `tool_trace_summary`,只能标注为历史读取兼容,不能描述为新 Verifier 输入/持久化源。 + +### 3. Demo script 是 final E2E executable contract + +扩展 `run-interview-demo-check.ps1`: + +1. 从 Chat response 获取 runId。 +2. 查询 exact Trace。 +3. 要求 `trace.data.run.orchestrationTrace` 非空。 +4. 要求 version/final_node/termination_reason 非空,transitions 为数组。 +5. 把 finalNode、terminationReason、degraded、transitionCount、evidenceRetryCount 写入 summary。 +6. 保留 feedback 和 Prompt/Gatekeeper summary。 + +脚本必须 fail fast,不能用 session self-evaluation 或日志合成缺失的 Graph trace。 + +### 4. 自动化门禁先于 live E2E + +顺序固定:cleanup source check → docs/script tests → Graph/Chat/Trace/Gatekeeper/Composer/Controller/Repository tests → diagnosis eval baseline/diff → test compilation → OpenSpec/static gates → live startup/E2E → logs → DB → process cleanup。 + +在 live 前发现的失败按 OpenSpec/code/test/documentation分类;live 后失败使用 diagnose loop,以 exact run evidence 定位,不通过放宽断言绕过。 + +### 5. Live E2E 使用唯一 identity 和可回收后台 Maven + +- sessionId:`iss-011-stage5-`。 +- Maven:`spring-boot:run` + `mvp-demo` profile,后台 hidden process,stdout/stderr 写入 `target/` 临时文件。 +- readiness:轮询 9900,最长明确超时,不阻塞超过 60 秒且持续汇报。 +- 执行:复用 interview demo script,输出写入 `target/iss-011-stage5-output/`,避免覆盖已提交 demo sample。 +- 最终:停止应用进程树,确认 9900 无监听,删除临时启动/output文件。 + +如果 9900 启动前已被其他进程占用,先识别而不是杀掉未知进程;只有本轮启动的 PID 可被清理。 + +### 6. 日志与数据库证据只认本次 Run + +启动前记录 `logs/application.log`、`application-error.log`、`chat.log` 长度和时间;E2E 后只读新增片段,并搜索 sessionId/runId、Graph执行、Flyway/JPA 和 ERROR。测试阶段写入的旧日志不计入 live 证据。 + +数据库通过 `scripts/query_mysql.py` 执行 read-only queries: + +- `SHOW COLUMNS ... orchestration_trace`。 +- exact `diagnosis_run` status/flow/answer/metrics/self_evaluation/orchestration trace/feedback。 +- JSON_EXTRACT version/final_node/termination_reason/degraded/evidence_retry_count。 +- exact run AgentStep/ToolInvocation count 和 distinct ownership。 +- wrong session ownership count=0。 + +所有查询必须带 E2E sessionId/runId 或 schema column 条件;不接受全局 latest 代替。 + +### 7. Issue 归档是最后一个实现动作 + +E2E/log/DB 任一失败时 ISS-011 保持 active。全部通过后更新验收 checkbox和状态,将文件 move 到 `mvp/issues/archived/`,并把 issues index 从 active 移到 archived。OpenSpec archive 在 Issue move 后执行,Git commit 是阶段 5最后门禁。 + +## Interface Impact + +- 等级:L2 内部类型删除 + demo/docs 增强。 +- 外部 `/api/chat`、Trace、feedback、DB schema 和 JSON 字段:不变。 +- 内部消费者:旧 Hook/ThreadLocal/service 无消费者,删除无需迁移调用点。 +- Demo script 行为:新增 fail-fast orchestration trace contract和 summary 字段;旧 Chat/Trace/feedback outputs保留。 +- 回滚:revert 阶段 5代码/文档;DB/生产数据无变更。E2E产生的 demo Run 是正常审计数据,不执行破坏性回滚。 + +## Risks / Trade-offs + +- [隐藏反射/配置消费者未被 rg发现] → test compilation + Spring startup 是最终验证;发现即恢复并修正规格。 +- [文档仍有历史术语] → current docs 白名单扫描;history/design-notes允许但索引明确非当前真理源。 +- [live LLM 输出波动] → mock logs/metrics提供稳定工具证据;Graph允许安全 Fallback,但验收仍要求非空真实 orchestration trace和一致 Run生命周期。 +- [外部 DB/Redis/Milvus/LLM不可用] → readiness/log/DB diagnose;不伪造通过,不修改验收口径。 +- [停止进程误伤] → 只记录和终止本轮 Maven/Java PID,结束后用端口复核。 +- [E2E output覆盖仓库样例] → 输出放 target临时目录,证据摘要提炼进 devflow 后删除临时文件。 + +## Migration Plan + +1. 删除旧闭包,运行 source/test compilation/Graph安全回归。 +2. 更新 current docs 和 demo script/checklist,增加静态脚本契约 test。 +3. 运行完整 focused tests、diagnosis eval baseline/diff、OpenSpec/static gates。 +4. Maven `mvp-demo` startup,运行 unique session E2E。 +5. 检查新增日志片段和 exact DB数据,停止进程、清理临时文件。 +6. 更新/移动 ISS-011,回填 devflow,archive OpenSpec,独立提交。 + +运行时回滚为 Git revert;数据库 V012列和 E2E Run可安全保留。 + +## Open Questions + +无。清理闭包、current/history docs边界、E2E身份/证据和 Issue关闭门禁均已确定。 diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/proposal.md b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/proposal.md new file mode 100644 index 0000000..6062d5f --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/proposal.md @@ -0,0 +1,69 @@ +# Chat Diagnosis StateGraph Cleanup, Final Acceptance And Documentation + +## Why + +阶段 0–4 已完成设计冻结、Graph 骨架、真实 Nodes、ChatService 生产切换和新测试体系,但仓库仍保留只服务旧 Sequential/Hook 链路的生产类,当前架构/评测/demo 文档仍把 Chat 描述为固定顺序、Hook Gatekeeper 和 full `tool_trace_summary`。阶段 5 需要删除这组死代码、将当前文档与真实 StateGraph 对齐,并首次运行统一 Maven live E2E、日志和数据库验收,形成 ISS-011 最终闭环。 + +## What Changes + +- 删除无生产消费者的 `VerifierInputHook`、`VerifierContextHolder`、`ToolTraceSummaryService` 及其 focused test;保留仍被 Graph 使用的共享 parser/Gatekeeper/verified input 组件。 +- 更新当前架构、Trace 生命周期、证据管线、质量门禁、反馈、评测和 demo 文档,明确 StateGraph 条件边、verified-only Verifier、Run `orchestration_trace` 和三层测试体系。 +- 更新 interview demo check,使其强制读取 `run.orchestrationTrace`,并在 summary 输出 final node、termination reason、degraded、transition count 和 evidence retry count。 +- 运行新的 Graph/Chat/Trace/Controller/Repository/Gatekeeper/Composer/Eval 回归和固定 diagnosis eval baseline。 +- 使用 `mvn spring-boot:run` 启动 `mvp-demo` profile,运行唯一 session 的 Chat→exact Trace→feedback E2E。 +- 只检查本次 E2E 新增日志片段,并用 `scripts/query_mysql.py` 核验 V012、当前 Run、AgentStep/ToolInvocation、self-evaluation/orchestration trace、feedback 和跨 Run 隔离。 +- E2E 全部通过后将 ISS-011 从 active 移到 archived,更新 issues index 和验收 checkbox。 + +## Capabilities + +### New Capabilities + +- `chat-diagnosis-stategraph-cleanup-docs`:规定旧编排死代码清理、当前架构文档对齐、最终自动化回归、Maven live E2E、日志/数据库证据和 Issue 关闭门禁。 + +### Modified Capabilities + +- `mvp-demo-trace-acceptance`:最终 interview demo 必须断言精确 Run 的非空 orchestration trace,并把 StateGraph 路由摘要纳入可复现证据包和检查清单。 + +## Scope + +### In Scope + +- 旧 Hook/ThreadLocal/trace-summary producer 闭包删除和引用清理。 +- 当前(非 archive/design history)架构、eval、demo 与 Issue 文档更新。 +- demo preflight 脚本的 orchestration trace assertion/summary 扩展。 +- 所有阶段 5单元/集成/评测门禁和最终 live E2E/log/DB 验收。 +- ISS-011 状态关闭、归档和阶段 5独立 Git commit。 + +### Out of Scope + +- 不重写历史 archived issues、旧 design-notes 或旧 fixture payload;它们保留时间点语义。 +- 不删除 Trace UI/eval 对历史 `tool_trace_summary` 的只读兼容支持。 +- 不改变 Graph 路由、Prompt、API、DTO、DB schema 或诊断业务行为;发现生产偏差时按实现期冲突规则处理。 +- 不 push,不清理仓库已有凭据/基础设施配置;这些不属于 ISS-011。 + +## Context Constraints + +- `VerifierInputHook` 与 `VerifierContextHolder` 当前只互相引用;`ToolTraceSummaryService` 只被该 Hook 和自身测试使用,因此四文件构成可删除闭包。 +- `ExecutorEvidenceParser`、`ExecutorGatekeeperService`、`GatekeeperNode`、`VerifiedInputNode`、`VerifierNodeAdapter` 和 `DiagnosisGraphResultMapper` 是新链路真理源,不得随旧闭包删除。 +- 新 StateGraph Chat run 的 `run.orchestrationTrace` 必须非空;顶层/session 不重复,历史 null 仍兼容。 +- demo summary 必须从 exact run trace 读取 Graph 摘要,不能从日志反推路由。 +- 日志验收只分析本次启动/Run 的新增内容;数据库查询必须使用 exact sessionId+runId,不能以 latest 全局记录代替。 +- live E2E 必须在所有实现、单元测试、eval、static gates 通过后执行,并在结束后关闭应用进程/端口。 + +## Acceptance + +- 生产/测试源码中不存在 `VerifierInputHook`、`VerifierContextHolder`、`ToolTraceSummaryService` 定义、import、实例化或类型依赖;负向契约测试 MAY 保留名称字符串以阻止回归;Graph verified-only、安全 Fallback 和 Gatekeeper tests仍全绿。 +- 当前架构/Trace/eval/demo 文档不再把 Chat 描述为 SequentialAgent/Hook Gatekeeper/full tool trace Verifier。 +- demo check 对 `run.orchestrationTrace` 缺失 fail fast,并把路由摘要写入 interview summary。 +- 新 Graph suite、保留安全回归、test compilation、fixed diagnosis eval baseline、OpenSpec/static gates全部通过。 +- Maven `mvp-demo` live startup 成功;Chat/Trace/feedback 请求成功并返回精确 runId。 +- 新日志片段可关联本次 sessionId/runId,无未解释 ERROR;数据库证明 Run=SUCCESS/CHAT、answer/metrics/evaluation/trace 非空、feedback=useful、步骤/工具属于当前 run、V012 列存在且 trace JSON 可解析。 +- 应用进程和 9900 端口最终清理;ISS-011 移到 archived 并更新 index/checklist。 + +## Risks + +- 删除 `ToolTraceSummaryService` 可能遗漏隐藏消费者;删除前后使用全仓引用搜索、test compilation 和 retained tests验证。 +- 当前架构文档引用面广;只更新 architecture index 声明的 current docs 和 active demo/eval docs,历史 archive/design-notes 保持不变。 +- live LLM/外部基础设施存在波动;先做 readiness,失败时按日志/Run/Trace/DB 证据 diagnose,禁止把环境失败伪装为验收通过。 +- E2E 可能污染已有 demo session;使用时间戳唯一 sessionId 和 exact runId 查询,并在证据中记录。 +- Maven/Java 子进程可能遗留;以端口和进程双重检查清理,避免影响后续工作。 diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md new file mode 100644 index 0000000..0a75141 --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md @@ -0,0 +1,88 @@ +## ADDED Requirements + +### Requirement: Legacy implicit verifier orchestration SHALL be removed + +The final StateGraph implementation SHALL have no production or test definition, import, instantiation, or executable type dependency for the legacy Verifier Hook, its ThreadLocal context, or its dedicated full-trace summary producer. Negative contract tests MAY retain identifier strings solely to prevent regression. + +#### Scenario: Legacy source inventory is inspected +- **WHEN** stage 5 cleanup completes +- **THEN** `VerifierInputHook`, `VerifierContextHolder`, and `ToolTraceSummaryService` source files SHALL NOT exist +- **AND** no production or test source SHALL import, instantiate, extend, or type-reference those types +- **AND** identifier string literals SHALL only be allowed in negative source/Prompt contract assertions + +#### Scenario: Graph safety contracts are rerun +- **WHEN** the legacy closure is removed +- **THEN** Executor parser, Gatekeeper service/node, Verified Input, Verifier, Composer, Fallback, Trace, and Chat integration tests SHALL pass +- **AND** Maven test compilation and live Spring startup SHALL succeed without the removed beans + +### Requirement: Current documentation SHALL describe the real StateGraph architecture + +Current architecture, eval, and demo documentation SHALL describe complex Chat as explicit bounded Diagnosis StateGraph orchestration with run-owned events and verified-only Verifier input. + +#### Scenario: Current docs are inspected +- **WHEN** a maintainer follows the architecture index and active eval/demo guides +- **THEN** Chat orchestration SHALL be described as conditional StateGraph Nodes rather than SequentialAgent or Hook Gatekeeper +- **AND** Verifier input SHALL use verified Executor projection/evidence rather than full tool trace +- **AND** `run.orchestrationTrace` SHALL be documented separately from self-evaluation and detailed Agent/tool Trace + +#### Scenario: Historical docs are inspected +- **WHEN** a maintainer reads archived issues, design notes, or legacy fixtures +- **THEN** those materials MAY retain their original Hook/full-trace terminology +- **AND** they SHALL NOT be indexed as the current implementation truth + +### Requirement: Final deterministic regression SHALL remain green + +The project SHALL run the authoritative Graph suite, retained public/security contracts, Maven test compilation, and fixed diagnosis eval baseline before live acceptance. + +#### Scenario: Final automated gates run +- **WHEN** stage 5 implementation and documentation are complete +- **THEN** Workflow, Node Contract, Chat Integration, Trace, Controller, Repository, Gatekeeper, Composer, parser/projection, and Eval tests SHALL pass +- **AND** the fixed diagnosis eval report/diff SHALL show no unexpected regression +- **AND** OpenSpec strict and source/whitespace checks SHALL pass + +### Requirement: Final live E2E SHALL prove exact StateGraph Run ownership + +The final acceptance SHALL start the application through Maven with the `mvp-demo` profile and SHALL execute Chat, exact Trace, and feedback requests using one unique sessionId and the returned runId. + +#### Scenario: Live Chat Graph completes +- **WHEN** the unique payment-timeout demo request completes +- **THEN** the response SHALL be successful with a non-empty answer/sessionId/runId +- **AND** exact Trace SHALL return the same runId with a non-empty `run.orchestrationTrace` +- **AND** orchestration trace SHALL include version, final node, termination reason, transitions, degraded, and evidence retry count +- **AND** feedback SHALL attach to the same runId + +#### Scenario: Live application is cleaned up +- **WHEN** live verification finishes or fails +- **THEN** the Maven/Java processes started by this stage SHALL be stopped +- **AND** port 9900 SHALL no longer be owned by the stage 5 process +- **AND** temporary target output/log files SHALL be removed after evidence extraction + +### Requirement: Final logs and database SHALL corroborate the E2E Run + +Stage 5 SHALL inspect only the current live run's new log segment and SHALL query exact database rows with the repository MySQL tool. + +#### Scenario: New logs are inspected +- **WHEN** E2E returns sessionId and runId +- **THEN** new application/chat logs SHALL contain evidence for the current request/run lifecycle +- **AND** no unexplained ERROR in the new stage 5 segment SHALL invalidate acceptance + +#### Scenario: Exact database run is queried +- **WHEN** `scripts/query_mysql.py` queries the E2E sessionId/runId +- **THEN** V012 orchestration column SHALL exist +- **AND** DiagnosisRun SHALL be CHAT/SUCCESS with non-empty answer, metrics, self-evaluation, orchestration trace, and useful feedback +- **AND** orchestration JSON fields SHALL match the exact Trace response +- **AND** AgentStep/ToolInvocation ownership checks SHALL contain no row from another run/session + +### Requirement: ISS-011 SHALL close only after all final gates pass + +The Issue SHALL remain active until cleanup, docs, deterministic regression, live E2E, logs, database, process cleanup, and OpenSpec validation are complete. + +#### Scenario: Any final gate fails +- **WHEN** a required stage 5 gate is incomplete or failed +- **THEN** ISS-011 SHALL remain active +- **AND** acceptance SHALL record the blocker without marking the OpenSpec complete + +#### Scenario: All final gates pass +- **WHEN** all stage 5 acceptance evidence is archived +- **THEN** ISS-011 checkboxes/status SHALL be completed +- **AND** the Issue SHALL move from active to archived with its index entry updated diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/mvp-demo-trace-acceptance/spec.md b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/mvp-demo-trace-acceptance/spec.md new file mode 100644 index 0000000..5676d3a --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/specs/mvp-demo-trace-acceptance/spec.md @@ -0,0 +1,39 @@ +## MODIFIED Requirements + +### Requirement: End-to-end MVP acceptance case is documented +The project SHALL include an end-to-end acceptance case that demonstrates Maven start-up, chat diagnosis, exact trace query, orchestration trace inspection, feedback submission, log inspection, and exact database verification using the same `sessionId + runId`. + +#### Scenario: Reviewer follows the acceptance case +- **WHEN** a reviewer follows the documented MVP demo acceptance steps +- **THEN** they can run the application, submit a diagnosis question, query the exact trace endpoint, inspect `run.orchestrationTrace`, and submit feedback for the same run +- **AND** they can correlate that run with new logs and exact DiagnosisRun/AgentStep/ToolInvocation database records + +### Requirement: MVP demo SHALL be reproducible for interviews +The MVP demo SHALL provide a repeatable way to show a diagnosis answer, exact run trace, StateGraph orchestration summary, verifier evaluation, and feedback. + +#### Scenario: interview demo check script records an evidence bundle +- **WHEN** the user runs the interview demo check script against a running `mvp-demo` service +- **THEN** the script SHALL submit a fixed Chat diagnosis request +- **AND** it SHALL fetch the trace for the same `sessionId + runId` +- **AND** it SHALL fail if `run.orchestrationTrace` or its version/final node/termination reason is missing +- **AND** it SHALL submit useful feedback for that run +- **AND** it SHALL write chat, trace, feedback, and summary outputs under the configured output directory +- **AND** summary SHALL include final node, termination reason, degraded, transition count, and evidence retry count + +#### Scenario: interview demo check fails with actionable readiness output +- **WHEN** the target service is not reachable +- **THEN** the script SHALL fail before issuing diagnosis requests +- **AND** the failure message SHALL name the base URL and the expected startup profile + +#### Scenario: interview documentation explains audit fields +- **WHEN** an interviewer asks how Prompt, Gatekeeper, or Graph routing changes are audited +- **THEN** the demo documentation SHALL point to `prompt_audit.version`, `gatekeeper_result.rule_set_version`, and `run.orchestrationTrace` +- **AND** it SHALL explain that deterministic eval fixtures and the Graph test suite are the regression source of truth + +### Requirement: MVP demo SHALL provide a trace inspection checklist +The MVP demo SHALL document which trace fields to inspect for StateGraph routing, evidence, verifier behavior, and run-level auditability. + +#### Scenario: Checklist maps fields to interview claims +- **WHEN** a developer reviews an exact trace response +- **THEN** the checklist SHALL map `run.orchestrationTrace` version/transitions/final node/termination reason/degraded/evidence retry count to Graph routing claims +- **AND** it SHALL map AgentStep, ToolInvocation, self-evaluation, Prompt audit, Gatekeeper audit, answer, and feedback paths to their separate responsibilities diff --git a/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/tasks.md b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/tasks.md new file mode 100644 index 0000000..c135a7d --- /dev/null +++ b/openspec/changes/archive/2026-07-20-chat-diagnosis-stategraph-cleanup-docs/tasks.md @@ -0,0 +1,52 @@ +## 1. Legacy Orchestration Closure Removal + +- [x] 1.1 Delete `VerifierInputHook`, `VerifierContextHolder`, `ToolTraceSummaryService`, and `ToolTraceSummaryServiceTest` as one proven-unused closure. +- [x] 1.2 Run whole-repository definition/import/instantiation/type-use searches proving no executable production/test source depends on the removed types; allow only negative guard strings. +- [x] 1.3 Run Executor parser, Gatekeeper service/node, VerifiedInput, Verifier, Composer, Fallback, Chat integration, and test compilation regressions after deletion. +- [x] 1.4 Confirm no Graph/shared protocol component or Spring bean required by the current path was removed. + +## 2. Current Architecture And Eval Documentation + +- [x] 2.1 Update architecture index and `agent-orchestration.md` to explicit bounded StateGraph, verified-only Verifier input, conditional retry/Fallback, and orchestration trace. +- [x] 2.2 Update `session-trace-lifecycle.md` and `current-mvp-architecture.md` for StateGraph Run lifecycle, `orchestration_trace`, explicit Gatekeeper Node, and new test layers. +- [x] 2.3 Update current evidence-pipeline/quality-gate/feedback/retrieval docs so full `tool_trace_summary` is historical compatibility rather than new Verifier input. +- [x] 2.4 Update `mvp/eval/README.md` regression commands and architecture descriptions to the authoritative Graph suite. +- [x] 2.5 Run a current-doc source scan proving no current truth document describes Chat as SequentialAgent/Hook Gatekeeper/full-trace Verifier. + +## 3. Demo Orchestration Trace Contract + +- [x] 3.1 Extend `run-interview-demo-check.ps1` to fail fast when exact `run.orchestrationTrace` or required routing fields are missing. +- [x] 3.2 Add final node, termination reason, degraded, transition count, and evidence retry count to interview summary output. +- [x] 3.3 Update demo README and trace checklist with exact run orchestration trace fields and responsibility separation. +- [x] 3.4 Add/extend static script contract tests for exact runId use, orchestration assertion, and summary fields without requiring a live service. + +## 4. Final Deterministic Regression + +- [x] 4.1 Run authoritative Workflow/Node Contract/Chat Integration and all focused Graph/Trace/Gatekeeper/Composer/Controller/Repository tests. +- [x] 4.2 Run fixed `DiagnosisTraceEvaluatorTest` and `DiagnosisEvalBaselineDiffTest`; confirm baseline count/distribution and no unexpected diff. +- [x] 4.3 Run Maven test compilation and any docs/script contract tests touched by cleanup. +- [x] 4.4 Run current change strict validation, all main specs strict validation, `git diff --check`, legacy-source scan, current-doc scan, and no-unexpected-schema check. + +## 5. Maven Live E2E Startup And Requests + +- [x] 5.1 Confirm port 9900 is free, record pre-start log lengths/timestamps, and create a unique `iss-011-stage5-` sessionId. +- [x] 5.2 Start Maven `spring-boot:run` with the `mvp-demo` profile as a hidden tracked background process and wait for readiness with bounded polling. +- [x] 5.3 Run the interview demo check into a target-only output directory and capture Chat/Trace/feedback/summary for the exact returned runId. +- [x] 5.4 Assert live response success, non-empty answer, exact runId, Graph summary fields, Agent/tool evidence, evaluation audit, and useful feedback. + +## 6. Live Logs And Database Evidence + +- [x] 6.1 Read only the new stage-5 application/chat/error log segments and correlate the current sessionId/runId/Graph lifecycle. +- [x] 6.2 Classify every ERROR in the new segment; leave no unexplained error in accepted evidence. +- [x] 6.3 Use `scripts/query_mysql.py` to prove V012 column existence and exact Run CHAT/SUCCESS/answer/metrics/self-evaluation/orchestration trace/feedback fields. +- [x] 6.4 Query JSON routing fields and compare version/final node/termination/degraded/evidence retry count with the exact Trace response. +- [x] 6.5 Query AgentStep/ToolInvocation counts and distinct session/run ownership; prove wrong-session/run rows are zero. + +## 7. Cleanup, Issue Closure And Handoff + +- [x] 7.1 Stop only the Maven/Java processes started by stage 5 and confirm port 9900 is released even on failure. +- [x] 7.2 Remove target-only E2E output/startup files after extracting durable acceptance evidence. +- [x] 7.3 Update ISS-011 final checklist/status and move it from `mvp/issues/active` to `mvp/issues/archived` only after every gate passes. +- [x] 7.4 Update `mvp/issues/README.md` and current documentation links/status for archived ISS-011. +- [x] 7.5 Record deterministic, live E2E, log, database, process cleanup, known limits, and exact identities in devflow acceptance/evidence. +- [x] 7.6 Run final OpenSpec/static/worktree scope checks and prepare stage 5 Archive/independent Git commit without pushing. diff --git a/openspec/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md b/openspec/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md new file mode 100644 index 0000000..399a02e --- /dev/null +++ b/openspec/specs/chat-diagnosis-stategraph-cleanup-docs/spec.md @@ -0,0 +1,91 @@ +# chat-diagnosis-stategraph-cleanup-docs Specification + +## Purpose +TBD - created by archiving change chat-diagnosis-stategraph-cleanup-docs. Update Purpose after archive. +## Requirements +### Requirement: Legacy implicit verifier orchestration SHALL be removed + +The final StateGraph implementation SHALL have no production or test definition, import, instantiation, or executable type dependency for the legacy Verifier Hook, its ThreadLocal context, or its dedicated full-trace summary producer. Negative contract tests MAY retain identifier strings solely to prevent regression. + +#### Scenario: Legacy source inventory is inspected +- **WHEN** stage 5 cleanup completes +- **THEN** `VerifierInputHook`, `VerifierContextHolder`, and `ToolTraceSummaryService` source files SHALL NOT exist +- **AND** no production or test source SHALL import, instantiate, extend, or type-reference those types +- **AND** identifier string literals SHALL only be allowed in negative source/Prompt contract assertions + +#### Scenario: Graph safety contracts are rerun +- **WHEN** the legacy closure is removed +- **THEN** Executor parser, Gatekeeper service/node, Verified Input, Verifier, Composer, Fallback, Trace, and Chat integration tests SHALL pass +- **AND** Maven test compilation and live Spring startup SHALL succeed without the removed beans + +### Requirement: Current documentation SHALL describe the real StateGraph architecture + +Current architecture, eval, and demo documentation SHALL describe complex Chat as explicit bounded Diagnosis StateGraph orchestration with run-owned events and verified-only Verifier input. + +#### Scenario: Current docs are inspected +- **WHEN** a maintainer follows the architecture index and active eval/demo guides +- **THEN** Chat orchestration SHALL be described as conditional StateGraph Nodes rather than SequentialAgent or Hook Gatekeeper +- **AND** Verifier input SHALL use verified Executor projection/evidence rather than full tool trace +- **AND** `run.orchestrationTrace` SHALL be documented separately from self-evaluation and detailed Agent/tool Trace + +#### Scenario: Historical docs are inspected +- **WHEN** a maintainer reads archived issues, design notes, or legacy fixtures +- **THEN** those materials MAY retain their original Hook/full-trace terminology +- **AND** they SHALL NOT be indexed as the current implementation truth + +### Requirement: Final deterministic regression SHALL remain green + +The project SHALL run the authoritative Graph suite, retained public/security contracts, Maven test compilation, and fixed diagnosis eval baseline before live acceptance. + +#### Scenario: Final automated gates run +- **WHEN** stage 5 implementation and documentation are complete +- **THEN** Workflow, Node Contract, Chat Integration, Trace, Controller, Repository, Gatekeeper, Composer, parser/projection, and Eval tests SHALL pass +- **AND** the fixed diagnosis eval report/diff SHALL show no unexpected regression +- **AND** OpenSpec strict and source/whitespace checks SHALL pass + +### Requirement: Final live E2E SHALL prove exact StateGraph Run ownership + +The final acceptance SHALL start the application through Maven with the `mvp-demo` profile and SHALL execute Chat, exact Trace, and feedback requests using one unique sessionId and the returned runId. + +#### Scenario: Live Chat Graph completes +- **WHEN** the unique payment-timeout demo request completes +- **THEN** the response SHALL be successful with a non-empty answer/sessionId/runId +- **AND** exact Trace SHALL return the same runId with a non-empty `run.orchestrationTrace` +- **AND** orchestration trace SHALL include version, final node, termination reason, transitions, degraded, and evidence retry count +- **AND** feedback SHALL attach to the same runId + +#### Scenario: Live application is cleaned up +- **WHEN** live verification finishes or fails +- **THEN** the Maven/Java processes started by this stage SHALL be stopped +- **AND** port 9900 SHALL no longer be owned by the stage 5 process +- **AND** temporary target output/log files SHALL be removed after evidence extraction + +### Requirement: Final logs and database SHALL corroborate the E2E Run + +Stage 5 SHALL inspect only the current live run's new log segment and SHALL query exact database rows with the repository MySQL tool. + +#### Scenario: New logs are inspected +- **WHEN** E2E returns sessionId and runId +- **THEN** new application/chat logs SHALL contain evidence for the current request/run lifecycle +- **AND** no unexplained ERROR in the new stage 5 segment SHALL invalidate acceptance + +#### Scenario: Exact database run is queried +- **WHEN** `scripts/query_mysql.py` queries the E2E sessionId/runId +- **THEN** V012 orchestration column SHALL exist +- **AND** DiagnosisRun SHALL be CHAT/SUCCESS with non-empty answer, metrics, self-evaluation, orchestration trace, and useful feedback +- **AND** orchestration JSON fields SHALL match the exact Trace response +- **AND** AgentStep/ToolInvocation ownership checks SHALL contain no row from another run/session + +### Requirement: ISS-011 SHALL close only after all final gates pass + +The Issue SHALL remain active until cleanup, docs, deterministic regression, live E2E, logs, database, process cleanup, and OpenSpec validation are complete. + +#### Scenario: Any final gate fails +- **WHEN** a required stage 5 gate is incomplete or failed +- **THEN** ISS-011 SHALL remain active +- **AND** acceptance SHALL record the blocker without marking the OpenSpec complete + +#### Scenario: All final gates pass +- **WHEN** all stage 5 acceptance evidence is archived +- **THEN** ISS-011 checkboxes/status SHALL be completed +- **AND** the Issue SHALL move from active to archived with its index entry updated diff --git a/openspec/specs/mvp-demo-trace-acceptance/spec.md b/openspec/specs/mvp-demo-trace-acceptance/spec.md index 79fdf9a..a55a814 100644 --- a/openspec/specs/mvp-demo-trace-acceptance/spec.md +++ b/openspec/specs/mvp-demo-trace-acceptance/spec.md @@ -37,11 +37,12 @@ The system SHALL provide an `mvp-demo` Spring profile that documents the demo ru - **THEN** `prometheus.mock-enabled` and `cls.mock-enabled` are enabled by profile configuration ### Requirement: End-to-end MVP acceptance case is documented -The project SHALL include an end-to-end acceptance case that demonstrates start-up, chat diagnosis, trace query, and feedback submission using the same `sessionId + runId`. +The project SHALL include an end-to-end acceptance case that demonstrates Maven start-up, chat diagnosis, exact trace query, orchestration trace inspection, feedback submission, log inspection, and exact database verification using the same `sessionId + runId`. #### Scenario: Reviewer follows the acceptance case - **WHEN** a reviewer follows the documented MVP demo acceptance steps -- **THEN** they can run the application, submit a diagnosis question, query the exact trace endpoint, and submit feedback for the same run +- **THEN** they can run the application, submit a diagnosis question, query the exact trace endpoint, inspect `run.orchestrationTrace`, and submit feedback for the same run +- **AND** they can correlate that run with new logs and exact DiagnosisRun/AgentStep/ToolInvocation database records ### Requirement: MVP demo SHALL provide an interview runbook The MVP demo SHALL include a concise interview runbook that explains how to demonstrate the Agent flow and how to narrate the engineering value. @@ -67,14 +68,16 @@ The MVP demo SHALL provide scripts and request payloads for running the payment- - **THEN** it SHALL write chat, exact trace, and feedback responses under a demo output directory ### Requirement: MVP demo SHALL be reproducible for interviews -The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, verifier evaluation, and feedback. +The MVP demo SHALL provide a repeatable way to show a diagnosis answer, exact run trace, StateGraph orchestration summary, verifier evaluation, and feedback. #### Scenario: interview demo check script records an evidence bundle - **WHEN** the user runs the interview demo check script against a running `mvp-demo` service - **THEN** the script SHALL submit a fixed Chat diagnosis request - **AND** it SHALL fetch the trace for the same `sessionId + runId` +- **AND** it SHALL fail if `run.orchestrationTrace` or its version/final node/termination reason is missing - **AND** it SHALL submit useful feedback for that run -- **AND** it SHALL write chat, trace, feedback, and summary outputs under `mvp/demo/output/` +- **AND** it SHALL write chat, trace, feedback, and summary outputs under the configured output directory +- **AND** summary SHALL include final node, termination reason, degraded, transition count, and evidence retry count #### Scenario: interview demo check fails with actionable readiness output - **WHEN** the target service is not reachable @@ -82,16 +85,17 @@ The MVP demo SHALL provide a repeatable way to show a diagnosis answer, trace, v - **AND** the failure message SHALL name the base URL and the expected startup profile #### Scenario: interview documentation explains audit fields -- **WHEN** an interviewer asks how prompt or Gatekeeper changes are audited -- **THEN** the demo documentation SHALL point to `prompt_audit.version` and `gatekeeper_result.rule_set_version` -- **AND** it SHALL explain that deterministic eval fixtures are the regression source of truth +- **WHEN** an interviewer asks how Prompt, Gatekeeper, or Graph routing changes are audited +- **THEN** the demo documentation SHALL point to `prompt_audit.version`, `gatekeeper_result.rule_set_version`, and `run.orchestrationTrace` +- **AND** it SHALL explain that deterministic eval fixtures and the Graph test suite are the regression source of truth ### Requirement: MVP demo SHALL provide a trace inspection checklist -The MVP demo SHALL document which trace fields to inspect for evidence, verifier behavior, and run-level auditability. +The MVP demo SHALL document which trace fields to inspect for StateGraph routing, evidence, verifier behavior, and run-level auditability. #### Scenario: Checklist maps fields to interview claims -- **WHEN** a developer reviews a trace response -- **THEN** the checklist SHALL map concrete JSON paths to the claims made in the interview walkthrough +- **WHEN** a developer reviews an exact trace response +- **THEN** the checklist SHALL map `run.orchestrationTrace` version/transitions/final node/termination reason/degraded/evidence retry count to Graph routing claims +- **AND** it SHALL map AgentStep, ToolInvocation, self-evaluation, Prompt audit, Gatekeeper audit, answer, and feedback paths to their separate responsibilities ### Requirement: MVP demo SHALL provide a browser trace workbench The MVP demo SHALL provide a browser-accessible static page for inspecting one diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/AbstractDiagnosisAgentNodeAdapter.java b/src/main/java/com/superbiz/agent/graph/diagnosis/AbstractDiagnosisAgentNodeAdapter.java index ae8dc0e..57f13c4 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/AbstractDiagnosisAgentNodeAdapter.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/AbstractDiagnosisAgentNodeAdapter.java @@ -76,7 +76,8 @@ abstract class AbstractDiagnosisAgentNodeAdapter Map update = new LinkedHashMap<>(result.values()); update.put(DiagnosisGraphState.ORCHESTRATION_EVENTS, List.of(new OrchestrationEvent( - nodeName, result.outcome(), result.reasonCode(), attempt))); + nodeName, result.outcome(), result.reasonCode(), attempt) + .toMap())); return update; } diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilder.java b/src/main/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilder.java index 26e2f7c..db91777 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilder.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilder.java @@ -4,6 +4,7 @@ import com.alibaba.cloud.ai.graph.OverAllState; import java.util.ArrayList; import java.util.List; +import java.util.Map; public final class DiagnosisOrchestrationTraceBuilder { @@ -19,12 +20,7 @@ public final class DiagnosisOrchestrationTraceBuilder { List events = new ArrayList<>(values.size()); for (Object value : values) { - if (!(value instanceof OrchestrationEvent event)) { - throw new IllegalArgumentException( - "orchestration events contain unsupported value: " - + valueType(value)); - } - events.add(event); + events.add(toEvent(value)); } List transitions = new ArrayList<>( @@ -52,6 +48,49 @@ public final class DiagnosisOrchestrationTraceBuilder { evidenceRetryCount); } + private OrchestrationEvent toEvent(Object value) { + if (value instanceof OrchestrationEvent event) { + return event; + } + if (value instanceof Map map) { + try { + return new OrchestrationEvent( + text(map.get("node")), + text(map.get("outcome")), + text(map.get("reason_code")), + intValue(map.get("attempt"))); + } catch (RuntimeException invalidMap) { + throw unsupported(value, invalidMap); + } + } + throw unsupported(value, null); + } + + private IllegalArgumentException unsupported(Object value, Throwable cause) { + return new IllegalArgumentException( + "orchestration events contain unsupported value: " + + valueType(value), + cause); + } + + private String text(Object value) { + return value == null ? null : String.valueOf(value); + } + + private int intValue(Object value) { + if (value instanceof Number number) { + return number.intValue(); + } + if (value == null) { + return 0; + } + try { + return Integer.parseInt(String.valueOf(value)); + } catch (NumberFormatException ignored) { + return 0; + } + } + private String valueType(Object value) { return value == null ? "null" : value.getClass().getName(); } diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/EvidenceRetryPrepareNode.java b/src/main/java/com/superbiz/agent/graph/diagnosis/EvidenceRetryPrepareNode.java index 6e4d17d..1daeb2d 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/EvidenceRetryPrepareNode.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/EvidenceRetryPrepareNode.java @@ -58,7 +58,7 @@ public final class EvidenceRetryPrepareNode implements AsyncNodeActionWithConfig DiagnosisGraphTopology.Node.EVIDENCE_RETRY, "COMPLETED", DiagnosisGraphTopology.Reason.EVIDENCE_RETRY, - attempt))); + attempt).toMap())); return CompletableFuture.completedFuture(update); } diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/FallbackNode.java b/src/main/java/com/superbiz/agent/graph/diagnosis/FallbackNode.java index 2e6e684..dc4ee67 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/FallbackNode.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/FallbackNode.java @@ -53,7 +53,7 @@ public final class FallbackNode implements AsyncNodeActionWithConfig { DiagnosisGraphTopology.Node.FALLBACK, "COMPLETED", DiagnosisGraphTopology.Reason.FALLBACK_COMPLETED, - attempt))); + attempt).toMap())); return CompletableFuture.completedFuture(update); } diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/GatekeeperNode.java b/src/main/java/com/superbiz/agent/graph/diagnosis/GatekeeperNode.java index 2374bd0..b7deca2 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/GatekeeperNode.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/GatekeeperNode.java @@ -70,7 +70,7 @@ public final class GatekeeperNode implements AsyncNodeActionWithConfig { DiagnosisGraphTopology.Node.GATEKEEPER, status.name(), reasonCode(status), - attempt))); + attempt).toMap())); return CompletableFuture.completedFuture(update); } diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvoker.java b/src/main/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvoker.java index 361082e..4ef42f5 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvoker.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvoker.java @@ -4,6 +4,7 @@ import com.alibaba.cloud.ai.graph.RunnableConfig; import com.alibaba.cloud.ai.graph.agent.ReactAgent; import org.springframework.ai.chat.messages.AssistantMessage; +import java.util.Map; import java.util.Objects; public final class ReactAgentDiagnosisInvoker implements DiagnosisAgentInvoker { @@ -16,7 +17,23 @@ public final class ReactAgentDiagnosisInvoker implements DiagnosisAgentInvoker { @Override public String invoke(String input, RunnableConfig config) throws Exception { - AssistantMessage message = agent.call(input, config); + AssistantMessage message = agent.call(input, nestedConfig(config)); return Objects.requireNonNull(message, "assistant message").getText(); } + + private RunnableConfig nestedConfig(RunnableConfig outerConfig) { + RunnableConfig.Builder builder = RunnableConfig.builder(); + outerConfig.threadId().ifPresent(builder::threadId); + for (Map.Entry metadata : outerConfig.metadata() + .orElse(Map.of()).entrySet()) { + if (!RunnableConfig.HUMAN_FEEDBACK_METADATA_KEY.equals(metadata.getKey()) + && !RunnableConfig.STATE_UPDATE_METADATA_KEY.equals(metadata.getKey())) { + builder.addMetadata(metadata.getKey(), metadata.getValue()); + } + } + if (outerConfig.store() != null) { + builder.store(outerConfig.store()); + } + return builder.build(); + } } diff --git a/src/main/java/com/superbiz/agent/graph/diagnosis/VerifiedInputNode.java b/src/main/java/com/superbiz/agent/graph/diagnosis/VerifiedInputNode.java index ddda6f5..49367bc 100644 --- a/src/main/java/com/superbiz/agent/graph/diagnosis/VerifiedInputNode.java +++ b/src/main/java/com/superbiz/agent/graph/diagnosis/VerifiedInputNode.java @@ -51,7 +51,7 @@ public final class VerifiedInputNode implements AsyncNodeActionWithConfig { DiagnosisGraphTopology.Node.VERIFIED_INPUT, "COMPLETED", "verified_input_completed", - attempt))); + attempt).toMap())); return CompletableFuture.completedFuture(update); } diff --git a/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java b/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java deleted file mode 100644 index b03070e..0000000 --- a/src/main/java/com/superbiz/agent/hook/VerifierInputHook.java +++ /dev/null @@ -1,174 +0,0 @@ -package com.superbiz.agent.hook; - -import com.alibaba.cloud.ai.graph.RunnableConfig; -import com.alibaba.cloud.ai.graph.agent.hook.HookPosition; -import com.alibaba.cloud.ai.graph.agent.hook.HookPositions; -import com.alibaba.cloud.ai.graph.agent.hook.messages.AgentCommand; -import com.alibaba.cloud.ai.graph.agent.hook.messages.MessagesModelHook; -import com.fasterxml.jackson.databind.ObjectMapper; -import com.superbiz.agent.diagnosis.protocol.ExecutorEvidenceParser; -import com.superbiz.agent.service.ExecutorGatekeeperService; -import com.superbiz.agent.service.GatekeeperRuleCatalog; -import com.superbiz.agent.service.ToolTraceSummaryService; -import com.superbiz.agent.util.SessionContextHolder; -import com.superbiz.agent.util.VerifierContextHolder; -import lombok.extern.slf4j.Slf4j; -import org.springframework.ai.chat.messages.AssistantMessage; -import org.springframework.ai.chat.messages.Message; -import org.springframework.ai.chat.messages.UserMessage; - -import java.util.LinkedHashMap; -import java.util.List; -import java.util.Map; - -/** - * Replaces verifier history with an explicit structured payload. - */ -@Slf4j -@HookPositions(HookPosition.BEFORE_MODEL) -public class VerifierInputHook extends MessagesModelHook { - - private final ToolTraceSummaryService toolTraceSummaryService; - private final ExecutorGatekeeperService executorGatekeeperService; - private final ExecutorEvidenceParser executorEvidenceParser; - private final ObjectMapper objectMapper = new ObjectMapper(); - - public VerifierInputHook(ToolTraceSummaryService toolTraceSummaryService) { - this(toolTraceSummaryService, null); - } - - public VerifierInputHook(ToolTraceSummaryService toolTraceSummaryService, - ExecutorGatekeeperService executorGatekeeperService) { - this(toolTraceSummaryService, executorGatekeeperService, - new ExecutorEvidenceParser()); - } - - VerifierInputHook(ToolTraceSummaryService toolTraceSummaryService, - ExecutorGatekeeperService executorGatekeeperService, - ExecutorEvidenceParser executorEvidenceParser) { - this.toolTraceSummaryService = toolTraceSummaryService; - this.executorGatekeeperService = executorGatekeeperService; - this.executorEvidenceParser = executorEvidenceParser; - } - - @Override - public String getName() { - return "verifier_input_hook"; - } - - @Override - public AgentCommand beforeModel(List previousMessages, RunnableConfig config) { - try { - String sessionId = config.metadata("sessionId") - .map(Object::toString) - .orElseGet(SessionContextHolder::getSessionId); - String runId = config.metadata("runId") - .map(Object::toString) - .orElseGet(SessionContextHolder::getRunId); - String executorFinalAnswer = VerifierContextHolder.getExecutorFinalAnswer(); - if (executorFinalAnswer == null || executorFinalAnswer.isBlank()) { - executorFinalAnswer = extractLastAssistantText(previousMessages); - } - - List> toolTraceSummary = runId == null || runId.isBlank() - ? toolTraceSummaryService.buildVerifierTraceSummary(sessionId, executorFinalAnswer) - : toolTraceSummaryService.buildVerifierTraceSummaryForRun(runId, executorFinalAnswer); - VerifierContextHolder.setToolTraceSummary(toolTraceSummary); - - ExecutorEvidenceParser.ParseResult parseResult = - executorEvidenceParser.enrich( - executorEvidenceParser.parse(executorFinalAnswer), - toolTraceSummary); - VerifierContextHolder.setExecutorStructuredOutput(parseResult.structuredOutput()); - VerifierContextHolder.setExecutorOutputParseStatus(parseResult.status()); - - Map gatekeeperResult = runGatekeeper(sessionId, runId, parseResult); - VerifierContextHolder.setGatekeeperResult(gatekeeperResult); - - Map verifierInput = new LinkedHashMap<>(); - verifierInput.put("original_query", VerifierContextHolder.getOriginalQuery()); - verifierInput.put("executor_final_answer", executorFinalAnswer); - verifierInput.put("executor_structured_output", parseResult.structuredOutput()); - verifierInput.put("executor_output_parse_status", parseResult.status()); - verifierInput.put("tool_trace_summary", toolTraceSummary); - verifierInput.put("gatekeeper_result", gatekeeperResult); - verifierInput.put("retry_context", VerifierContextHolder.getRetryContext()); - - String payload = objectMapper.writerWithDefaultPrettyPrinter().writeValueAsString(verifierInput); - return new AgentCommand(List.of(new UserMessage(payload))); - } catch (Exception e) { - log.error("Failed to build verifier input, fallback to original messages", e); - return new AgentCommand(previousMessages); - } - } - - private Map runGatekeeper(String sessionId, String runId, - ExecutorEvidenceParser.ParseResult parseResult) { - if (executorGatekeeperService == null) { - return passGatekeeperResult(); - } - try { - if (runId != null && !runId.isBlank()) { - return executorGatekeeperService.validateRun(runId, parseResult.structuredOutput(), parseResult.status()); - } - return executorGatekeeperService.validate(sessionId, parseResult.structuredOutput(), parseResult.status()); - } catch (Exception e) { - log.error("Gatekeeper validation failed unexpectedly", e); - return executorGatekeeperService.fail("gatekeeper.internal_error", - "gatekeeper", - e.getMessage() == null ? "gatekeeper validation failed" : e.getMessage()); - } - } - - private Map passGatekeeperResult() { - GatekeeperRuleCatalog catalog = GatekeeperRuleCatalog.fallback(); - Map result = new LinkedHashMap<>(); - result.put("status", "pass"); - result.put("severity", "none"); - result.put("rule_set_version", catalog.version()); - result.put("rules", catalog.auditRules()); - result.put("checked_bindings", List.of()); - result.put("failed_rules", List.of()); - result.put("warnings", List.of()); - result.put("errors", List.of()); - return result; - } - - private String extractLastAssistantText(List previousMessages) { - for (int i = previousMessages.size() - 1; i >= 0; i--) { - if (previousMessages.get(i) instanceof AssistantMessage assistantMessage) { - String text = extractTextContent(assistantMessage); - if (text != null && !text.isBlank()) { - return text; - } - } - } - return ""; - } - - private String extractTextContent(AssistantMessage message) { - try { - try { - return message.getText(); - } catch (Exception ignore) { - // Fallback for older implementations. - } - - for (String methodName : List.of("getText", "getContent")) { - try { - var method = message.getClass().getMethod(methodName); - Object value = method.invoke(message); - if (value != null) { - return value.toString(); - } - } catch (NoSuchMethodException ignore) { - // continue - } - } - } catch (Exception e) { - log.debug("Failed to extract verifier assistant text", e); - } - return message.toString(); - } - -} diff --git a/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java b/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java deleted file mode 100644 index 6da9598..0000000 --- a/src/main/java/com/superbiz/agent/service/ToolTraceSummaryService.java +++ /dev/null @@ -1,546 +0,0 @@ -package com.superbiz.agent.service; - -import com.fasterxml.jackson.core.type.TypeReference; -import com.fasterxml.jackson.databind.ObjectMapper; -import com.superbiz.agent.domain.entity.ToolInvocation; -import com.superbiz.agent.repository.ToolInvocationRepository; -import lombok.extern.slf4j.Slf4j; -import org.springframework.stereotype.Service; - -import java.util.ArrayList; -import java.util.Comparator; -import java.util.LinkedHashMap; -import java.util.LinkedHashSet; -import java.util.List; -import java.util.Locale; -import java.util.Map; -import java.util.Set; -import java.util.regex.Matcher; -import java.util.regex.Pattern; - -/** - * Builds a verifier-facing evidence index from persisted tool invocations. - */ -@Slf4j -@Service -public class ToolTraceSummaryService { - - private static final TypeReference> MAP_TYPE = new TypeReference<>() {}; - private static final Set EVIDENCE_TOOLS = Set.of("lookup_knowledge", "query_logs", "query_metrics", "query_order"); - private static final Pattern JSON_STRING_FIELD = Pattern.compile("\"%s\"\\s*:\\s*\"((?:\\\\.|[^\"])*)\""); - - private final ToolInvocationRepository toolInvocationRepository; - private final ObjectMapper objectMapper = new ObjectMapper(); - - public ToolTraceSummaryService(ToolInvocationRepository toolInvocationRepository) { - this.toolInvocationRepository = toolInvocationRepository; - } - - public List> buildVerifierTraceSummary(String sessionId, String executorFinalAnswer) { - if (sessionId == null || sessionId.isBlank()) { - return List.of(); - } - - List invocations = toolInvocationRepository.findBySessionIdOrderByIdAsc(sessionId); - return buildVerifierTraceSummary(invocations, executorFinalAnswer); - } - - public List> buildVerifierTraceSummaryForRun(String runId, String executorFinalAnswer) { - if (runId == null || runId.isBlank()) { - return List.of(); - } - - List invocations = toolInvocationRepository.findByRunIdOrderByIdAsc(runId); - return buildVerifierTraceSummary(invocations, executorFinalAnswer); - } - - private List> buildVerifierTraceSummary(List invocations, String executorFinalAnswer) { - if (invocations.isEmpty()) { - return List.of(); - } - - Map grouped = new LinkedHashMap<>(); - for (ToolInvocation invocation : invocations) { - if (!EVIDENCE_TOOLS.contains(invocation.getToolName())) { - continue; - } - String topicDomain = extractTopicDomain(invocation); - String key = invocation.getToolName() + "|" + topicDomain; - AggregateEntry entry = grouped.computeIfAbsent( - key, - ignored -> new AggregateEntry(invocation.getToolName(), topicDomain)); - entry.absorb(invocation); - } - - List rankedEntries = grouped.values().stream() - .sorted(Comparator.comparingInt((AggregateEntry entry) -> entry.relevanceScore(executorFinalAnswer)).reversed()) - .limit(8) - .toList(); - - List> summaries = new ArrayList<>(); - for (int i = 0; i < rankedEntries.size(); i++) { - summaries.add(rankedEntries.get(i).toSummary("trace-" + (i + 1))); - } - return summaries; - } - - private String extractTopicDomain(ToolInvocation invocation) { - try { - if (invocation.getRetrievalDetails() != null && !invocation.getRetrievalDetails().isBlank()) { - Map details = objectMapper.readValue(invocation.getRetrievalDetails(), MAP_TYPE); - Object domains = details.get("retrieved_domains"); - if (domains instanceof List domainList && !domainList.isEmpty()) { - return String.valueOf(domainList.get(0)); - } - } - } catch (Exception e) { - log.debug("Failed to parse retrieved_domains, fallback to general", e); - } - return "general"; - } - - private String extractInputSummary(ToolInvocation invocation) { - String query = extractQuery(invocation); - if (query != null && !query.isBlank()) { - return "query=" + truncate(query, 120); - } - return invocation.getToolName() + " invoked"; - } - - private String extractQuery(ToolInvocation invocation) { - try { - if (invocation.getInputParams() != null && !invocation.getInputParams().isBlank()) { - Map params = objectMapper.readValue(invocation.getInputParams(), MAP_TYPE); - Object query = params.get("query"); - if (query != null) { - return String.valueOf(query); - } - } - } catch (Exception e) { - log.debug("Failed to parse invocation query", e); - } - return null; - } - - private String extractOutputSummary(ToolInvocation invocation, String topicDomain) { - String evidenceStatus = extractEvidenceStatus(invocation); - if (!Boolean.TRUE.equals(invocation.getSuccess())) { - if (invocation.getErrorMessage() != null && !invocation.getErrorMessage().isBlank()) { - return "call failed: " + truncate(invocation.getErrorMessage(), 120); - } - return "no usable evidence returned"; - } - - if (ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) { - return "retrieval skipped because the same document was already used in this session"; - } - - if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus)) { - if (invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank()) { - return "completed without usable evidence: " + truncate(invocation.getOutputPreview(), 120); - } - return "completed without usable evidence"; - } - - if ("lookup_knowledge".equals(invocation.getToolName())) { - String relevance = invocation.getRelevanceLevel() != null ? invocation.getRelevanceLevel() : "UNKNOWN"; - String trace = extractLookupTraceSummary(invocation); - String preview = invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank() - ? truncate(invocation.getOutputPreview(), 240) - : "no preview"; - return "matched domain=" + topicDomain + ", relevance=" + relevance + trace + ", preview=" + preview; - } - - if ("query_logs".equals(invocation.getToolName())) { - String concrete = extractLogEvidenceSummary(invocation.getOutputPreview()); - if (!concrete.isBlank()) { - return concrete; - } - } - - if ("query_metrics".equals(invocation.getToolName())) { - String concrete = extractMetricsEvidenceSummary(invocation.getOutputPreview()); - if (!concrete.isBlank()) { - return concrete; - } - } - - if (invocation.getOutputPreview() != null && !invocation.getOutputPreview().isBlank()) { - return truncate(invocation.getOutputPreview(), 240); - } - return "evidence retrieved without preview"; - } - - private String extractLookupTraceSummary(ToolInvocation invocation) { - if (invocation.getRetrievalDetails() == null || invocation.getRetrievalDetails().isBlank()) { - return ""; - } - try { - Map details = objectMapper.readValue(invocation.getRetrievalDetails(), MAP_TYPE); - List sources = new ArrayList<>(); - Object evidenceBlocks = details.get("evidence_blocks"); - if (evidenceBlocks instanceof List blocks) { - for (Object block : blocks) { - if (block instanceof Map blockMap) { - Object title = blockMap.get("title"); - Object source = blockMap.get("source"); - String label = title != null ? String.valueOf(title) : String.valueOf(source); - if (label != null && !label.isBlank() && !"null".equals(label)) { - sources.add(label); - } - } - if (sources.size() >= 3) { - break; - } - } - } - Object retrievalTrace = details.get("retrieval_trace"); - String selectedAttempt = ""; - if (retrievalTrace instanceof Map traceMap && traceMap.get("selected_attempt") != null) { - selectedAttempt = ", selected_attempt=" + traceMap.get("selected_attempt"); - } - return (sources.isEmpty() ? "" : ", sources=" + truncate(String.join("|", sources), 180)) + selectedAttempt; - } catch (Exception e) { - log.debug("Failed to parse lookup retrieval details", e); - return ""; - } - } - - private String extractLogEvidenceSummary(String outputPreview) { - if (outputPreview == null || outputPreview.isBlank()) { - return ""; - } - List messages = extractJsonStringFields(outputPreview, "message", 3); - List services = extractJsonStringFields(outputPreview, "service", 3); - List levels = extractJsonStringFields(outputPreview, "level", 3); - List timestamps = extractJsonStringFields(outputPreview, "timestamp", 3); - if (messages.isEmpty()) { - return ""; - } - List rows = new ArrayList<>(); - for (int i = 0; i < messages.size(); i++) { - String prefix = labelAt(timestamps, i) + labelAt(levels, i) + labelAt(services, i); - rows.add((prefix.isBlank() ? "" : prefix + " ") + truncate(messages.get(i), 220)); - } - return "log_evidence: " + truncate(String.join(" | ", rows), 520); - } - - private boolean hasOnlyGenericMockLogMessages(String outputPreview) { - List messages = extractJsonStringFields(outputPreview, "message", 3); - if (messages.isEmpty()) { - return false; - } - return messages.stream() - .allMatch(message -> message.startsWith("日志消息 #") && message.contains("查询条件:")); - } - - private String extractMetricsEvidenceSummary(String outputPreview) { - if (outputPreview == null || outputPreview.isBlank()) { - return ""; - } - List alertNames = extractJsonStringFields(outputPreview, "alert_name", 5); - List descriptions = extractJsonStringFields(outputPreview, "description", 5); - List services = extractJsonStringFields(outputPreview, "service", 5); - if (alertNames.isEmpty() && descriptions.isEmpty()) { - return ""; - } - List rows = new ArrayList<>(); - int count = Math.max(alertNames.size(), descriptions.size()); - for (int i = 0; i < Math.min(5, count); i++) { - StringBuilder row = new StringBuilder(); - if (i < alertNames.size()) { - row.append(alertNames.get(i)); - } - if (i < services.size()) { - if (!row.isEmpty()) { - row.append(" "); - } - row.append("service=").append(services.get(i)); - } - if (i < descriptions.size()) { - if (!row.isEmpty()) { - row.append(": "); - } - row.append(descriptions.get(i)); - } - rows.add(truncate(row.toString(), 220)); - } - return "metric_evidence: " + truncate(String.join(" | ", rows), 520); - } - - private List extractJsonStringFields(String text, String field, int limit) { - Pattern pattern = Pattern.compile(String.format(JSON_STRING_FIELD.pattern(), Pattern.quote(field))); - Matcher matcher = pattern.matcher(text); - List values = new ArrayList<>(); - while (matcher.find() && values.size() < limit) { - values.add(unescapeJsonString(matcher.group(1))); - } - return values; - } - - private String unescapeJsonString(String value) { - return value == null ? "" : value - .replace("\\\"", "\"") - .replace("\\\\", "\\") - .replace("\\n", "\n") - .replace("\\r", "\r") - .replace("\\t", "\t"); - } - - private String labelAt(List values, int index) { - if (index >= values.size() || values.get(index) == null || values.get(index).isBlank()) { - return ""; - } - return "[" + values.get(index) + "]"; - } - - private String determineEvidenceLevel(ToolInvocation invocation) { - String evidenceStatus = extractEvidenceStatus(invocation); - if (!Boolean.TRUE.equals(invocation.getSuccess())) { - return "none"; - } - if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus) - || ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) { - return "none"; - } - if ("query_logs".equals(invocation.getToolName()) - && hasOnlyGenericMockLogMessages(invocation.getOutputPreview())) { - return "none"; - } - if ("PRECISE".equals(invocation.getRelevanceLevel()) || "HIGHLY_RELEVANT".equals(invocation.getRelevanceLevel())) { - return "direct"; - } - if ("REFERENCE".equals(invocation.getRelevanceLevel())) { - return "indirect"; - } - if (EVIDENCE_TOOLS.contains(invocation.getToolName())) { - return "direct"; - } - return "none"; - } - - private String extractEvidenceStatus(ToolInvocation invocation) { - if (invocation.getRetrievalDetails() == null || invocation.getRetrievalDetails().isBlank()) { - return Boolean.TRUE.equals(invocation.getSuccess()) - ? ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED - : ToolInvocationRecorder.EVIDENCE_STATUS_FAILED; - } - try { - Map details = objectMapper.readValue(invocation.getRetrievalDetails(), MAP_TYPE); - Object evidenceStatus = details.get("evidence_status"); - if (evidenceStatus != null) { - return String.valueOf(evidenceStatus); - } - } catch (Exception e) { - log.debug("Failed to parse evidence_status", e); - } - return Boolean.TRUE.equals(invocation.getSuccess()) - ? ToolInvocationRecorder.EVIDENCE_STATUS_SUPPORTED - : ToolInvocationRecorder.EVIDENCE_STATUS_FAILED; - } - - private List extractStringList(Object value) { - if (!(value instanceof List list) || list.isEmpty()) { - return List.of(); - } - List result = new ArrayList<>(); - for (Object item : list) { - if (item != null) { - result.add(String.valueOf(item)); - } - } - return result; - } - - private List extractSourceDocuments(ToolInvocation invocation) { - if (invocation.getRetrievalDetails() == null || invocation.getRetrievalDetails().isBlank()) { - return List.of(); - } - try { - Map details = objectMapper.readValue(invocation.getRetrievalDetails(), MAP_TYPE); - List paths = extractStringList(details.get("l0_paths")); - if (!paths.isEmpty()) { - return paths; - } - List titles = extractStringList(details.get("l0_titles")); - if (!titles.isEmpty()) { - return titles; - } - } catch (Exception e) { - log.debug("Failed to parse source documents", e); - } - return List.of(); - } - - private String truncate(String text, int maxLength) { - if (text == null) { - return ""; - } - return text.length() <= maxLength ? text : text.substring(0, maxLength) + "..."; - } - - private final class AggregateEntry { - private final String toolName; - private final String topicDomain; - private String inputSummary; - private String outputSummary; - private boolean success; - private String evidenceLevel = "none"; - private int invocationCount; - private int failedCount; - private int noHitCount; - private final List sourceInvocationIds = new ArrayList<>(); - private final LinkedHashSet querySamples = new LinkedHashSet<>(); - private final LinkedHashSet retrievalLayers = new LinkedHashSet<>(); - private final LinkedHashSet relevanceLevels = new LinkedHashSet<>(); - private final LinkedHashSet sourceDocuments = new LinkedHashSet<>(); - - private AggregateEntry(String toolName, String topicDomain) { - this.toolName = toolName; - this.topicDomain = topicDomain; - } - - void absorb(ToolInvocation invocation) { - invocationCount++; - if (invocation.getId() != null) { - sourceInvocationIds.add(invocation.getId()); - } - - String query = extractQuery(invocation); - if (query != null && !query.isBlank()) { - querySamples.add(query); - } - if (invocation.getRetrievalLayer() != null && !invocation.getRetrievalLayer().isBlank()) { - retrievalLayers.add(invocation.getRetrievalLayer()); - } - if (invocation.getRelevanceLevel() != null && !invocation.getRelevanceLevel().isBlank()) { - relevanceLevels.add(invocation.getRelevanceLevel()); - } - sourceDocuments.addAll(extractSourceDocuments(invocation)); - - if (inputSummary == null || inputSummary.isBlank()) { - inputSummary = extractInputSummary(invocation); - } - - String evidenceStatus = extractEvidenceStatus(invocation); - if (!Boolean.TRUE.equals(invocation.getSuccess())) { - failedCount++; - if (outputSummary == null || outputSummary.isBlank()) { - outputSummary = extractOutputSummary(invocation, topicDomain); - } - return; - } - - if (ToolInvocationRecorder.EVIDENCE_STATUS_NO_EVIDENCE.equals(evidenceStatus) - || ToolInvocationRecorder.EVIDENCE_STATUS_DEDUPED.equals(evidenceStatus)) { - noHitCount++; - if (outputSummary == null || outputSummary.isBlank()) { - outputSummary = extractOutputSummary(invocation, topicDomain); - } - return; - } - - String invocationEvidenceLevel = determineEvidenceLevel(invocation); - if ("none".equals(invocationEvidenceLevel)) { - noHitCount++; - if (outputSummary == null || outputSummary.isBlank()) { - outputSummary = extractOutputSummary(invocation, topicDomain); - } - return; - } - if (!success || evidenceRank(invocationEvidenceLevel) > evidenceRank(evidenceLevel)) { - success = true; - evidenceLevel = invocationEvidenceLevel; - outputSummary = extractOutputSummary(invocation, topicDomain); - } else if (evidenceRank(invocationEvidenceLevel) == evidenceRank(evidenceLevel)) { - String candidateSummary = extractOutputSummary(invocation, topicDomain); - if (isMoreConcrete(candidateSummary, outputSummary)) { - outputSummary = candidateSummary; - } - } - } - - private boolean isMoreConcrete(String candidate, String current) { - return concretenessScore(candidate) > concretenessScore(current); - } - - private int concretenessScore(String summary) { - if (summary == null || summary.isBlank()) { - return 0; - } - int score = summary.length() > 160 ? 2 : 1; - if (summary.contains("log_evidence") || summary.contains("metric_evidence")) { - score += 5; - } - if (summary.contains("连接池耗尽") || summary.contains("OutOfMemoryError") - || summary.contains("扫描行数") || summary.contains("HighMemoryUsage") - || summary.contains("HighCPUUsage")) { - score += 4; - } - return score; - } - - int relevanceScore(String answer) { - int score = success ? 10 : 0; - if ("direct".equals(evidenceLevel)) { - score += 10; - } else if ("indirect".equals(evidenceLevel)) { - score += 5; - } - if (answer != null) { - String normalized = answer.toLowerCase(Locale.ROOT); - if (normalized.contains(topicDomain.toLowerCase(Locale.ROOT))) { - score += 8; - } - if (normalized.contains(toolName.toLowerCase(Locale.ROOT))) { - score += 3; - } - } - return score; - } - - Map toSummary(String traceRef) { - String mergedOutput = outputSummary == null ? "no summarized evidence" : outputSummary; - if (invocationCount > 1) { - StringBuilder builder = new StringBuilder(mergedOutput); - builder.append(" (merged ").append(invocationCount).append(" invocations"); - if (failedCount > 0) { - builder.append(", failed=").append(failedCount); - } - if (noHitCount > 0) { - builder.append(", no_hit=").append(noHitCount); - } - builder.append(")"); - mergedOutput = builder.toString(); - } - - Map summary = new LinkedHashMap<>(); - summary.put("trace_ref", traceRef); - summary.put("tool_name", toolName); - summary.put("success", success); - summary.put("input_summary", inputSummary); - summary.put("output_summary", mergedOutput); - summary.put("evidence_level", evidenceLevel); - summary.put("topic_domain", topicDomain); - summary.put("source_invocation_ids", new ArrayList<>(sourceInvocationIds)); - summary.put("invocation_count", invocationCount); - summary.put("failed_invocation_count", failedCount); - summary.put("no_hit_invocation_count", noHitCount); - summary.put("query_samples", new ArrayList<>(querySamples)); - summary.put("retrieval_layers", new ArrayList<>(retrievalLayers)); - summary.put("relevance_levels", new ArrayList<>(relevanceLevels)); - summary.put("source_documents", new ArrayList<>(sourceDocuments)); - return summary; - } - - private int evidenceRank(String level) { - if ("direct".equals(level)) { - return 2; - } - if ("indirect".equals(level)) { - return 1; - } - return 0; - } - } -} diff --git a/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java b/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java deleted file mode 100644 index eda86ca..0000000 --- a/src/main/java/com/superbiz/agent/util/VerifierContextHolder.java +++ /dev/null @@ -1,87 +0,0 @@ -package com.superbiz.agent.util; - -import java.util.List; -import java.util.Map; - -/** - * Thread-local verifier context shared across one planner/executor/verifier round. - */ -public final class VerifierContextHolder { - - private static final ThreadLocal ORIGINAL_QUERY = new ThreadLocal<>(); - private static final ThreadLocal RETRY_CONTEXT = new ThreadLocal<>(); - private static final ThreadLocal EXECUTOR_FINAL_ANSWER = new ThreadLocal<>(); - private static final ThreadLocal> EXECUTOR_STRUCTURED_OUTPUT = new ThreadLocal<>(); - private static final ThreadLocal> EXECUTOR_OUTPUT_PARSE_STATUS = new ThreadLocal<>(); - private static final ThreadLocal>> TOOL_TRACE_SUMMARY = new ThreadLocal<>(); - private static final ThreadLocal> GATEKEEPER_RESULT = new ThreadLocal<>(); - - private VerifierContextHolder() { - } - - public static void setOriginalQuery(String originalQuery) { - ORIGINAL_QUERY.set(originalQuery); - } - - public static String getOriginalQuery() { - return ORIGINAL_QUERY.get(); - } - - public static void setRetryContext(String retryContext) { - RETRY_CONTEXT.set(retryContext); - } - - public static String getRetryContext() { - return RETRY_CONTEXT.get(); - } - - public static void setExecutorFinalAnswer(String executorFinalAnswer) { - EXECUTOR_FINAL_ANSWER.set(executorFinalAnswer); - } - - public static String getExecutorFinalAnswer() { - return EXECUTOR_FINAL_ANSWER.get(); - } - - public static void setExecutorStructuredOutput(Map executorStructuredOutput) { - EXECUTOR_STRUCTURED_OUTPUT.set(executorStructuredOutput); - } - - public static Map getExecutorStructuredOutput() { - return EXECUTOR_STRUCTURED_OUTPUT.get(); - } - - public static void setExecutorOutputParseStatus(Map executorOutputParseStatus) { - EXECUTOR_OUTPUT_PARSE_STATUS.set(executorOutputParseStatus); - } - - public static Map getExecutorOutputParseStatus() { - return EXECUTOR_OUTPUT_PARSE_STATUS.get(); - } - - public static void setToolTraceSummary(List> toolTraceSummary) { - TOOL_TRACE_SUMMARY.set(toolTraceSummary); - } - - public static List> getToolTraceSummary() { - return TOOL_TRACE_SUMMARY.get(); - } - - public static void setGatekeeperResult(Map gatekeeperResult) { - GATEKEEPER_RESULT.set(gatekeeperResult); - } - - public static Map getGatekeeperResult() { - return GATEKEEPER_RESULT.get(); - } - - public static void clear() { - ORIGINAL_QUERY.remove(); - RETRY_CONTEXT.remove(); - EXECUTOR_FINAL_ANSWER.remove(); - EXECUTOR_STRUCTURED_OUTPUT.remove(); - EXECUTOR_OUTPUT_PARSE_STATUS.remove(); - TOOL_TRACE_SUMMARY.remove(); - GATEKEEPER_RESULT.remove(); - } -} diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java index 02995d5..ffb012c 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphNodeContractTest.java @@ -322,7 +322,7 @@ class DiagnosisGraphNodeContractTest { state, DiagnosisGraphState.ORCHESTRATION_EVENTS); assertEquals(List.of("planner", "executor", "gatekeeper", "verified_input", "verifier", "composer"), - events.stream().map(event -> ((OrchestrationEvent) event).node()).toList()); + events.stream().map(this::eventNode).toList()); verify(gatekeeper, times(1)).validateRun( eq("run-real-pass"), anyMap(), anyMap()); } @@ -352,10 +352,14 @@ class DiagnosisGraphNodeContractTest { private List eventNodes(OverAllState state) { return DiagnosisGraphState.listValue(state, DiagnosisGraphState.ORCHESTRATION_EVENTS) .stream() - .map(event -> ((OrchestrationEvent) event).node()) + .map(this::eventNode) .toList(); } + private String eventNode(Object event) { + return String.valueOf(((Map) event).get("node")); + } + private String plannerOutput() { return """ {"selected_skill":null,"selection_reason":"none", diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java index 3e90712..b6940de 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisGraphWorkflowTest.java @@ -447,10 +447,11 @@ class DiagnosisGraphWorkflowTest { initialState(), RunnableConfig.builder().threadId(THREAD_ID).build()) .orElseThrow(); - assertEquals(script.sequence(), DiagnosisGraphState.listValue( + assertEquals(script.sequence(), DiagnosisGraphState.listValue( state, DiagnosisGraphState.ORCHESTRATION_EVENTS) .stream() - .map(event -> ((OrchestrationEvent) event).node()) + .map(event -> String.valueOf( + ((Map) event).get("node"))) .toList()); return state; } diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilderTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilderTest.java index a364b8d..23bcb59 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilderTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/DiagnosisOrchestrationTraceBuilderTest.java @@ -82,6 +82,21 @@ class DiagnosisOrchestrationTraceBuilderTest { assertTrue(trace.degraded()); } + @Test + void buildsTraceFromPortableEventMapsUsedAtGraphStateBoundary() { + OverAllState state = stateWithEventValues(List.of( + event("planner", "COMPLETED", "planner_completed", 1).toMap(), + event("composer", "COMPLETED", "composer_completed", 1).toMap()), + 0); + + DiagnosisOrchestrationTrace trace = builder.build(state); + + assertEquals("composer", trace.finalNode()); + assertEquals("planner->composer", + trace.transitions().get(0).from() + "->" + + trace.transitions().get(0).to()); + } + @Test void emptyEventsFailExplicitly() { OverAllState state = new OverAllState(Map.of()); @@ -140,6 +155,12 @@ class DiagnosisOrchestrationTraceBuilderTest { private OverAllState stateWithEvents( List events, int evidenceRetryCount) { + return stateWithEventValues(events, evidenceRetryCount); + } + + private OverAllState stateWithEventValues( + List events, + int evidenceRetryCount) { return new OverAllState(Map.of( DiagnosisGraphState.ORCHESTRATION_EVENTS, events, DiagnosisGraphState.EVIDENCE_RETRY_COUNT, evidenceRetryCount)); diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/ExecutorNodeAdapterTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/ExecutorNodeAdapterTest.java index 334ae62..f516455 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/ExecutorNodeAdapterTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/ExecutorNodeAdapterTest.java @@ -52,9 +52,9 @@ class ExecutorNodeAdapterTest { assertEquals("COMPLETED", update.get(DiagnosisGraphState.EXECUTOR_STATUS)); assertEquals(List.of(), ((Map) update.get( DiagnosisGraphState.EXECUTOR_OUTPUT)).get("claims")); - OrchestrationEvent event = (OrchestrationEvent) ((List) update.get( + Map event = (Map) ((List) update.get( DiagnosisGraphState.ORCHESTRATION_EVENTS)).get(0); - assertEquals(2, event.attempt()); + assertEquals(2, event.get("attempt")); } private static final class CapturingInvoker implements DiagnosisAgentInvoker { diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/InterviewDemoScriptContractTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/InterviewDemoScriptContractTest.java new file mode 100644 index 0000000..bd11efe --- /dev/null +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/InterviewDemoScriptContractTest.java @@ -0,0 +1,47 @@ +package com.superbiz.agent.graph.diagnosis; + +import org.junit.jupiter.api.Test; + +import java.io.IOException; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.List; + +import static org.junit.jupiter.api.Assertions.assertTrue; + +class InterviewDemoScriptContractTest { + + private static final Path SCRIPT = Path.of( + "mvp", "demo", "scripts", "run-interview-demo-check.ps1"); + + @Test + void scriptPinsExactRunAndRequiresOrchestrationTrace() throws IOException { + String source = Files.readString(SCRIPT); + + assertTrue(source.contains("trace?runId=$([System.Uri]::EscapeDataString($runId))")); + assertTrue(source.contains("$traceData.runId -ne $runId")); + assertTrue(source.contains("$traceData.run.runId -ne $runId")); + assertTrue(source.contains("$traceData.run.orchestrationTrace")); + assertTrue(source.contains("data.run.orchestrationTrace is missing")); + + for (String field : List.of( + "version", "final_node", "termination_reason", "transitions", + "degraded", "evidence_retry_count")) { + assertTrue(source.contains("\"" + field + "\""), field); + } + } + + @Test + void scriptWritesStateGraphSummaryFields() throws IOException { + String source = Files.readString(SCRIPT); + + for (String assignment : List.of( + "finalNode = $orchestrationTrace.final_node", + "terminationReason = $orchestrationTrace.termination_reason", + "degraded = [bool]$orchestrationTrace.degraded", + "transitionCount = $transitionCount", + "evidenceRetryCount = [int]$orchestrationTrace.evidence_retry_count")) { + assertTrue(source.contains(assignment), assignment); + } + } +} diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvokerTest.java b/src/test/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvokerTest.java index e173153..ad3f0d1 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvokerTest.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/ReactAgentDiagnosisInvokerTest.java @@ -6,25 +6,47 @@ import org.junit.jupiter.api.Test; import org.springframework.ai.chat.messages.AssistantMessage; import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNotSame; +import static org.junit.jupiter.api.Assertions.assertTrue; import static org.mockito.ArgumentMatchers.eq; -import static org.mockito.ArgumentMatchers.same; +import static org.mockito.ArgumentMatchers.any; import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.verify; import static org.mockito.Mockito.when; +import org.mockito.ArgumentCaptor; + class ReactAgentDiagnosisInvokerTest { @Test - void forwardsExactInputAndRunnableConfig() throws Exception { + void isolatesNestedAgentFromOuterResumeMetadata() throws Exception { ReactAgent agent = mock(ReactAgent.class); - RunnableConfig config = RunnableConfig.builder() + RunnableConfig outerConfig = RunnableConfig.builder() .threadId("run-invoker") + .addMetadata("sessionId", "session-invoker") .addMetadata("runId", "run-invoker") + .resume() .build(); - when(agent.call(eq("projected input"), same(config))) + when(agent.call(eq("projected input"), any(RunnableConfig.class))) .thenReturn(new AssistantMessage("agent output")); DiagnosisAgentInvoker invoker = new ReactAgentDiagnosisInvoker(agent); - assertEquals("agent output", invoker.invoke("projected input", config)); + assertEquals("agent output", invoker.invoke("projected input", outerConfig)); + + ArgumentCaptor captor = ArgumentCaptor.forClass( + RunnableConfig.class); + verify(agent).call(eq("projected input"), captor.capture()); + RunnableConfig nestedConfig = captor.getValue(); + assertNotSame(outerConfig, nestedConfig); + assertEquals("run-invoker", nestedConfig.threadId().orElseThrow()); + assertEquals("session-invoker", + nestedConfig.metadata("sessionId").orElseThrow()); + assertEquals("run-invoker", + nestedConfig.metadata("runId").orElseThrow()); + assertFalse(nestedConfig.metadata( + RunnableConfig.HUMAN_FEEDBACK_METADATA_KEY).isPresent()); + assertTrue(nestedConfig.checkPointId().isEmpty()); } } diff --git a/src/test/java/com/superbiz/agent/graph/diagnosis/ScriptedDiagnosisGraphActions.java b/src/test/java/com/superbiz/agent/graph/diagnosis/ScriptedDiagnosisGraphActions.java index 85e2ec1..7df7d49 100644 --- a/src/test/java/com/superbiz/agent/graph/diagnosis/ScriptedDiagnosisGraphActions.java +++ b/src/test/java/com/superbiz/agent/graph/diagnosis/ScriptedDiagnosisGraphActions.java @@ -125,7 +125,7 @@ final class ScriptedDiagnosisGraphActions { name, step.outcome(), step.reasonCode(), - calls))); + calls).toMap())); return CompletableFuture.completedFuture(update); } } diff --git a/src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java b/src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java deleted file mode 100644 index 791fb75..0000000 --- a/src/test/java/com/superbiz/agent/service/ToolTraceSummaryServiceTest.java +++ /dev/null @@ -1,209 +0,0 @@ -package com.superbiz.agent.service; - -import com.superbiz.agent.domain.entity.ToolInvocation; -import com.superbiz.agent.repository.ToolInvocationRepository; -import org.junit.jupiter.api.Test; - -import java.util.List; -import java.util.Map; - -import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertFalse; -import static org.junit.jupiter.api.Assertions.assertTrue; -import static org.mockito.Mockito.mock; -import static org.mockito.Mockito.when; - -class ToolTraceSummaryServiceTest { - - @Test - void buildVerifierTraceSummaryForRunUsesRunScopedToolRows() { - ToolInvocationRepository repository = mock(ToolInvocationRepository.class); - when(repository.findByRunIdOrderByIdAsc("run-summary-1")).thenReturn(List.of( - ToolInvocation.builder() - .id(101L) - .sessionId("session-1") - .runId("run-summary-1") - .toolName("query_metrics") - .inputParams("{\"query\":\"active_prometheus_alerts\"}") - .outputPreview("active=50 max=50") - .retrievalDetails("{\"retrieved_domains\":[\"prometheus_alerts\"],\"evidence_status\":\"supported\"}") - .success(true) - .build() - )); - - ToolTraceSummaryService service = new ToolTraceSummaryService(repository); - - List> summaries = service.buildVerifierTraceSummaryForRun( - "run-summary-1", "active=50 max=50"); - - assertEquals(1, summaries.size()); - assertEquals("query_metrics", summaries.get(0).get("tool_name")); - assertEquals(List.of(101L), summaries.get(0).get("source_invocation_ids")); - } - - @Test - void buildVerifierTraceSummaryTreatsNoEvidenceAsGapWithoutLosingSuccessfulEvidence() { - ToolInvocationRepository repository = mock(ToolInvocationRepository.class); - when(repository.findBySessionIdOrderByIdAsc("session-1")).thenReturn(List.of( - ToolInvocation.builder() - .id(1L) - .sessionId("session-1") - .toolName("query_logs") - .inputParams("{\"query\":\"timeout\"}") - .outputPreview("payment timeout stack trace") - .retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"supported\"}") - .success(true) - .build(), - ToolInvocation.builder() - .id(2L) - .sessionId("session-1") - .toolName("query_logs") - .inputParams("{\"query\":\"timeout\"}") - .outputPreview("{\"success\":false,\"message\":\"未找到匹配的日志\"}") - .retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"no_evidence\"}") - .success(true) - .build(), - ToolInvocation.builder() - .id(3L) - .sessionId("session-1") - .toolName("query_metrics") - .inputParams("{\"query\":\"active_prometheus_alerts\"}") - .errorMessage("prometheus timeout") - .retrievalDetails("{\"retrieved_domains\":[\"prometheus_alerts\"],\"evidence_status\":\"failed\"}") - .success(false) - .build() - )); - - ToolTraceSummaryService service = new ToolTraceSummaryService(repository); - - List> summaries = service.buildVerifierTraceSummary("session-1", "application-logs point to timeout"); - - assertEquals(2, summaries.size()); - - Map logsSummary = summaries.stream() - .filter(item -> "query_logs".equals(item.get("tool_name"))) - .findFirst() - .orElseThrow(); - assertEquals(Boolean.TRUE, logsSummary.get("success")); - assertEquals("direct", logsSummary.get("evidence_level")); - assertEquals(2, logsSummary.get("invocation_count")); - assertEquals(1, logsSummary.get("no_hit_invocation_count")); - assertTrue(String.valueOf(logsSummary.get("output_summary")).contains("payment timeout stack trace")); - - Map metricsSummary = summaries.stream() - .filter(item -> "query_metrics".equals(item.get("tool_name"))) - .findFirst() - .orElseThrow(); - assertEquals(Boolean.FALSE, metricsSummary.get("success")); - assertEquals("none", metricsSummary.get("evidence_level")); - assertEquals(1, metricsSummary.get("failed_invocation_count")); - assertTrue(String.valueOf(metricsSummary.get("output_summary")).contains("call failed")); - } - - @Test - void buildVerifierTraceSummaryPreservesConcreteFactsFromTruncatedLogAndMetricRows() { - ToolInvocationRepository repository = mock(ToolInvocationRepository.class); - String logPreview = """ - { - "success" : true, - "logs" : [ { - "timestamp" : "2026-07-06 22:15:45", - "level" : "ERROR", - "service" : "order-service", - "message" : "数据库连接池耗尽: Cannot acquire connection from pool, active: 50/50, waiting: 23, timeout: 30000ms" - } ] - } - """; - String metricPreview = """ - { - "success" : true, - "alerts" : [ { - "alert_name" : "HighCPUUsage", - "service" : "payment-service", - "description" : "服务 payment-service 的 CPU 使用率持续超过 80%,当前值为 92%。" - } ] - } - """; - when(repository.findBySessionIdOrderByIdAsc("session-2")).thenReturn(List.of( - ToolInvocation.builder() - .id(10L) - .sessionId("session-2") - .toolName("query_logs") - .inputParams("{\"query\":\"pool\"}") - .outputPreview(logPreview) - .retrievalDetails("{\"retrieved_domains\":[\"application-logs\"],\"evidence_status\":\"supported\"}") - .isTruncated(true) - .success(true) - .build(), - ToolInvocation.builder() - .id(11L) - .sessionId("session-2") - .toolName("query_metrics") - .inputParams("{\"query\":\"active_prometheus_alerts\"}") - .outputPreview(metricPreview) - .retrievalDetails("{\"retrieved_domains\":[\"prometheus_alerts\"],\"evidence_status\":\"supported\"}") - .isTruncated(true) - .success(true) - .build() - )); - - ToolTraceSummaryService service = new ToolTraceSummaryService(repository); - - List> summaries = service.buildVerifierTraceSummary("session-2", "连接池耗尽 HighCPUUsage"); - - Map logsSummary = summaries.stream() - .filter(item -> "query_logs".equals(item.get("tool_name"))) - .findFirst() - .orElseThrow(); - assertTrue(String.valueOf(logsSummary.get("output_summary")).contains("连接池耗尽")); - assertTrue(String.valueOf(logsSummary.get("output_summary")).contains("active: 50/50")); - assertEquals(List.of(10L), logsSummary.get("source_invocation_ids")); - - Map metricsSummary = summaries.stream() - .filter(item -> "query_metrics".equals(item.get("tool_name"))) - .findFirst() - .orElseThrow(); - assertTrue(String.valueOf(metricsSummary.get("output_summary")).contains("HighCPUUsage")); - assertTrue(String.valueOf(metricsSummary.get("output_summary")).contains("payment-service")); - } - - @Test - void buildVerifierTraceSummaryDoesNotTreatGenericMockLogsAsDirectEvidence() { - ToolInvocationRepository repository = mock(ToolInvocationRepository.class); - String genericLogPreview = """ - { - "success" : true, - "logs" : [ { - "timestamp" : "2026-07-06 23:44:41", - "level" : "ERROR", - "service" : "generic-service", - "message" : "日志消息 #0, 查询条件: service:payment-service" - } ] - } - """; - when(repository.findBySessionIdOrderByIdAsc("session-3")).thenReturn(List.of( - ToolInvocation.builder() - .id(20L) - .sessionId("session-3") - .toolName("query_logs") - .inputParams("{\"query\":\"service:payment-service\"}") - .outputPreview(genericLogPreview) - .retrievalDetails("{\"retrieved_domains\":[\"system-metrics\"],\"evidence_status\":\"supported\"}") - .success(true) - .build() - )); - - ToolTraceSummaryService service = new ToolTraceSummaryService(repository); - - List> summaries = service.buildVerifierTraceSummary("session-3", "payment-service timeout"); - - Map logsSummary = summaries.stream() - .filter(item -> "query_logs".equals(item.get("tool_name"))) - .findFirst() - .orElseThrow(); - assertEquals(Boolean.FALSE, logsSummary.get("success")); - assertEquals("none", logsSummary.get("evidence_level")); - assertEquals(1, logsSummary.get("no_hit_invocation_count")); - assertTrue(String.valueOf(logsSummary.get("output_summary")).contains("日志消息 #0")); - } -}