diff --git a/devflow/index.md b/devflow/index.md index f1c24d4..6a64a82 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -4,6 +4,7 @@ | 日期 | slug | 说明 | 领域 | 关键词 | 关联 OpenSpec | 状态 | |---|---|---|---|---|---|---| +| 2026-07-21 | single-react-tool-invocation-store | 建立统一 ToolBoundary 与 Redis canonical invocation store,集中生命周期、证据状态、TTL、容量和 Run 所有权。 | Harness/Tool boundary/Canonical store | ISS-014, ToolBoundary, canonical invocation, PROJECTING, READY, ERROR, TTL, RESULT_TOO_LARGE | openspec/changes/archive/2026-07-21-single-react-tool-invocation-store | archived | | 2026-07-21 | single-react-harness-run-context | 建立显式 RunContext、Harness Core、预算、取消、类型化重试和 Tool Store 基础。 | Harness/Run lifecycle/Budget | ISS-014, RunContext, deadline, cancellation, budget, retry, ToolCallKey | openspec/changes/archive/2026-07-21-single-react-harness-run-context | archived | | 2026-07-21 | single-react-aci-tool-contracts | 冻结 RAG、日志和 MySQL evidence Tool 的 Agent-facing ACI Schema、状态、框架调用引用和描述边界。 | Harness/Agent Tool contract | ISS-014, ACI, tool_call_id, evidence_status, RAG, query_logs, query_mysql, MOCK | openspec/changes/archive/2026-07-21-single-react-aci-tool-contracts | archived | | 2026-07-21 | single-react-design-freeze | 冻结单体 Diagnosis Agent、Harness、Guard、工具证据与阶段门禁契约。 | Chat/Harness/Agent contract | ISS-014, single ReactAgent, Harness, EvidenceGuard, SemanticGuard, tool_call_id, evidence_status | openspec/changes/archive/2026-07-21-single-react-design-freeze | archived | diff --git a/devflow/projects/2026-07-21-single-react-tool-invocation-store/acceptance.md b/devflow/projects/2026-07-21-single-react-tool-invocation-store/acceptance.md new file mode 100644 index 0000000..a38db7d --- /dev/null +++ b/devflow/projects/2026-07-21-single-react-tool-invocation-store/acceptance.md @@ -0,0 +1,43 @@ +# Acceptance: single-react-tool-invocation-store + +## 实现结果 + +- 新增 canonical record/limits/exceptions/store interface/Redis JSON adapter。 +- 新增 ToolBoundary envelope/result、executor/projector interfaces 和稳定错误码。 +- 实现 preflight、Run/Tool budget、Run capacity、PROJECTING、READY/ERROR、evidence semantics、TTL 和 UTF-8 limits。 +- 旧 ToolInvocationRecorder、JPA、Chat/AIOps、Controller、Repository 和公开协议未修改。 + +## 静态验证 + +- `openspec validate single-react-tool-invocation-store --strict`:通过。 +- 新 Harness 包 Redis 引用仅为 `RedisCanonicalInvocationStore`。 +- legacy recorder/JPA/Chat/AIOps/Controller/repository/resources diff:为空。 +- staged diff check 与 Secret scan:提交前执行并通过。 + +## 脚本验证 + +- `mvn -q -DskipTests compile`:通过。 +- `mvn -q '-Dtest=CanonicalInvocationStoreTest,ToolBoundaryTest' test`:通过。 +- 综合阶段 0/1/2/3A suite(含 Harness/ACI/Core/retry/key/config/ChatController):通过。 + +## 浏览器/人工验证 + +- 不适用。本阶段没有 UI、Controller、SSE 或公开协议变化。 + +## 未验证 + +- 未连接真实 Redis,未做 ACL/network/TTL live 验证;最终 E2E 阶段执行。 +- 未接入 Alibaba ToolInterceptor,真实框架 ID 传播留给后续 Agent/application stage。 +- 未实现 RAG/log/MySQL projector,留给 3B/3C。 +- canonical update 的 read-TTL-write 并发窗口已记录为风险,尚未 Lua/CAS 化。 + +## 剩余风险与后续门禁 + +- 新旧 JPA audit 与 canonical store 短期并存,后续 projector 必须只以新 boundary 的 READY record 作为引用来源。 +- 下一阶段 3B/3C 必须复用本 ToolBoundary,不复制 Redis 状态机。 + +## 状态 + +- Stage acceptance: accepted +- OpenSpec archive: archived at `openspec/changes/archive/2026-07-21-single-react-tool-invocation-store` +- Main spec sync: `openspec/specs/canonical-tool-invocation-store/spec.md`(7 added requirements) diff --git a/devflow/projects/2026-07-21-single-react-tool-invocation-store/brief.md b/devflow/projects/2026-07-21-single-react-tool-invocation-store/brief.md new file mode 100644 index 0000000..125db4d --- /dev/null +++ b/devflow/projects/2026-07-21-single-react-tool-invocation-store/brief.md @@ -0,0 +1,29 @@ +# Brief: single-react-tool-invocation-store + +## 背景 + +旧 `ToolInvocationRecorder` 依赖 ThreadLocal 和 JPA preview,不能证明完整 Tool 结果、生命周期和当前 Run 所有权。阶段 2 已提供 RunContext/Key/Capacity,需要统一 canonical ToolBoundary 和 Redis store 供后续 projector 复用。 + +## 目标 + +- 统一 Pre-Tool 门禁、PROJECTING/READY/ERROR 状态和 evidence semantics。 +- 在同一 Redis record 保存 request/raw_response/agent_result、框架 ID、Run、时间和错误。 +- 固定 TTL 不续期、容量/结果大小 fail-closed、raw 不静默截断。 +- 用 Fake Tool/Projector/Store 覆盖 duplicate、cross-run、no-evidence、error、TTL 和 oversize。 + +## 范围 + +- Canonical invocation model/store、Redis JSON adapter、ToolBoundary 和 focused tests。 +- 复用阶段 2 Core、budget、capacity、ToolCallKeyFactory。 + +## 非目标 + +- 不实现 RAG/log/MySQL projector。 +- 不修改旧 recorder/JPA、Chat/AIOps、Controller/SSE 或公开协议。 + +## 元数据 + +- 分档:complex +- 接口影响:L2 内部 Harness boundary/store +- 关联 Issue:ISS-014 阶段 3A +- 关联 OpenSpec:`openspec/changes/single-react-tool-invocation-store` diff --git a/devflow/projects/2026-07-21-single-react-tool-invocation-store/decisions.md b/devflow/projects/2026-07-21-single-react-tool-invocation-store/decisions.md new file mode 100644 index 0000000..a6044d2 --- /dev/null +++ b/devflow/projects/2026-07-21-single-react-tool-invocation-store/decisions.md @@ -0,0 +1,106 @@ +# Decisions: single-react-tool-invocation-store + +## 规模与入口 + +- 分档:complex。 +- 入口:ISS-014 阶段 3A;阶段 0/1/2 已 Archive 并由 `58c3910`、`4274f33`、`6b74990` 提交。 +- 目标:统一 ToolBoundary 和 Redis canonical invocation store,不实现 Tool-specific projection。 + +## Context + +- 阶段 1 主规格已冻结 RAG/log/MySQL Agent-facing Request/Result 和 `EvidenceStatus`。 +- 阶段 2 主规格已冻结 `RunContext`、预算、取消、Key Factory 和 strict retry。 +- 旧 `ToolInvocationRecorder` 依赖 `SessionContextHolder`、JPA `ToolInvocation` 和 500 字符 preview,属于 durable audit 兼容路径,不是 canonical store。 +- 既有 `SessionConfiguration` 提供 `RedisTemplate` + JSON serializer;新 store 复用 bean,不新增连接配置。 + +## Question Pool + +| 维度 | 问题 | 模式 | 证据与结论 | 状态 | +|---|---|---|---|---| +| 术语 | canonical invocation 与旧 JPA ToolInvocation 是否同一记录? | evidence-driven | ISS-014 数据分层明确 Redis canonical 保存完整 request/raw/agent,JPA 只做 durable audit;两者分离。 | 已解决并汇报 | +| 术语 | `PROJECTING/READY/ERROR` 与 evidence status 如何组合? | evidence-driven | 阶段 0/1 规格:只有 READY 可为 FOUND/NO_EVIDENCE,ERROR 不可引用;PROJECTING 是内部暂态。 | 已解决并汇报 | +| 边界 | 阶段 3A 是否实现 RAG/log projector? | evidence-driven | ISS-014 3A 明确不实现 Tool-specific projection,3B/3C 单独接入。 | 已解决并汇报 | +| 边界 | Redis 读取是否刷新 TTL? | evidence-driven | ISS-014 固定“创建设置、读取/更新不续期”;更新采用当前剩余 TTL,不恢复初始 TTL。 | 已解决并汇报 | +| 验收 | raw 超限是否静默截断? | evidence-driven | ISS-014 明确 `RESULT_TOO_LARGE` ERROR,raw 不静默截断;agent projection 才可按 projector 预算截断并标记。 | 已解决并汇报 | +| 验收 | 缺失/重复/cross-run ID 如何处理? | evidence-driven | 阶段 3A 任务明确覆盖;Key Factory 保留框架 ID,ToolBoundary 在 store 创建前校验 Run/ID 和 duplicate。 | 已解决并汇报 | +| 技术 | Redis 如何避免 update 重置 TTL? | evidence-driven | 既有 RedisTemplate;begin 使用 setIfAbsent + TTL,update 先读取剩余 TTL 再写回相同/更短 TTL,get 不调用 expire。 | 已解决并汇报 | +| 技术 | 是否修改旧 recorder 以复用新 store? | evidence-driven | 旧链路大量测试依赖 JPA preview/evidence_refs;本阶段零消费者,保持旧 recorder 不变,避免行为回归。 | 已解决并汇报 | + +## Grill 结论 + +- 所有术语、边界、验收和技术问题均由 ISS、阶段规格、旧代码和 Redis 配置事实证明。 +- 没有新增产品偏好或兼容性取舍需要 user-interview;并发更新窗口作为已接受风险记录。 +- `grill-with-docs` 的代码可证结论已回写 proposal;没有未确认问题。 + +## 能力与工具限制 + +- Discover 能力来源:`sm-flow` + `grill-with-docs`。 +- 当前无 `codebase-retrieval`/LSP;使用 `rg`、源码、既有测试、本地 Redis 配置和 focused fake tests 进行等价核对。 + +## Cross-artifact 对齐 + +| 链路 | 状态 | 结论 | +|---|---|---| +| brief 目标/范围/非目标 -> proposal | 已对齐 | Boundary、canonical record、TTL/size、Fake tests 和 legacy isolation 全部覆盖。 | +| proposal 范围/约束/承诺 -> design | 已对齐 | Redis value、状态机、preflight 顺序、raw/projection 和失败处理均已设计。 | +| design 决策/接口影响/风险 -> specs/tasks | 已对齐 | L2 boundary、TTL 更新窗口、oversize/error、ID/Run 所有权有对应要求和任务。 | +| specs 可观察行为 -> tasks | 已对齐 | 7 条 requirements 分解为 store、boundary、limits/failure 和隔离验证纵向切片。 | + +## Architecture Audit + +- 能力来源:`zoom-out`,按 RunContext、Invocation Status、Evidence Status、canonical invocation 和 durable audit 术语审计。 +- ToolBoundary 只编排一次调用;CanonicalInvocationStore 独占 Redis 状态转换;DiagnosisHarnessCore 独占 Run budget/cancellation;旧 recorder 只写 JPA audit。 +- request/raw/agent result 归同一 canonical record,Agent 只获得 ToolBoundaryResult,不存在 raw 旁路。 +- Redis adapter 是唯一 Redis 访问点,接口/Fake 不依赖 Redis;后续 3B/3C 可直接复用而不复制状态机。 +- 风险集中在 read-TTL-write 并发窗口和 canonical raw 敏感性,已进入 design/spec/limits,无架构或 ADR 冲突。 + +## Commit Gate Preflight + +- proposal、design、specs、tasks 完整,OpenSpec status complete,strict validation 通过。 +- question pool 无未汇报 evidence-driven 或未确认 user-interview 项。 +- L2 内部接口影响已记录;旧 JPA/Chat/Controller/协议不改。 +- cross-artifact 无 gap,所有错误、TTL、size、ID/Run 所有权和隔离要求可由 Fake tests 验证。 +- Apply 已获持续授权,范围只包括新 store/boundary 与 focused tests。 + +## Pre-apply Research + +### 参考实现与复用 + +- 复用 `DiagnosisHarnessCore` 的 active/deadline/Tool budget/Run bytes 门禁。 +- 复用 `ToolCallKeyFactory` 精确保留框架 Tool Call ID 并隔离 Run key。 +- 复用阶段 0 `InvocationStatus` / `EvidenceStatus`,不创建字符串状态副本。 +- 复用 `SessionConfiguration` 提供的 `RedisTemplate` 和 Spring Boot ObjectMapper。 +- 旧 `ToolInvocationRecorder`/JPA preview 仅作为 durable audit 反例,本阶段不修改或调用。 + +### 技术栈清单 + +- canonical record:Java 17 record + Jackson JSON String,所有状态组合在 record transition 方法中校验。 +- Redis create:`ValueOperations.setIfAbsent` + TTL;update:读取当前 remaining TTL 后写回;get 不调用 expire。 +- 大小:UTF-8 bytes;record/agent result/store limits 与 RunContext 累计 capacity 双重门禁。 +- 测试:Mockito RedisTemplate/ValueOperations 验证 TTL API;In-memory fake store + Fake Tool/Projector 验证 boundary,不连接外部 Redis。 + +### 新建类型 + +- canonical model/limits/store/exceptions/Redis adapter。 +- ToolCallRequestEnvelope、ToolBoundaryResult、ProjectedToolResult、ToolExecutor、ToolResultProjector、ToolBoundary 和稳定 error codes。 +- Redis adapter tests 与 boundary fake tests。 + +### 影响半径 + +- 新 package 在阶段 3A 保持零现有消费者。 +- Redis 访问只允许出现在 `RedisCanonicalInvocationStore`;旧 Chat/AIOps/Controller/JPA/Recorder 不修改。 + +## Apply 结果 + +- 冲突分类:一次 focused test 断言错误(duplicate 场景应调用 `setIfAbsent` 两次)已修正并复跑通过;无规格偏离。 +- 新增 canonical record/limits/store exceptions、Redis JSON adapter、ToolBoundary envelope/result/interfaces 和 stable error codes。 +- ToolBoundary 已实现 preflight、Core budget/capacity、PROJECTING、raw/projection size、READY/ERROR 和安全返回边界;未接具体 projector。 +- 首模块对齐:store/boundary/limits/failure 与 design/tasks 全部完成;RAG/log/MySQL adapters 仍留给后续阶段。 + +## Apply 验证 + +- 编译:`mvn -q -DskipTests compile`:通过。 +- Store/Boundary focused:`mvn -q '-Dtest=CanonicalInvocationStoreTest,ToolBoundaryTest' test`:通过。 +- 综合回归:`mvn -q '-Dtest=CanonicalInvocationStoreTest,ToolBoundaryTest,HarnessContractTest,RagToolContractTest,QueryLogsToolContractTest,MysqlToolContractTest,RunContextTest,RunBudgetTest,DiagnosisHarnessCoreTest,HarnessRetryExecutorTest,ToolCallKeyFactoryTest,SpringAiRetryConfigurationTest,ChatControllerTest' test`:通过。 +- 静态隔离:新 Harness 包仅 `RedisCanonicalInvocationStore` 引用 Redis;legacy recorder/JPA/Chat/AIOps/Controller/repository/resources diff 为空。 +- OpenSpec:`openspec validate single-react-tool-invocation-store --strict`:通过。 diff --git a/devflow/projects/2026-07-21-single-react-tool-invocation-store/evidence.md b/devflow/projects/2026-07-21-single-react-tool-invocation-store/evidence.md new file mode 100644 index 0000000..78b1f6d --- /dev/null +++ b/devflow/projects/2026-07-21-single-react-tool-invocation-store/evidence.md @@ -0,0 +1,20 @@ +# Evidence: single-react-tool-invocation-store + +## 文档与代码证据 + +- ISS-014 阶段 3A 明确要求统一 ToolBoundary、canonical invocation、PROJECTING/READY/ERROR、TTL/容量、ID/Run 所有权,且不实现 3B/3C projector。 +- 阶段 2 已提供 `RunContext`、Run bytes capacity 和 `ToolCallKeyFactory`,本阶段直接复用。 +- 旧 `ToolInvocationRecorder` 使用 JPA preview 和 ThreadLocal fallback;新 canonical record 独立保存完整 request/raw/agent,不修改旧 recorder/JPA。 +- 既有 `SessionConfiguration` 提供 `RedisTemplate` JSON bean;`RedisCanonicalInvocationStore` 是新 Harness 包唯一 Redis 引用。 + +## Evidence-driven 结论 + +- `setIfAbsent` 确保同一 `runId+toolCallId` 不覆盖;读取不调用 expire;更新使用剩余 TTL。 +- canonical record transition 只允许 PROJECTING -> READY/ERROR;READY 只接受 FOUND/NO_EVIDENCE,ERROR 不可引用。 +- ToolBoundary 在执行前校验 Run、ID、JSON、授权、只读和预算;raw 只在可信 store 保存,不返回 Agent。 +- UTF-8 record/Agent limits 与 Run capacity 双门禁;raw oversize 跳过 projector,Agent oversize 不返回,均产生 RESULT_TOO_LARGE。 +- Fake store/Redis mock tests 已覆盖 duplicate、cross-run、unauthorized、writable、execution/projection error、NO_EVIDENCE、TTL、raw/agent oversize。 + +## 工具限制 + +- 当前无 `codebase-retrieval`/LSP;使用 `rg`、源码、Maven 编译、Mockito Redis API 和 in-memory fake 完成等价验证。 diff --git a/mvp/issues/active/ISS-014-single-react-agent-harness-aci-ptk-refactor.md b/mvp/issues/active/ISS-014-single-react-agent-harness-aci-ptk-refactor.md index 0baae9d..69f6dbd 100644 --- a/mvp/issues/active/ISS-014-single-react-agent-harness-aci-ptk-refactor.md +++ b/mvp/issues/active/ISS-014-single-react-agent-harness-aci-ptk-refactor.md @@ -1,6 +1,6 @@ # ISS-014 单体 ReAct Agent、Harness 与 ACI 工具瘦身 -**状态**:实施中(阶段 0-2 已归档,下一阶段 3) +**状态**:实施中(阶段 0-3A 已归档,下一阶段 3B) **严重程度**:高 **发现时间**:2026-07-20 **目标分支**:`refactor/chat-single-react-harness` diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.archive-ready b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.archive-ready new file mode 100644 index 0000000..746db3b --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.archive-ready @@ -0,0 +1 @@ +Archive-ready after implementation and focused verification on 2026-07-21. diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.committed b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.committed new file mode 100644 index 0000000..c501e20 --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.committed @@ -0,0 +1 @@ +Committed after strict validation on 2026-07-21. diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.openspec.yaml b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.openspec.yaml new file mode 100644 index 0000000..c0a8162 --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-21 diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/design.md b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/design.md new file mode 100644 index 0000000..478e068 --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/design.md @@ -0,0 +1,102 @@ +## Context + +当前 `ToolInvocationRecorder` 将 JPA `ToolInvocation` 作为旧链路的 durable audit,保存 `output_preview`(500 字符)和 retrieval details,并通过 `SessionContextHolder` 补 session/run。它不能作为 EvidenceGuard 的 canonical source:raw response 已被截断、生命周期没有 PROJECTING/READY/ERROR、没有框架 Tool Call ID,也没有单 Run 容量/TTL 门禁。 + +阶段 2 已提供 `RunContext`、budget、capacity 和 `ToolCallKeyFactory`。本阶段需把所有 Tool 共用的执行边界与 canonical invocation store 建起来,供阶段 3B/3C 的 projector 直接复用。 + +## Goals / Non-Goals + +**Goals:** + +- 在 Tool 调用前统一校验 Run 所有权、框架 ID、JSON object、授权、只读和 Tool/Run budget。 +- 保存同一 canonical record 的完整 request/raw_response/agent_result 及状态、证据语义、时间和错误。 +- 集中执行 PROJECTING -> READY/ERROR,防止未经投影的 raw 进入 Agent。 +- 固定 TTL 不续期、单记录/Agent result/Run bytes 上限和 RESULT_TOO_LARGE 语义。 +- 提供 Redis 实现和 store-independent Fake boundary tests。 + +**Non-Goals:** + +- 不实现 RAG/log/MySQL specific projector 或 adapter。 +- 不修改旧 ToolInvocationRecorder/JPA/数据库、Chat/AIOps、Controller/SSE。 +- 不让 Agent 访问 Redis/client/key/raw record。 +- 不实现并发 Lua/CAS 更新、永久 audit、脱敏或跨 Run 查询。 + +## Decisions + +### 1. Internal ToolBoundary envelope and result + +`ToolCallRequestEnvelope` 是 Harness 内部输入,包含 `runId`、框架 `toolCallId`、toolName、JSON request、`authorized` 和 `readOnly`。Agent-facing DTO 不暴露这些门禁字段;后续 Alibaba ToolInterceptor 负责组装 envelope。 + +`ToolBoundaryResult` 只返回 invocation status、evidence status、原始框架 ID、bounded `agentResult` 或安全 errorCode;raw 只进入 store,不返回 Agent。 + +### 2. Canonical record and lifecycle + +`CanonicalToolInvocation` 是 JSON serializable record,字段包含 toolCallId、runId、toolName、request、rawResponse、agentResult、InvocationStatus、EvidenceStatus、errorCode、startedAt、completedAt。`begin` 只接受 PROJECTING;`markReady` 只接受 PROJECTING + 非空 bounded projection + FOUND/NO_EVIDENCE;`markError` 将 evidence status 固定为 ERROR。 + +状态转换和 duplicate 检查在 `CanonicalInvocationStore` 内集中执行。ToolBoundary 不直接写 Redis。 + +### 3. Redis value and TTL + +Redis 实现复用现有 `RedisTemplate`,将 record 序列化为 JSON String。创建使用 `setIfAbsent(key,json,ttl)`,保证同一 `runId+toolCallId` 不覆盖;读取不调用 expire。更新先读取剩余毫秒 TTL,再用不大于该值的 TTL 写回,避免恢复初始 TTL。过期/缺失读取返回 empty。 + +替代方案是 Redis Hash;单 JSON value 能保证 request/raw/agent 原子同记录,并让 Fake/序列化 schema 与 canonical record 一致,故采用。并发 projector 更新窗口是已接受风险,后续需要时再升级 Lua/CAS。 + +### 4. Size and status rules + +`CanonicalInvocationLimits` 由 caller 提供 TTL、maxRecordBytes 和 maxAgentResultBytes。request/raw/agent 使用 UTF-8 bytes 计数;raw 超过 record 或 agent projection 超过独立上限,均不截断,记录 ERROR/RESULT_TOO_LARGE。Run bytes 通过阶段 2 Core 再做单 Run 累计门禁。 + +PROJECTING 时 evidence status 仅作为内部未知/ERROR 占位;READY 只接受 `EVIDENCE_FOUND` 或 `NO_EVIDENCE`;ERROR 永远不可引用。NO_EVIDENCE 不触发重试或成功解释。 + +### 5. Preflight and projector boundary + +ToolBoundary 顺序固定: + +```text +RunContext active/deadline + -> runId + toolCallId + authorization + read-only + JSON object + -> Core.beforeToolCall + request/run capacity + -> store.begin(PROJECTING) + -> ToolExecutor(raw) + -> raw size/capacity + -> ToolResultProjector(agentResult,evidenceStatus) + -> agent size/capacity + -> store.markReady or markError + -> bounded ToolBoundaryResult +``` + +执行或投影异常都会写 ERROR;raw 已在可信边界且未超限时保留在 canonical record,但不返回 Agent。preflight/duplicate/cross-run 错误在 begin 前返回安全 ERROR。 + +### 6. Old audit separation + +旧 recorder/JPA 继续接收旧 Tool 调用,阶段 3A 不改其字段和 ThreadLocal fallback。新 canonical store 没有旧消费者;阶段 3B/3C 接入时必须明确写新 boundary,并在需要 durable audit 时另行脱敏摘要。 + +## Module Flow + +```text +Alibaba ToolInterceptor (future) + -> ToolCallRequestEnvelope + -> ToolBoundary + -> DiagnosisHarnessCore + ToolCallKeyFactory + -> CanonicalInvocationStore (Redis JSON / Fake) + -> ToolExecutor + -> ToolResultProjector (future RAG/log/MySQL) + -> bounded ToolBoundaryResult +``` + +## Risks / Trade-offs + +- [Redis update read-TTL-write 存在并发窗口] -> 当前每个 invocation 只允许 boundary 顺序更新;后续并发需求升级 Lua/CAS。 +- [canonical raw 可能敏感] -> 仅 Harness store 访问,TTL/ACL/容量受限;脱敏在 projector/durable audit 阶段处理。 +- [旧 recorder 与新 store 短期并存] -> 包和接口隔离,spec 明确旧 preview 不能作为 canonical evidence。 +- [preflight 失败可能没有 canonical record] -> 返回安全 ERROR 且不执行 Tool;阶段 3A 的可引用记录只针对已通过 begin 的调用。 + +## Migration Plan + +1. 本阶段新增 boundary/store/Redis adapter 和 fake tests,旧运行链路不变。 +2. 阶段 3B/3C 将各 Tool adapter/projector 包装到本 boundary。 +3. 阶段 4 Diagnosis Agent 只接收 boundary 的 bounded result。 +4. 阶段 6A/7 再决定 durable audit 如何从 canonical 摘要回填,并清理旧 recorder/ThreadLocal。 + +## Open Questions + +无。真实 Redis 的 ACL、网络和 TTL 由最终运行/E2E 阶段验证;并发更新 Lua 化留作后续需求。 diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/proposal.md b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/proposal.md new file mode 100644 index 0000000..cdc68bf --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/proposal.md @@ -0,0 +1,45 @@ +## Why + +阶段 1 已冻结 Agent-facing Tool Contract,阶段 2 已提供 RunContext、预算、取消和 Tool Call Key 基础,但当前 `ToolInvocationRecorder` 仍把截断 preview 写入 JPA、依赖 ThreadLocal,并没有同一条记录中的 `request/raw_response/agent_result`、生命周期或当前 Run 所有权。RAG、日志和 MySQL 投影若各自保存调用,会重新复制状态机并让 EvidenceGuard 无法证明引用来自当前 Run。 + +## What Changes + +- 新增统一 `ToolBoundary`,在每个 Tool 调用前执行 JSON Schema/只读/Run/预算/Tool Call ID 门禁,执行后统一处理 raw、投影、状态和错误。 +- 新增 `CanonicalInvocationStore` 抽象与 Redis 实现,按阶段 2 Key Factory 保存一条完整 JSON 调用记录:`request`、`raw_response`、`agent_result`、`status`、`evidence_status`、时间和错误信息。 +- `PROJECTING -> READY/ERROR` 生命周期和独立 EvidenceStatus 在 store 中集中执行;READY 才允许 `EVIDENCE_FOUND/NO_EVIDENCE`,ERROR 不可引用。 +- 创建时设置 TTL,读取不刷新;更新只使用当前剩余 TTL,不延长生命周期;单记录、Agent projection 和单 Run 容量超限显式返回 `RESULT_TOO_LARGE`,不静默截断 raw。 +- 拒绝缺失/非法/重复 Tool Call ID、跨 Run 引用、不可解析 JSON、非只读请求和已超预算调用;不生成第二套 ID。 +- 使用 Fake Tool/Projector/In-memory Store 覆盖成功、no-evidence、projection error、execution error、duplicate/cross-run、TTL、容量和 raw oversize。 +- 本阶段不实现 RAG/log/MySQL specific projector,不修改旧 `ToolInvocationRecorder`、JPA entity、Controller、ChatService 或公开协议。 + +## Capabilities + +### New Capabilities + +- `canonical-tool-invocation-store`: 提供统一 ToolBoundary、canonical invocation 生命周期、Run 所有权、容量/TTL 和可引用状态边界,供后续 RAG/log/MySQL 投影复用。 + +### Modified Capabilities + +- None. 旧 JPA audit 记录继续服务旧链路;新 store 先作为零消费者 Harness foundation。 + +## Context Constraints + +- canonical store 只能由 Harness/ToolBoundary 访问,Agent 不获得 Redis client/key/raw record。 +- Redis key 固定由阶段 2 `ToolCallKeyFactory` 生成:`prefix:runId:toolCallId`。 +- 同一调用的完整 request/raw/agent projection 必须在同一记录;raw 不能只保存 preview,也不能未经 projector 返回 Agent。 +- 创建 TTL 默认配置由 caller 提供且必须大于 0;读取与更新不得续期。 +- `PROJECTING` 时 evidence_status 只能是内部暂态 ERROR/unknown;只有 READY 才能成为 `EVIDENCE_FOUND` 或 `NO_EVIDENCE`。 +- `NO_EVIDENCE` 仅作为结果语义,不可被 boundary 自动升级为成功事实或重试。 + +## Interface Impact + +- 等级:L2(内部 Harness/Tool boundary)。新增接口会被阶段 3B/3C 直接消费,旧调用方不变。 +- 不改变 JPA `tool_invocation`、数据库 Schema、旧审计 preview 或公开 HTTP/SSE。 +- Redis 是新增运行时依赖使用既有 `RedisTemplate` bean;真实连接验证留给阶段 7,focused tests 使用 fake/mocks。 + +## Risks + +- Redis JSON value 更新需要读取剩余 TTL 后再写回,存在并发更新窗口;当前单 Tool Call 只有 boundary 状态机写入,后续若并发 projector 必须升级 Lua/CAS。 +- canonical raw 可包含敏感内容;本 Issue 保留阶段 0 已确认的 Harness-only ACL/TTL 约束,持久化脱敏和 durable audit 留给后续阶段。 +- ToolBoundary 同时负责预算、store 状态和 projector 错误,若异常分类不清会产生错误状态;每个边界分支都有 Fake tests。 +- 当前旧 recorder 继续运行,新旧两条 audit 链短期并存;proposal 明确禁止把旧 JPA 记录当 canonical evidence。 diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/specs/canonical-tool-invocation-store/spec.md b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/specs/canonical-tool-invocation-store/spec.md new file mode 100644 index 0000000..319728d --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/specs/canonical-tool-invocation-store/spec.md @@ -0,0 +1,74 @@ +## ADDED Requirements + +### Requirement: ToolBoundary SHALL enforce explicit preflight before execution +The Harness SHALL reject a Tool call before invoking the executor when the envelope has a blank/unsafe framework `tool_call_id`, a run ID different from RunContext, invalid JSON object input, unauthorized access, non-read-only access, an inactive/deadline-expired Run, duplicate canonical key, or exhausted Tool/Run budget. + +#### Scenario: Cross-Run Tool Call is rejected +- **WHEN** an envelope run ID differs from the explicit RunContext run ID +- **THEN** ToolBoundary returns a safe `ERROR`, does not create a canonical record, and does not invoke the Tool + +#### Scenario: Unauthorized or writable Tool is rejected +- **WHEN** `authorized=false` or `readOnly=false` +- **THEN** ToolBoundary returns `ERROR` before execution and does not expose the request to an Agent + +### Requirement: Canonical invocation SHALL keep complete data in one record +The store SHALL save the complete request JSON, raw Tool response, bounded Agent result, framework `tool_call_id`, run ID, tool name, lifecycle status, evidence status, timestamps, and error code in one canonical record. The raw response SHALL NOT be replaced by a preview, and SHALL NOT be returned to the Agent. + +#### Scenario: Tool begins execution +- **WHEN** preflight succeeds and the Tool is about to execute +- **THEN** one record is created with `status=PROJECTING`, the complete request, the exact framework ID, and no Agent result yet + +#### Scenario: Projector succeeds +- **WHEN** the executor returns raw data and the projector returns a bounded result +- **THEN** the same record contains raw data and Agent result with `status=READY`, and ToolBoundary returns only the bounded Agent result + +### Requirement: Invocation lifecycle and evidence semantics SHALL be enforced centrally +The store SHALL allow only `PROJECTING -> READY` or `PROJECTING -> ERROR`. READY SHALL require `EVIDENCE_FOUND` or `NO_EVIDENCE`; ERROR SHALL use `evidence_status=ERROR` and SHALL NOT be referencable. `NO_EVIDENCE` SHALL remain a scoped negative observation and SHALL NOT be upgraded to success or retried by the boundary. + +#### Scenario: No evidence projection completes +- **WHEN** a projector returns a valid bounded result with `NO_EVIDENCE` +- **THEN** the record becomes `READY`, preserves the result scope, and remains eligible only for a negative observation + +#### Scenario: Projection fails +- **WHEN** the projector throws or returns an invalid evidence status +- **THEN** the same record becomes `ERROR`, stores a safe error code, and no Agent result is returned + +### Requirement: Tool Call ID and Run ownership SHALL be preserved +The boundary SHALL use the exact framework `tool_call_id` with the RunContext run ID to create the canonical key. It SHALL reject missing, unsafe, duplicate, and cross-Run references and SHALL never generate or replace a fallback ID. + +#### Scenario: Duplicate Tool Call ID is submitted +- **WHEN** a second invocation uses the same valid run ID and framework Tool Call ID +- **THEN** the second Tool is not executed and returns `ERROR` without overwriting the first record + +#### Scenario: Framework ID is preserved +- **WHEN** a valid envelope passes preflight +- **THEN** the key and canonical record contain the exact supplied `tool_call_id` + +### Requirement: TTL and size limits SHALL fail closed +The Redis implementation SHALL set TTL only when a canonical record is created. Reads SHALL NOT refresh TTL; updates SHALL preserve only the current remaining TTL. Request/raw/record and Agent-result byte limits SHALL be enforced in UTF-8. Raw overflow SHALL produce `ERROR/RESULT_TOO_LARGE` without silent truncation; Agent projection overflow SHALL produce the same error. + +#### Scenario: Read does not renew TTL +- **WHEN** a canonical record is read before expiration +- **THEN** its expiry remains at or before the original expiry and no expire/refresh operation is issued + +#### Scenario: Raw response is too large +- **WHEN** the executor returns raw data beyond the record limit +- **THEN** the record becomes `ERROR` with `RESULT_TOO_LARGE`, the raw payload is not silently truncated, and the projector is not invoked + +#### Scenario: Agent result is too large +- **WHEN** a projector returns a result beyond the Agent projection limit +- **THEN** the record becomes `ERROR/RESULT_TOO_LARGE` and the oversized result is not returned to the Agent + +### Requirement: Tool and store failures SHALL return safe errors +Execution, serialization, Redis, duplicate, preflight, and projection failures SHALL return a bounded error code without raw payload, credentials, internal stack trace, or Redis key to the Agent. If raw data was safely stored before a projection failure, it remains Harness-only canonical data. + +#### Scenario: Tool execution throws +- **WHEN** the executor raises an exception after PROJECTING begins +- **THEN** the record becomes `ERROR` with a stable execution error code and ToolBoundary returns no raw response + +### Requirement: Stage 3A SHALL remain reusable and independent from legacy audit +The boundary and canonical store SHALL be usable by later RAG/log/MySQL projectors without modifying legacy `ToolInvocationRecorder`, JPA `ToolInvocation`, Chat/AIOps, Controller, or public protocol. Redis access SHALL remain inside the store implementation and no Agent-facing API SHALL expose the Redis client. + +#### Scenario: Fake projector tests pass +- **WHEN** Fake Tool/Projector tests cover success, no evidence, error, duplicate, TTL, and capacity cases +- **THEN** later Tool-specific changes can depend on the same boundary without copying lifecycle or Redis logic, while legacy call sites remain unchanged diff --git a/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/tasks.md b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/tasks.md new file mode 100644 index 0000000..61f9456 --- /dev/null +++ b/openspec/changes/archive/2026-07-21-single-react-tool-invocation-store/tasks.md @@ -0,0 +1,24 @@ +## 1. Canonical Model and Store + +- [x] 1.1 Implement CanonicalToolInvocation, CanonicalInvocationLimits, store exceptions, and lifecycle/evidence transition validation. +- [x] 1.2 Implement CanonicalInvocationStore and Redis JSON-value adapter with set-if-absent creation, complete record persistence, and remaining-TTL updates. +- [x] 1.3 Add store tests for duplicate creation, PROJECTING/READY/ERROR transitions, evidence status rules, and read-without-TTL-refresh semantics. + +## 2. Tool Boundary + +- [x] 2.1 Implement ToolCallRequestEnvelope, ToolBoundaryResult, ProjectedToolResult, executor/projector interfaces, and safe error codes. +- [x] 2.2 Implement ToolBoundary preflight for Run ownership, framework ID, JSON object, authorization, read-only, budget, and duplicate gates. +- [x] 2.3 Implement execution -> raw size/capacity -> projection -> Agent size/capacity -> READY/ERROR flow without returning raw data. +- [x] 2.4 Add Fake Tool/Projector boundary tests for success, NO_EVIDENCE, execution/projection errors, duplicate/cross-run/unauthorized/writable requests. + +## 3. Limits and Failure Semantics + +- [x] 3.1 Enforce UTF-8 request/raw/record/Agent-result limits and `RESULT_TOO_LARGE` without silent raw truncation. +- [x] 3.2 Add tests proving raw overflow skips projector, Agent overflow is not returned, and Run capacity remains consistent. +- [x] 3.3 Verify ERROR records cannot become referencable READY evidence and NO_EVIDENCE remains scoped negative observation. + +## 4. Verification and Isolation + +- [x] 4.1 Run focused canonical store and ToolBoundary tests with fake store/Redis operations. +- [x] 4.2 Run stage 0/1/2 contract, Core, retry, key, and ChatController regression tests. +- [x] 4.3 Verify legacy ToolInvocationRecorder/JPA, Chat/AIOps, Controller, and public protocol files are unchanged; Redis access is confined to the new store adapter. diff --git a/openspec/specs/canonical-tool-invocation-store/spec.md b/openspec/specs/canonical-tool-invocation-store/spec.md new file mode 100644 index 0000000..55f96aa --- /dev/null +++ b/openspec/specs/canonical-tool-invocation-store/spec.md @@ -0,0 +1,78 @@ +# canonical-tool-invocation-store Specification + +## Purpose +定义 Harness ToolBoundary 与 canonical invocation store 的统一执行边界,包括 Run/Tool preflight、PROJECTING/READY/ERROR 生命周期、evidence status、框架 Tool Call ID、TTL、容量和有界 Agent 投影。 + +## Requirements +### Requirement: ToolBoundary SHALL enforce explicit preflight before execution +The Harness SHALL reject a Tool call before invoking the executor when the envelope has a blank/unsafe framework `tool_call_id`, a run ID different from RunContext, invalid JSON object input, unauthorized access, non-read-only access, an inactive/deadline-expired Run, duplicate canonical key, or exhausted Tool/Run budget. + +#### Scenario: Cross-Run Tool Call is rejected +- **WHEN** an envelope run ID differs from the explicit RunContext run ID +- **THEN** ToolBoundary returns a safe `ERROR`, does not create a canonical record, and does not invoke the Tool + +#### Scenario: Unauthorized or writable Tool is rejected +- **WHEN** `authorized=false` or `readOnly=false` +- **THEN** ToolBoundary returns `ERROR` before execution and does not expose the request to an Agent + +### Requirement: Canonical invocation SHALL keep complete data in one record +The store SHALL save the complete request JSON, raw Tool response, bounded Agent result, framework `tool_call_id`, run ID, tool name, lifecycle status, evidence status, timestamps, and error code in one canonical record. The raw response SHALL NOT be replaced by a preview, and SHALL NOT be returned to the Agent. + +#### Scenario: Tool begins execution +- **WHEN** preflight succeeds and the Tool is about to execute +- **THEN** one record is created with `status=PROJECTING`, the complete request, the exact framework ID, and no Agent result yet + +#### Scenario: Projector succeeds +- **WHEN** the executor returns raw data and the projector returns a bounded result +- **THEN** the same record contains raw data and Agent result with `status=READY`, and ToolBoundary returns only the bounded Agent result + +### Requirement: Invocation lifecycle and evidence semantics SHALL be enforced centrally +The store SHALL allow only `PROJECTING -> READY` or `PROJECTING -> ERROR`. READY SHALL require `EVIDENCE_FOUND` or `NO_EVIDENCE`; ERROR SHALL use `evidence_status=ERROR` and SHALL NOT be referencable. `NO_EVIDENCE` SHALL remain a scoped negative observation and SHALL NOT be upgraded to success or retried by the boundary. + +#### Scenario: No evidence projection completes +- **WHEN** a projector returns a valid bounded result with `NO_EVIDENCE` +- **THEN** the record becomes `READY`, preserves the result scope, and remains eligible only for a negative observation + +#### Scenario: Projection fails +- **WHEN** the projector throws or returns an invalid evidence status +- **THEN** the same record becomes `ERROR`, stores a safe error code, and no Agent result is returned + +### Requirement: Tool Call ID and Run ownership SHALL be preserved +The boundary SHALL use the exact framework `tool_call_id` with the RunContext run ID to create the canonical key. It SHALL reject missing, unsafe, duplicate, and cross-Run references and SHALL never generate or replace a fallback ID. + +#### Scenario: Duplicate Tool Call ID is submitted +- **WHEN** a second invocation uses the same valid run ID and framework Tool Call ID +- **THEN** the second Tool is not executed and returns `ERROR` without overwriting the first record + +#### Scenario: Framework ID is preserved +- **WHEN** a valid envelope passes preflight +- **THEN** the key and canonical record contain the exact supplied `tool_call_id` + +### Requirement: TTL and size limits SHALL fail closed +The Redis implementation SHALL set TTL only when a canonical record is created. Reads SHALL NOT refresh TTL; updates SHALL preserve only the current remaining TTL. Request/raw/record and Agent-result byte limits SHALL be enforced in UTF-8. Raw overflow SHALL produce `ERROR/RESULT_TOO_LARGE` without silent truncation; Agent projection overflow SHALL produce the same error. + +#### Scenario: Read does not renew TTL +- **WHEN** a canonical record is read before expiration +- **THEN** its expiry remains at or before the original expiry and no expire/refresh operation is issued + +#### Scenario: Raw response is too large +- **WHEN** the executor returns raw data beyond the record limit +- **THEN** the record becomes `ERROR` with `RESULT_TOO_LARGE`, the raw payload is not silently truncated, and the projector is not invoked + +#### Scenario: Agent result is too large +- **WHEN** a projector returns a result beyond the Agent projection limit +- **THEN** the record becomes `ERROR/RESULT_TOO_LARGE` and the oversized result is not returned to the Agent + +### Requirement: Tool and store failures SHALL return safe errors +Execution, serialization, Redis, duplicate, preflight, and projection failures SHALL return a bounded error code without raw payload, credentials, internal stack trace, or Redis key to the Agent. If raw data was safely stored before a projection failure, it remains Harness-only canonical data. + +#### Scenario: Tool execution throws +- **WHEN** the executor raises an exception after PROJECTING begins +- **THEN** the record becomes `ERROR` with a stable execution error code and ToolBoundary returns no raw response + +### Requirement: Stage 3A SHALL remain reusable and independent from legacy audit +The boundary and canonical store SHALL be usable by later RAG/log/MySQL projectors without modifying legacy `ToolInvocationRecorder`, JPA `ToolInvocation`, Chat/AIOps, Controller, or public protocol. Redis access SHALL remain inside the store implementation and no Agent-facing API SHALL expose the Redis client. + +#### Scenario: Fake projector tests pass +- **WHEN** Fake Tool/Projector tests cover success, no evidence, error, duplicate, TTL, and capacity cases +- **THEN** later Tool-specific changes can depend on the same boundary without copying lifecycle or Redis logic, while legacy call sites remain unchanged diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ProjectedToolResult.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ProjectedToolResult.java new file mode 100644 index 0000000..961822d --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ProjectedToolResult.java @@ -0,0 +1,19 @@ +package com.superbiz.agent.harness.tool.boundary; + +import com.superbiz.agent.harness.contract.EvidenceStatus; + +import java.util.Objects; + +public record ProjectedToolResult(String agentResult, EvidenceStatus evidenceStatus) { + + public ProjectedToolResult { + if (agentResult == null || agentResult.isBlank()) { + throw new IllegalArgumentException("agentResult must not be blank"); + } + Objects.requireNonNull(evidenceStatus, "evidenceStatus must not be null"); + if (evidenceStatus != EvidenceStatus.EVIDENCE_FOUND + && evidenceStatus != EvidenceStatus.NO_EVIDENCE) { + throw new IllegalArgumentException("projected result must be evidence or no-evidence"); + } + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundary.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundary.java new file mode 100644 index 0000000..c3ec62f --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundary.java @@ -0,0 +1,219 @@ +package com.superbiz.agent.harness.tool.boundary; + +import com.fasterxml.jackson.core.JsonProcessingException; +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.harness.contract.EvidenceStatus; +import com.superbiz.agent.harness.core.BudgetExceededException; +import com.superbiz.agent.harness.core.DiagnosisHarnessCore; +import com.superbiz.agent.harness.core.RunAbortedException; +import com.superbiz.agent.harness.core.RunContext; +import com.superbiz.agent.harness.tool.store.CanonicalInvocationStore; +import com.superbiz.agent.harness.tool.store.CanonicalStoreException; +import com.superbiz.agent.harness.tool.store.CanonicalToolInvocation; +import com.superbiz.agent.harness.tool.store.DuplicateInvocationException; +import com.superbiz.agent.harness.tool.store.ResultTooLargeException; +import com.superbiz.agent.harness.tool.store.ToolCallKeyFactory; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import java.time.Clock; +import java.util.Objects; + +public final class ToolBoundary { + + private static final Logger log = LoggerFactory.getLogger(ToolBoundary.class); + + private final DiagnosisHarnessCore core; + private final ToolCallKeyFactory keyFactory; + private final CanonicalInvocationStore store; + private final ObjectMapper objectMapper; + private final Clock clock; + + public ToolBoundary(DiagnosisHarnessCore core, + ToolCallKeyFactory keyFactory, + CanonicalInvocationStore store, + ObjectMapper objectMapper, + Clock clock) { + this.core = Objects.requireNonNull(core, "core must not be null"); + this.keyFactory = Objects.requireNonNull(keyFactory, "keyFactory must not be null"); + this.store = Objects.requireNonNull(store, "store must not be null"); + this.objectMapper = Objects.requireNonNull(objectMapper, "objectMapper must not be null"); + this.clock = Objects.requireNonNull(clock, "clock must not be null"); + } + + public ToolBoundaryResult execute(RunContext context, + ToolCallRequestEnvelope request, + ToolExecutor executor, + ToolResultProjector projector) { + String toolCallId = request == null ? null : request.toolCallId(); + String key; + try { + key = preflight(context, request); + core.beforeToolCall(context, request.toolName()); + long requestBytes = store.limits().utf8Bytes(request.requestJson()); + if (requestBytes > store.limits().maxRecordBytes()) { + return errorAndNoRecord(toolCallId, ToolBoundaryErrorCode.RESULT_TOO_LARGE); + } + core.reserveRunBytes(context, requestBytes); + store.begin(key, CanonicalToolInvocation.projecting( + request.toolCallId(), request.runId(), request.toolName(), + request.requestJson(), clock.instant())); + } catch (DuplicateInvocationException e) { + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.DUPLICATE_TOOL_CALL); + } catch (RunAbortedException | BudgetExceededException e) { + return ToolBoundaryResult.error(toolCallId, + e instanceof BudgetExceededException + ? ToolBoundaryErrorCode.BUDGET_EXHAUSTED + : ToolBoundaryErrorCode.RUN_INACTIVE); + } catch (IllegalArgumentException e) { + return ToolBoundaryResult.error(toolCallId, classifyPreflightError(e)); + } catch (CanonicalStoreException e) { + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.STORE_ERROR); + } + + String rawResponse; + try { + rawResponse = Objects.requireNonNull(executor, "executor must not be null") + .execute(request.requestJson()); + if (rawResponse == null) { + throw new IllegalArgumentException("executor returned null"); + } + } catch (Exception e) { + markErrorSafely(key, null, ToolBoundaryErrorCode.TOOL_EXECUTION_ERROR); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.TOOL_EXECUTION_ERROR); + } + + try { + store.limits().validateRawCandidate(request.requestJson(), rawResponse); + core.reserveRunBytes(context, store.limits().utf8Bytes(rawResponse)); + } catch (ResultTooLargeException e) { + markErrorSafely(key, null, ToolBoundaryErrorCode.RESULT_TOO_LARGE); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.RESULT_TOO_LARGE); + } catch (BudgetExceededException e) { + markErrorSafely(key, null, ToolBoundaryErrorCode.BUDGET_EXHAUSTED); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.BUDGET_EXHAUSTED); + } catch (RunAbortedException e) { + markErrorSafely(key, null, ToolBoundaryErrorCode.RUN_INACTIVE); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.RUN_INACTIVE); + } + + ProjectedToolResult projected; + try { + projected = Objects.requireNonNull(projector, "projector must not be null") + .project(rawResponse); + if (projected == null) { + throw new IllegalArgumentException("projector returned null"); + } + } catch (Exception e) { + markErrorSafely(key, rawResponse, ToolBoundaryErrorCode.PROJECTION_ERROR); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.PROJECTION_ERROR); + } + + try { + store.limits().validateAgentResult(projected.agentResult()); + core.reserveRunBytes(context, store.limits().utf8Bytes(projected.agentResult())); + store.markReady(key, rawResponse, projected.agentResult(), projected.evidenceStatus(), clock.instant()); + return ToolBoundaryResult.ready(toolCallId, projected.agentResult(), projected.evidenceStatus()); + } catch (ResultTooLargeException e) { + markErrorSafely(key, rawResponse, ToolBoundaryErrorCode.RESULT_TOO_LARGE); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.RESULT_TOO_LARGE); + } catch (BudgetExceededException e) { + markErrorSafely(key, rawResponse, ToolBoundaryErrorCode.BUDGET_EXHAUSTED); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.BUDGET_EXHAUSTED); + } catch (RunAbortedException e) { + markErrorSafely(key, rawResponse, ToolBoundaryErrorCode.RUN_INACTIVE); + return ToolBoundaryResult.error(toolCallId, ToolBoundaryErrorCode.RUN_INACTIVE); + } catch (CanonicalStoreException e) { + ToolBoundaryErrorCode code = e instanceof ResultTooLargeException + ? ToolBoundaryErrorCode.RESULT_TOO_LARGE + : ToolBoundaryErrorCode.PROJECTION_ERROR; + markErrorSafely(key, rawResponse, code); + return ToolBoundaryResult.error(toolCallId, code); + } + } + + private String preflight(RunContext context, ToolCallRequestEnvelope request) { + if (context == null || request == null) { + throw new IllegalArgumentException("request/context must not be null"); + } + if (!context.runId().equals(request.runId())) { + throw new RunMismatchException(); + } + if (!request.authorized()) { + throw new UnauthorizedException(); + } + if (!request.readOnly()) { + throw new NotReadOnlyException(); + } + if (request.toolName() == null || request.toolName().isBlank() + || request.requestJson() == null || request.requestJson().isBlank()) { + throw new IllegalArgumentException("tool name and request must not be blank"); + } + try { + JsonNode root = objectMapper.readTree(request.requestJson()); + if (root == null || !root.isObject()) { + throw new IllegalArgumentException("request must be a JSON object"); + } + } catch (JsonProcessingException e) { + throw new IllegalArgumentException("request must be valid JSON", e); + } + try { + return keyFactory.create(request.runId(), request.toolCallId()); + } catch (IllegalArgumentException e) { + throw new InvalidToolCallIdException(e); + } + } + + private void markErrorSafely(String key, String rawResponse, ToolBoundaryErrorCode code) { + try { + store.markError(key, rawResponse, code.name(), clock.instant()); + } catch (RuntimeException e) { + log.warn("Failed to persist canonical Tool error: code={}", code, e); + } + } + + private ToolBoundaryResult errorAndNoRecord(String toolCallId, ToolBoundaryErrorCode code) { + return ToolBoundaryResult.error(toolCallId, code); + } + + private ToolBoundaryErrorCode classifyPreflightError(IllegalArgumentException exception) { + if (exception instanceof InvalidToolCallIdException) { + return ToolBoundaryErrorCode.INVALID_TOOL_CALL_ID; + } + if (exception instanceof RunMismatchException) { + return ToolBoundaryErrorCode.RUN_MISMATCH; + } + if (exception instanceof UnauthorizedException) { + return ToolBoundaryErrorCode.UNAUTHORIZED; + } + if (exception instanceof NotReadOnlyException) { + return ToolBoundaryErrorCode.NOT_READ_ONLY; + } + return ToolBoundaryErrorCode.INVALID_REQUEST; + } + + private static final class InvalidToolCallIdException extends IllegalArgumentException { + private InvalidToolCallIdException(Throwable cause) { + super("Invalid tool call ID", cause); + } + } + + private static final class RunMismatchException extends IllegalArgumentException { + private RunMismatchException() { + super("Run ID does not match context"); + } + } + + private static final class UnauthorizedException extends IllegalArgumentException { + private UnauthorizedException() { + super("Tool call is not authorized"); + } + } + + private static final class NotReadOnlyException extends IllegalArgumentException { + private NotReadOnlyException() { + super("Tool call is not read-only"); + } + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryErrorCode.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryErrorCode.java new file mode 100644 index 0000000..96fcde0 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryErrorCode.java @@ -0,0 +1,17 @@ +package com.superbiz.agent.harness.tool.boundary; + +public enum ToolBoundaryErrorCode { + INVALID_REQUEST, + INVALID_TOOL_CALL_ID, + RUN_MISMATCH, + UNAUTHORIZED, + NOT_READ_ONLY, + RUN_INACTIVE, + BUDGET_EXHAUSTED, + DUPLICATE_TOOL_CALL, + RESULT_TOO_LARGE, + TOOL_EXECUTION_ERROR, + PROJECTION_ERROR, + INVALID_EVIDENCE_STATUS, + STORE_ERROR +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryResult.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryResult.java new file mode 100644 index 0000000..8bac1e2 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryResult.java @@ -0,0 +1,47 @@ +package com.superbiz.agent.harness.tool.boundary; + +import com.fasterxml.jackson.annotation.JsonProperty; +import com.superbiz.agent.harness.contract.EvidenceStatus; +import com.superbiz.agent.harness.contract.InvocationStatus; + +public record ToolBoundaryResult( + @JsonProperty("status") InvocationStatus status, + @JsonProperty("evidence_status") EvidenceStatus evidenceStatus, + @JsonProperty("tool_call_id") String toolCallId, + @JsonProperty("agent_result") String agentResult, + @JsonProperty("error_code") String errorCode) { + + public ToolBoundaryResult { + if (status == InvocationStatus.READY) { + if (agentResult == null || evidenceStatus == null + || (evidenceStatus != EvidenceStatus.EVIDENCE_FOUND + && evidenceStatus != EvidenceStatus.NO_EVIDENCE)) { + throw new IllegalArgumentException("READY result requires bounded evidence result"); + } + if (errorCode != null) { + throw new IllegalArgumentException("READY result must not contain errorCode"); + } + } else if (status == InvocationStatus.ERROR) { + if (evidenceStatus != EvidenceStatus.ERROR || errorCode == null || errorCode.isBlank()) { + throw new IllegalArgumentException("ERROR result requires errorCode and ERROR evidence status"); + } + if (agentResult != null) { + throw new IllegalArgumentException("ERROR result must not contain agent result"); + } + } else { + throw new IllegalArgumentException("ToolBoundaryResult must be READY or ERROR"); + } + } + + public static ToolBoundaryResult ready(String toolCallId, + String agentResult, + EvidenceStatus evidenceStatus) { + return new ToolBoundaryResult( + InvocationStatus.READY, evidenceStatus, toolCallId, agentResult, null); + } + + public static ToolBoundaryResult error(String toolCallId, ToolBoundaryErrorCode errorCode) { + return new ToolBoundaryResult( + InvocationStatus.ERROR, EvidenceStatus.ERROR, toolCallId, null, errorCode.name()); + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolCallRequestEnvelope.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolCallRequestEnvelope.java new file mode 100644 index 0000000..b95e847 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolCallRequestEnvelope.java @@ -0,0 +1,12 @@ +package com.superbiz.agent.harness.tool.boundary; + +import com.fasterxml.jackson.annotation.JsonProperty; + +public record ToolCallRequestEnvelope( + @JsonProperty("run_id") String runId, + @JsonProperty("tool_call_id") String toolCallId, + @JsonProperty("tool_name") String toolName, + @JsonProperty("request") String requestJson, + @JsonProperty("authorized") boolean authorized, + @JsonProperty("read_only") boolean readOnly) { +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolExecutor.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolExecutor.java new file mode 100644 index 0000000..1e72822 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolExecutor.java @@ -0,0 +1,6 @@ +package com.superbiz.agent.harness.tool.boundary; + +@FunctionalInterface +public interface ToolExecutor { + String execute(String requestJson) throws Exception; +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolResultProjector.java b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolResultProjector.java new file mode 100644 index 0000000..9d7eb7e --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/boundary/ToolResultProjector.java @@ -0,0 +1,6 @@ +package com.superbiz.agent.harness.tool.boundary; + +@FunctionalInterface +public interface ToolResultProjector { + ProjectedToolResult project(String rawResponse) throws Exception; +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationLimits.java b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationLimits.java new file mode 100644 index 0000000..78a91de --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationLimits.java @@ -0,0 +1,50 @@ +package com.superbiz.agent.harness.tool.store; + +import java.nio.charset.StandardCharsets; +import java.time.Duration; +import java.util.Objects; + +public record CanonicalInvocationLimits( + Duration ttl, + long maxRecordBytes, + long maxAgentResultBytes) { + + public CanonicalInvocationLimits { + Objects.requireNonNull(ttl, "ttl must not be null"); + if (ttl.isZero() || ttl.isNegative()) { + throw new IllegalArgumentException("ttl must be positive"); + } + if (maxRecordBytes <= 0 || maxAgentResultBytes <= 0) { + throw new IllegalArgumentException("byte limits must be positive"); + } + if (maxAgentResultBytes > maxRecordBytes) { + throw new IllegalArgumentException("maxAgentResultBytes must not exceed maxRecordBytes"); + } + ttl.toMillis(); + } + + public void validateRawCandidate(String request, String rawResponse) { + long actual = utf8Bytes(request) + utf8Bytes(rawResponse); + if (actual > maxRecordBytes) { + throw new ResultTooLargeException("raw_response", maxRecordBytes, actual); + } + } + + public void validateAgentResult(String agentResult) { + long actual = utf8Bytes(agentResult); + if (actual > maxAgentResultBytes) { + throw new ResultTooLargeException("agent_result", maxAgentResultBytes, actual); + } + } + + public void validateSerializedRecord(String json) { + long actual = utf8Bytes(json); + if (actual > maxRecordBytes) { + throw new ResultTooLargeException("canonical_record", maxRecordBytes, actual); + } + } + + public static long utf8Bytes(String value) { + return value == null ? 0 : value.getBytes(StandardCharsets.UTF_8).length; + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationStore.java b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationStore.java new file mode 100644 index 0000000..d722920 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationStore.java @@ -0,0 +1,26 @@ +package com.superbiz.agent.harness.tool.store; + +import com.superbiz.agent.harness.contract.EvidenceStatus; + +import java.time.Instant; +import java.util.Optional; + +public interface CanonicalInvocationStore { + + CanonicalInvocationLimits limits(); + + void begin(String key, CanonicalToolInvocation invocation); + + Optional find(String key); + + CanonicalToolInvocation markReady(String key, + String rawResponse, + String agentResult, + EvidenceStatus evidenceStatus, + Instant completedAt); + + CanonicalToolInvocation markError(String key, + String rawResponse, + String errorCode, + Instant completedAt); +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalStoreException.java b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalStoreException.java new file mode 100644 index 0000000..ccb8960 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalStoreException.java @@ -0,0 +1,12 @@ +package com.superbiz.agent.harness.tool.store; + +public class CanonicalStoreException extends RuntimeException { + + public CanonicalStoreException(String message) { + super(message); + } + + public CanonicalStoreException(String message, Throwable cause) { + super(message, cause); + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalToolInvocation.java b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalToolInvocation.java new file mode 100644 index 0000000..68a2d61 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/CanonicalToolInvocation.java @@ -0,0 +1,117 @@ +package com.superbiz.agent.harness.tool.store; + +import com.fasterxml.jackson.annotation.JsonProperty; +import com.superbiz.agent.harness.contract.EvidenceStatus; +import com.superbiz.agent.harness.contract.InvocationStatus; + +import java.time.Instant; +import java.util.Objects; + +public record CanonicalToolInvocation( + @JsonProperty("tool_call_id") String toolCallId, + @JsonProperty("run_id") String runId, + @JsonProperty("tool_name") String toolName, + @JsonProperty("request") String request, + @JsonProperty("raw_response") String rawResponse, + @JsonProperty("agent_result") String agentResult, + @JsonProperty("status") InvocationStatus status, + @JsonProperty("evidence_status") EvidenceStatus evidenceStatus, + @JsonProperty("error_code") String errorCode, + @JsonProperty("started_at") Instant startedAt, + @JsonProperty("completed_at") Instant completedAt) { + + public CanonicalToolInvocation { + requireText(toolCallId, "toolCallId"); + requireText(runId, "runId"); + requireText(toolName, "toolName"); + requireText(request, "request"); + Objects.requireNonNull(status, "status must not be null"); + Objects.requireNonNull(startedAt, "startedAt must not be null"); + validateState(status, evidenceStatus, rawResponse, agentResult, errorCode, completedAt); + } + + public static CanonicalToolInvocation projecting(String toolCallId, + String runId, + String toolName, + String request, + Instant startedAt) { + return new CanonicalToolInvocation( + toolCallId, runId, toolName, request, null, null, + InvocationStatus.PROJECTING, null, null, startedAt, null); + } + + public CanonicalToolInvocation markReady(String rawResponse, + String agentResult, + EvidenceStatus evidenceStatus, + Instant completedAt) { + requireProjecting(); + return new CanonicalToolInvocation( + toolCallId, runId, toolName, request, rawResponse, agentResult, + InvocationStatus.READY, evidenceStatus, null, startedAt, completedAt); + } + + public CanonicalToolInvocation markError(String rawResponse, + String errorCode, + Instant completedAt) { + requireProjecting(); + return new CanonicalToolInvocation( + toolCallId, runId, toolName, request, rawResponse, null, + InvocationStatus.ERROR, EvidenceStatus.ERROR, errorCode, startedAt, completedAt); + } + + public boolean isReferencableBy(String expectedRunId) { + return status == InvocationStatus.READY + && Objects.equals(runId, expectedRunId) + && agentResult != null + && (evidenceStatus == EvidenceStatus.EVIDENCE_FOUND + || evidenceStatus == EvidenceStatus.NO_EVIDENCE); + } + + private void requireProjecting() { + if (status != InvocationStatus.PROJECTING) { + throw new InvocationStateException("Only PROJECTING invocation can transition"); + } + } + + private static void validateState(InvocationStatus status, + EvidenceStatus evidenceStatus, + String rawResponse, + String agentResult, + String errorCode, + Instant completedAt) { + switch (status) { + case PROJECTING -> { + if (evidenceStatus != null || agentResult != null || errorCode != null || completedAt != null) { + throw new InvocationStateException("PROJECTING invocation contains terminal fields"); + } + } + case READY -> { + if (rawResponse == null || agentResult == null || completedAt == null) { + throw new InvocationStateException("READY invocation requires raw, agent result and completion"); + } + if (evidenceStatus != EvidenceStatus.EVIDENCE_FOUND + && evidenceStatus != EvidenceStatus.NO_EVIDENCE) { + throw new InvocationStateException("READY invocation has invalid evidence status"); + } + if (errorCode != null) { + throw new InvocationStateException("READY invocation must not contain errorCode"); + } + } + case ERROR -> { + if (evidenceStatus != EvidenceStatus.ERROR || completedAt == null) { + throw new InvocationStateException("ERROR invocation requires ERROR evidence status and completion"); + } + requireText(errorCode, "errorCode"); + if (agentResult != null) { + throw new InvocationStateException("ERROR invocation must not contain agent result"); + } + } + } + } + + private static void requireText(String value, String name) { + if (value == null || value.isBlank()) { + throw new IllegalArgumentException(name + " must not be blank"); + } + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/DuplicateInvocationException.java b/src/main/java/com/superbiz/agent/harness/tool/store/DuplicateInvocationException.java new file mode 100644 index 0000000..5c125a3 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/DuplicateInvocationException.java @@ -0,0 +1,8 @@ +package com.superbiz.agent.harness.tool.store; + +public final class DuplicateInvocationException extends CanonicalStoreException { + + public DuplicateInvocationException() { + super("Canonical invocation already exists"); + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/InvocationStateException.java b/src/main/java/com/superbiz/agent/harness/tool/store/InvocationStateException.java new file mode 100644 index 0000000..a5e2768 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/InvocationStateException.java @@ -0,0 +1,8 @@ +package com.superbiz.agent.harness.tool.store; + +public final class InvocationStateException extends CanonicalStoreException { + + public InvocationStateException(String message) { + super(message); + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/RedisCanonicalInvocationStore.java b/src/main/java/com/superbiz/agent/harness/tool/store/RedisCanonicalInvocationStore.java new file mode 100644 index 0000000..05799d1 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/RedisCanonicalInvocationStore.java @@ -0,0 +1,144 @@ +package com.superbiz.agent.harness.tool.store; + +import com.fasterxml.jackson.core.JsonProcessingException; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.harness.contract.EvidenceStatus; +import com.superbiz.agent.harness.contract.InvocationStatus; +import org.springframework.data.redis.core.RedisTemplate; +import org.springframework.data.redis.core.ValueOperations; + +import java.time.Instant; +import java.util.Objects; +import java.util.Optional; +import java.util.concurrent.TimeUnit; +import java.util.function.UnaryOperator; + +public final class RedisCanonicalInvocationStore implements CanonicalInvocationStore { + + private final RedisTemplate redisTemplate; + private final ValueOperations values; + private final ObjectMapper objectMapper; + private final CanonicalInvocationLimits limits; + + public RedisCanonicalInvocationStore(RedisTemplate redisTemplate, + ObjectMapper objectMapper, + CanonicalInvocationLimits limits) { + this.redisTemplate = Objects.requireNonNull(redisTemplate, "redisTemplate must not be null"); + this.values = redisTemplate.opsForValue(); + this.objectMapper = Objects.requireNonNull(objectMapper, "objectMapper must not be null"); + this.limits = Objects.requireNonNull(limits, "limits must not be null"); + } + + @Override + public CanonicalInvocationLimits limits() { + return limits; + } + + @Override + public void begin(String key, CanonicalToolInvocation invocation) { + requireKey(key); + Objects.requireNonNull(invocation, "invocation must not be null"); + if (invocation.status() != InvocationStatus.PROJECTING) { + throw new InvocationStateException("begin requires PROJECTING invocation"); + } + String json = serialize(invocation); + limits.validateSerializedRecord(json); + Boolean created = values.setIfAbsent( + key, json, limits.ttl().toMillis(), TimeUnit.MILLISECONDS); + if (!Boolean.TRUE.equals(created)) { + throw new DuplicateInvocationException(); + } + } + + @Override + public Optional find(String key) { + requireKey(key); + Object stored = values.get(key); + if (stored == null) { + return Optional.empty(); + } + if (!(stored instanceof String json)) { + throw new CanonicalStoreException("Canonical invocation value is not JSON text"); + } + return Optional.of(deserialize(json)); + } + + @Override + public CanonicalToolInvocation markReady(String key, + String rawResponse, + String agentResult, + EvidenceStatus evidenceStatus, + Instant completedAt) { + CanonicalToolInvocation existing = find(key) + .orElseThrow(() -> new InvocationStateException("Canonical invocation is missing or expired")); + limits.validateRawCandidate(existing.request(), requireValue(rawResponse, "rawResponse")); + limits.validateAgentResult(requireValue(agentResult, "agentResult")); + return update(key, current -> current.markReady( + rawResponse, agentResult, evidenceStatus, completedAt)); + } + + @Override + public CanonicalToolInvocation markError(String key, + String rawResponse, + String errorCode, + Instant completedAt) { + try { + return update(key, current -> current.markError(rawResponse, errorCode, completedAt)); + } catch (ResultTooLargeException e) { + if (rawResponse == null) { + throw e; + } + return update(key, current -> current.markError( + null, ResultTooLargeException.ERROR_CODE, completedAt)); + } + } + + private CanonicalToolInvocation update(String key, + UnaryOperator transition) { + CanonicalToolInvocation current = find(key) + .orElseThrow(() -> new InvocationStateException("Canonical invocation is missing or expired")); + long remainingTtl = remainingTtlMillis(key); + CanonicalToolInvocation updated = transition.apply(current); + String json = serialize(updated); + limits.validateSerializedRecord(json); + values.set(key, json, remainingTtl, TimeUnit.MILLISECONDS); + return updated; + } + + private long remainingTtlMillis(String key) { + Long remaining = redisTemplate.getExpire(key, TimeUnit.MILLISECONDS); + if (remaining == null || remaining <= 0) { + throw new InvocationStateException("Canonical invocation is missing or expired"); + } + return Math.min(remaining, limits.ttl().toMillis()); + } + + private String serialize(CanonicalToolInvocation invocation) { + try { + return objectMapper.writeValueAsString(invocation); + } catch (JsonProcessingException e) { + throw new CanonicalStoreException("Failed to serialize canonical invocation", e); + } + } + + private CanonicalToolInvocation deserialize(String json) { + try { + return objectMapper.readValue(json, CanonicalToolInvocation.class); + } catch (JsonProcessingException e) { + throw new CanonicalStoreException("Failed to deserialize canonical invocation", e); + } + } + + private static String requireValue(String value, String name) { + if (value == null) { + throw new IllegalArgumentException(name + " must not be null"); + } + return value; + } + + private static void requireKey(String key) { + if (key == null || key.isBlank()) { + throw new IllegalArgumentException("key must not be blank"); + } + } +} diff --git a/src/main/java/com/superbiz/agent/harness/tool/store/ResultTooLargeException.java b/src/main/java/com/superbiz/agent/harness/tool/store/ResultTooLargeException.java new file mode 100644 index 0000000..58cd620 --- /dev/null +++ b/src/main/java/com/superbiz/agent/harness/tool/store/ResultTooLargeException.java @@ -0,0 +1,10 @@ +package com.superbiz.agent.harness.tool.store; + +public final class ResultTooLargeException extends CanonicalStoreException { + + public static final String ERROR_CODE = "RESULT_TOO_LARGE"; + + public ResultTooLargeException(String field, long limit, long actual) { + super(field + " exceeds byte limit: limit=" + limit + ", actual=" + actual); + } +} diff --git a/src/test/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryTest.java b/src/test/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryTest.java new file mode 100644 index 0000000..c37467a --- /dev/null +++ b/src/test/java/com/superbiz/agent/harness/tool/boundary/ToolBoundaryTest.java @@ -0,0 +1,215 @@ +package com.superbiz.agent.harness.tool.boundary; + +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.harness.contract.EvidenceStatus; +import com.superbiz.agent.harness.contract.InvocationStatus; +import com.superbiz.agent.harness.core.DiagnosisHarnessCore; +import com.superbiz.agent.harness.core.HarnessCoreFixtures; +import com.superbiz.agent.harness.core.MutableClock; +import com.superbiz.agent.harness.core.RunContext; +import com.superbiz.agent.harness.tool.store.CanonicalInvocationLimits; +import com.superbiz.agent.harness.tool.store.CanonicalInvocationStore; +import com.superbiz.agent.harness.tool.store.CanonicalToolInvocation; +import com.superbiz.agent.harness.tool.store.DuplicateInvocationException; +import com.superbiz.agent.harness.tool.store.ToolCallKeyFactory; +import org.junit.jupiter.api.Test; + +import java.time.Clock; +import java.time.Duration; +import java.time.Instant; +import java.util.HashMap; +import java.util.Map; +import java.util.Optional; +import java.util.concurrent.atomic.AtomicInteger; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +class ToolBoundaryTest { + + private final ObjectMapper objectMapper = new ObjectMapper().findAndRegisterModules(); + + @Test + void storesRawAndReturnsOnlyProjectedReadyResult() { + MutableClock clock = new MutableClock(Instant.parse("2026-07-21T10:00:00Z")); + FakeStore store = new FakeStore(new CanonicalInvocationLimits(Duration.ofHours(2), 4096, 1024)); + ToolBoundary boundary = boundary(clock, store); + RunContext context = HarnessCoreFixtures.core(clock).startRun("session-1", "run-1"); + AtomicInteger executions = new AtomicInteger(); + + ToolBoundaryResult result = boundary.execute( + context, + request("run-1", "call-1", true, true), + json -> { + executions.incrementAndGet(); + return "{\"secret_raw\":true}"; + }, + raw -> new ProjectedToolResult("{\"evidence\":true}", EvidenceStatus.EVIDENCE_FOUND)); + + assertEquals(InvocationStatus.READY, result.status()); + assertEquals("call-1", result.toolCallId()); + assertNull(result.errorCode()); + assertFalse(String.valueOf(result.agentResult()).contains("secret_raw")); + assertEquals(1, executions.get()); + CanonicalToolInvocation saved = store.find("superbiz:harness:tool-call:run-1:call-1").orElseThrow(); + assertEquals("{\"secret_raw\":true}", saved.rawResponse()); + assertTrue(saved.isReferencableBy("run-1")); + } + + @Test + void preservesNoEvidenceAndRejectsDuplicateAndCrossRun() { + MutableClock clock = new MutableClock(Instant.parse("2026-07-21T10:00:00Z")); + FakeStore store = new FakeStore(new CanonicalInvocationLimits(Duration.ofHours(2), 4096, 1024)); + ToolBoundary boundary = boundary(clock, store); + RunContext context = HarnessCoreFixtures.core(clock).startRun("session-1", "run-1"); + AtomicInteger executions = new AtomicInteger(); + ToolExecutor executor = request -> { + executions.incrementAndGet(); + return "raw"; + }; + ToolResultProjector projector = raw -> new ProjectedToolResult( + "{\"scope\":\"none\"}", EvidenceStatus.NO_EVIDENCE); + + ToolBoundaryResult first = boundary.execute(context, request("run-1", "call-2", true, true), executor, projector); + ToolBoundaryResult duplicate = boundary.execute(context, request("run-1", "call-2", true, true), executor, projector); + ToolBoundaryResult crossRun = boundary.execute(context, request("other-run", "call-3", true, true), executor, projector); + + assertEquals(EvidenceStatus.NO_EVIDENCE, first.evidenceStatus()); + assertEquals(ToolBoundaryErrorCode.DUPLICATE_TOOL_CALL.name(), duplicate.errorCode()); + assertEquals(ToolBoundaryErrorCode.RUN_MISMATCH.name(), crossRun.errorCode()); + assertEquals(1, executions.get()); + assertTrue(store.find("superbiz:harness:tool-call:run-1:call-2").orElseThrow().isReferencableBy("run-1")); + } + + @Test + void rejectsUnauthorizedWritableAndInvalidIdBeforeTool() { + MutableClock clock = new MutableClock(Instant.parse("2026-07-21T10:00:00Z")); + FakeStore store = new FakeStore(new CanonicalInvocationLimits(Duration.ofHours(2), 4096, 1024)); + ToolBoundary boundary = boundary(clock, store); + RunContext context = HarnessCoreFixtures.core(clock).startRun("session-1", "run-1"); + AtomicInteger executions = new AtomicInteger(); + ToolExecutor executor = request -> { + executions.incrementAndGet(); + return "raw"; + }; + ToolResultProjector projector = raw -> new ProjectedToolResult("agent", EvidenceStatus.EVIDENCE_FOUND); + + ToolBoundaryResult unauthorized = boundary.execute(context, request("run-1", "call-4", false, true), executor, projector); + ToolBoundaryResult writable = boundary.execute(context, request("run-1", "call-5", true, false), executor, projector); + ToolBoundaryResult invalidId = boundary.execute(context, request("run-1", "bad:id", true, true), executor, projector); + + assertEquals(ToolBoundaryErrorCode.UNAUTHORIZED.name(), unauthorized.errorCode()); + assertEquals(ToolBoundaryErrorCode.NOT_READ_ONLY.name(), writable.errorCode()); + assertEquals(ToolBoundaryErrorCode.INVALID_TOOL_CALL_ID.name(), invalidId.errorCode()); + assertEquals(0, executions.get()); + } + + @Test + void rawOverflowSkipsProjectorAndAgentOverflowIsNotReturned() { + MutableClock clock = new MutableClock(Instant.parse("2026-07-21T10:00:00Z")); + RunContext context = HarnessCoreFixtures.core(clock).startRun("session-1", "run-1"); + AtomicInteger projectorCalls = new AtomicInteger(); + FakeStore rawStore = new FakeStore(new CanonicalInvocationLimits(Duration.ofHours(2), 20, 10)); + ToolBoundary rawBoundary = boundary(clock, rawStore); + ToolBoundaryResult rawOverflow = rawBoundary.execute( + context, request("run-1", "call-6", true, true), + requestJson -> "x".repeat(100), + raw -> { + projectorCalls.incrementAndGet(); + return new ProjectedToolResult("agent", EvidenceStatus.EVIDENCE_FOUND); + }); + + FakeStore agentStore = new FakeStore(new CanonicalInvocationLimits(Duration.ofHours(2), 4096, 4)); + ToolBoundary agentBoundary = boundary(clock, agentStore); + ToolBoundaryResult agentOverflow = agentBoundary.execute( + context, request("run-1", "call-7", true, true), + requestJson -> "raw", + raw -> new ProjectedToolResult("too-large", EvidenceStatus.EVIDENCE_FOUND)); + + assertEquals(ToolBoundaryErrorCode.RESULT_TOO_LARGE.name(), rawOverflow.errorCode()); + assertEquals(0, projectorCalls.get()); + assertEquals(ToolBoundaryErrorCode.RESULT_TOO_LARGE.name(), agentOverflow.errorCode()); + assertNull(agentOverflow.agentResult()); + assertEquals(InvocationStatus.ERROR, agentStore.find("superbiz:harness:tool-call:run-1:call-7").orElseThrow().status()); + } + + @Test + void executionAndProjectionErrorsAreCanonicalErrors() { + MutableClock clock = new MutableClock(Instant.parse("2026-07-21T10:00:00Z")); + FakeStore store = new FakeStore(new CanonicalInvocationLimits(Duration.ofHours(2), 4096, 1024)); + ToolBoundary boundary = boundary(clock, store); + RunContext context = HarnessCoreFixtures.core(clock).startRun("session-1", "run-1"); + + ToolBoundaryResult executionError = boundary.execute( + context, request("run-1", "call-8", true, true), + requestJson -> { throw new IllegalStateException("internal raw error"); }, + raw -> new ProjectedToolResult("agent", EvidenceStatus.EVIDENCE_FOUND)); + ToolBoundaryResult projectionError = boundary.execute( + context, request("run-1", "call-9", true, true), + requestJson -> "raw", + raw -> { throw new IllegalStateException("projection error"); }); + + assertEquals(ToolBoundaryErrorCode.TOOL_EXECUTION_ERROR.name(), executionError.errorCode()); + assertEquals(ToolBoundaryErrorCode.PROJECTION_ERROR.name(), projectionError.errorCode()); + assertFalse(store.find("superbiz:harness:tool-call:run-1:call-8").orElseThrow().isReferencableBy("run-1")); + assertFalse(store.find("superbiz:harness:tool-call:run-1:call-9").orElseThrow().isReferencableBy("run-1")); + } + + private ToolBoundary boundary(MutableClock clock, FakeStore store) { + DiagnosisHarnessCore core = HarnessCoreFixtures.core(clock); + return new ToolBoundary(core, new ToolCallKeyFactory("superbiz:harness:tool-call"), + store, objectMapper, clock); + } + + private ToolCallRequestEnvelope request(String runId, String toolCallId, + boolean authorized, boolean readOnly) { + return new ToolCallRequestEnvelope(runId, toolCallId, "query_logs", "{\"query\":\"timeout\"}", + authorized, readOnly); + } + + private static final class FakeStore implements CanonicalInvocationStore { + private final CanonicalInvocationLimits limits; + private final Map records = new HashMap<>(); + + private FakeStore(CanonicalInvocationLimits limits) { + this.limits = limits; + } + + @Override + public CanonicalInvocationLimits limits() { + return limits; + } + + @Override + public void begin(String key, CanonicalToolInvocation invocation) { + if (records.putIfAbsent(key, invocation) != null) { + throw new DuplicateInvocationException(); + } + } + + @Override + public Optional find(String key) { + return Optional.ofNullable(records.get(key)); + } + + @Override + public CanonicalToolInvocation markReady(String key, String rawResponse, String agentResult, + EvidenceStatus evidenceStatus, Instant completedAt) { + CanonicalToolInvocation current = records.get(key); + CanonicalToolInvocation updated = current.markReady(rawResponse, agentResult, evidenceStatus, completedAt); + records.put(key, updated); + return updated; + } + + @Override + public CanonicalToolInvocation markError(String key, String rawResponse, String errorCode, + Instant completedAt) { + CanonicalToolInvocation current = records.get(key); + CanonicalToolInvocation updated = current.markError(rawResponse, errorCode, completedAt); + records.put(key, updated); + return updated; + } + } +} diff --git a/src/test/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationStoreTest.java b/src/test/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationStoreTest.java new file mode 100644 index 0000000..91b8e85 --- /dev/null +++ b/src/test/java/com/superbiz/agent/harness/tool/store/CanonicalInvocationStoreTest.java @@ -0,0 +1,100 @@ +package com.superbiz.agent.harness.tool.store; + +import com.fasterxml.jackson.databind.ObjectMapper; +import com.superbiz.agent.harness.contract.EvidenceStatus; +import com.superbiz.agent.harness.contract.InvocationStatus; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; +import org.springframework.data.redis.core.RedisTemplate; +import org.springframework.data.redis.core.ValueOperations; + +import java.time.Duration; +import java.time.Instant; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.mockito.ArgumentMatchers.any; +import static org.mockito.ArgumentMatchers.eq; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.never; +import static org.mockito.Mockito.times; +import static org.mockito.Mockito.verify; +import static org.mockito.Mockito.when; + +class CanonicalInvocationStoreTest { + + private RedisTemplate redisTemplate; + private ValueOperations values; + private RedisCanonicalInvocationStore store; + private CanonicalToolInvocation projecting; + private final ObjectMapper objectMapper = new ObjectMapper().findAndRegisterModules(); + + @BeforeEach + void setUp() { + redisTemplate = mock(RedisTemplate.class); + values = mock(ValueOperations.class); + when(redisTemplate.opsForValue()).thenReturn(values); + store = new RedisCanonicalInvocationStore( + redisTemplate, + objectMapper, + new CanonicalInvocationLimits(Duration.ofHours(2), 4096, 1024)); + projecting = CanonicalToolInvocation.projecting( + "call-1", "run-1", "query_logs", "{\"query\":\"timeout\"}", + Instant.parse("2026-07-21T10:00:00Z")); + } + + @Test + void createsCompleteProjectingRecordAndRejectsDuplicate() throws Exception { + when(values.setIfAbsent(any(), any(), any(Long.class), any())).thenReturn(true, false); + store.begin("prefix:run-1:call-1", projecting); + + assertEquals(InvocationStatus.PROJECTING, projecting.status()); + assertFalse(projecting.request().isBlank()); + assertThrows(DuplicateInvocationException.class, + () -> store.begin("prefix:run-1:call-1", projecting)); + verify(values, times(2)).setIfAbsent( + eq("prefix:run-1:call-1"), any(), eq(Duration.ofHours(2).toMillis()), any()); + } + + @Test + void updatesSameRecordToReadyWithRemainingTtlAndReadDoesNotRefresh() throws Exception { + String projectingJson = objectMapper.writeValueAsString(projecting); + when(values.get("key")).thenReturn(projectingJson, projectingJson); + when(redisTemplate.getExpire("key", java.util.concurrent.TimeUnit.MILLISECONDS)).thenReturn(3210L); + + CanonicalToolInvocation ready = store.markReady( + "key", "{\"raw\":true}", "{\"evidence\":true}", + EvidenceStatus.EVIDENCE_FOUND, Instant.parse("2026-07-21T10:00:01Z")); + + assertEquals(InvocationStatus.READY, ready.status()); + assertEquals("call-1", ready.toolCallId()); + assertTrueJson(ready.agentResult()); + verify(values).set(eq("key"), any(), eq(3210L), eq(java.util.concurrent.TimeUnit.MILLISECONDS)); + + when(values.get("read-only")).thenReturn(projectingJson); + store.find("read-only"); + verify(redisTemplate, never()).getExpire("read-only", java.util.concurrent.TimeUnit.MILLISECONDS); + } + + @Test + void rejectsInvalidReadyEvidenceAndWritesErrorState() throws Exception { + String projectingJson = objectMapper.writeValueAsString(projecting); + when(values.get("key")).thenReturn(projectingJson, projectingJson); + when(redisTemplate.getExpire("key", java.util.concurrent.TimeUnit.MILLISECONDS)).thenReturn(3210L); + + assertThrows(InvocationStateException.class, () -> store.markReady( + "key", "raw", "agent", EvidenceStatus.ERROR, + Instant.parse("2026-07-21T10:00:01Z"))); + + CanonicalToolInvocation error = store.markError( + "key", "raw", "PROJECTION_ERROR", Instant.parse("2026-07-21T10:00:02Z")); + assertEquals(InvocationStatus.ERROR, error.status()); + assertEquals(EvidenceStatus.ERROR, error.evidenceStatus()); + assertFalse(error.isReferencableBy("run-1")); + } + + private void assertTrueJson(String json) throws Exception { + assertEquals(true, objectMapper.readTree(json).path("evidence").asBoolean()); + } +}