docs(harness): add retry guide and learning roadmap; annotate retry core

This commit is contained in:
wdm1802
2026-08-03 01:29:36 +08:00
parent d084202166
commit e564863c43
8 changed files with 478 additions and 0 deletions
@@ -9,6 +9,20 @@ import com.superbiz.agent.harness.core.RunState;
import java.util.Objects;
import java.util.function.Consumer;
/**
* 统一重试执行器:把「失败分类 + 策略裁决 + attempt 记录」集中到一个循环里。
*
* <p>为什么重试必须在这里,而不是 SDK 内部:
* <ul>
* <li>SDK 隐式重试(Spring AI 默认 maxAttempts=10)已通过
* {@code spring.ai.retry.max-attempts: 1} 关闭,重试所有权上移到 Harness;</li>
* <li>每次 attempt 前都 {@code checkActive}——Run 已终止(取消/预算/超时)时立即停止,
* 不会在 Run 死后继续烧预算;</li>
* <li>{@code RunAbortedException} / {@code BudgetExceededException} 永不重试,直接透出;
* 其他异常先由 {@code classifier} 分类,再由 {@code policy} 裁决是否再试;</li>
* <li>每个 attempt 都经 {@code recorder} 记录,Trace 可回放「试了几次、为什么停」。</li>
* </ul>
*/
public final class HarnessRetryExecutor {
private final DiagnosisHarnessCore core;
@@ -17,6 +31,15 @@ public final class HarnessRetryExecutor {
this.core = Objects.requireNonNull(core, "core must not be null");
}
/**
* 执行带重试的操作。
*
* @param context RunContext(每次 attempt 前检查 active 用)
* @param policy 重试策略:maxAttempts + 可重试失败类型
* @param operation 一次操作,通常是「模型调用 + 严格解析」
* @param classifier 异常 → RetryFailure 分类器
* @param recorder 每次 attempt 的收据(写 Trace / 账本)
*/
public <T> T execute(RunContext context,
RetryPolicy policy,
RetryOperation<T> operation,
@@ -29,32 +52,39 @@ public final class HarnessRetryExecutor {
Objects.requireNonNull(recorder, "recorder must not be null");
for (int attempt = 1; attempt <= policy.maxAttempts(); attempt++) {
// 每个 attempt 前先确认 Run 仍可执行;Run 已死则这里直接抛 RunAbortedException
core.checkActive(context);
try {
T result = operation.execute();
recorder.accept(RetryAttempt.succeeded(attempt));
return result;
} catch (RunAbortedException exception) {
// Run 已终止:从终态快照区分预算耗尽还是取消,立即透出,绝不重试
RetryFailure failure = exception.termination().state() == RunState.BUDGET_EXHAUSTED
? RetryFailure.BUDGET_EXHAUSTED
: RetryFailure.CANCELLED;
recorder.accept(RetryAttempt.failed(attempt, failure));
throw new RetryExecutionException(attempt, failure, exception);
} catch (BudgetExceededException exception) {
// 预算超限:重试只会再烧预算,立即透出,绝不重试
recorder.accept(RetryAttempt.failed(attempt, RetryFailure.BUDGET_EXHAUSTED));
throw new RetryExecutionException(
attempt, RetryFailure.BUDGET_EXHAUSTED, exception);
} catch (Exception exception) {
// 其他异常:先分类(null → UNKNOWN),再由策略裁决是否允许下一轮 attempt
RetryFailure failure = classifier.classify(exception);
if (failure == null) {
failure = RetryFailure.UNKNOWN;
}
recorder.accept(RetryAttempt.failed(attempt, failure));
if (!policy.allowsRetry(attempt, failure)) {
// 裁决失败:要么次数用尽,要么失败类型不可重试(如业务拒绝/无证据)
throw new RetryExecutionException(attempt, failure, exception);
}
// 允许 → 继续下一轮循环
}
}
// 理论上不可达:policy.maxAttempts >= 1,且循环内要么 return 要么 throw
throw new IllegalStateException("retry loop exited without a result");
}
}
@@ -3,6 +3,16 @@ package com.superbiz.agent.harness.retry;
import java.util.Objects;
import java.util.Set;
/**
* 每个组件独立的重试策略集合(不可变)。
*
* <p>五个组件各有自己的 {@code maxAttempts + retryableFailures},
* 因为「是否允许重试」取决于调用者知道的信息:
* <ul>
* <li>Agent / 业务 Tool 可能有副作用或多轮上下文,不重试;</li>
* <li>Router / SemanticGuard 是单轮无副作用的技术判定,允许一次技术重试。</li>
* </ul>
*/
public record HarnessRetryPolicies(
RetryPolicy intentRouter,
RetryPolicy diagnosisAgent,
@@ -18,6 +28,17 @@ public record HarnessRetryPolicies(
Objects.requireNonNull(evidenceRepair, "evidenceRepair must not be null");
}
/**
* 严格默认策略:
*
* <pre>
* intentRouter : 2 次(超时 / 传输 / 非法输出可重试)——路由判据单轮无副作用
* semanticGuard : 2 次(超时 / 传输 / 解析 / schema 可重试)——语义审查单轮无副作用
* diagnosisAgent : 1 次——多轮 ReAct,失败会破坏循环上下文,不重试
* toolCall : 1 次——业务 Tool 可能有副作用,重试会重复副作用
* evidenceRepair : 1 次——只修引用,失败直接走 Fallback,不重试
* </pre>
*/
public static HarnessRetryPolicies strict() {
RetryPolicy oneAttempt = new RetryPolicy(1, Set.of());
return new HarnessRetryPolicies(
@@ -1,5 +1,18 @@
package com.superbiz.agent.harness.retry;
/**
* 单次重试 attempt 的不可变收据(记录):第几次、成败、失败类型。
*
* <p>由 {@link HarnessRetryExecutor} 在每次尝试后产生,经调用方的 recorder
* ({@code Consumer<RetryAttempt>})写入 Trace(routingAttempt / semanticAttempt /
* evidenceRepairAttempt 等事件),让「试了几次、每次什么失败」完全可回放。
*
* <p>与 {@link RetryExecutionException} 互补:RetryAttempt 是每一步的脚印(过程),
* RetryExecutionException 是最终定格(attempts 总数 + 最后失败类型)。
*
* <p>构造校验保证记录必然自洽:成功不能带失败类型、失败必须带失败类型,
* 避免把自相矛盾的脏记录写进 Trace。
*/
public record RetryAttempt(int attemptNumber, boolean success, RetryFailure failure) {
public RetryAttempt {
@@ -14,10 +27,12 @@ public record RetryAttempt(int attemptNumber, boolean success, RetryFailure fail
}
}
/** 成功收据:failure 固定为 null(构造校验保证)。 */
public static RetryAttempt succeeded(int attemptNumber) {
return new RetryAttempt(attemptNumber, true, null);
}
/** 失败收据:必须携带失败类型,供 Trace 和策略裁决参考。 */
public static RetryAttempt failed(int attemptNumber, RetryFailure failure) {
return new RetryAttempt(attemptNumber, false, failure);
}
@@ -11,6 +11,8 @@ package com.superbiz.agent.harness.retry;
*/
public enum RetryFailure {
// ===== 可重试组:技术类失败,重试可能成功 =====
/** 单次 attempt 或总超时。 */
TIMEOUT,
@@ -26,6 +28,8 @@ public enum RetryFailure {
/** 结构符合 JSON 但 schema/字段约束失败。 */
SCHEMA_INVALID,
// ===== 绝不重试组:业务/系统事实,重试不会改变结果 =====
/** 业务上判定无可用证据(若某组件使用该分类)。 */
NO_EVIDENCE,
@@ -2,6 +2,17 @@ package com.superbiz.agent.harness.retry;
import java.util.Set;
/**
* 重试策略:不可变值对象,表达「最多试几次 + 哪些失败类型可重试」。
*
* <p>不用布尔 {@code retry=true},而用 {@code maxAttempts + retryableFailures} 组合,
* 因为重试必须同时回答两个问题:
* <ul>
* <li>还能不能再试(次数维度:{@code completedAttempts < maxAttempts});</li>
* <li>这次失败值不值得试(类型维度:失败是否在可重试集合里)。</li>
* </ul>
* {@code maxAttempts} 限定为 1 或 2,防止配置膨胀成不可控的隐式重试。
*/
public record RetryPolicy(int maxAttempts, Set<RetryFailure> retryableFailures) {
public RetryPolicy {
@@ -11,6 +22,13 @@ public record RetryPolicy(int maxAttempts, Set<RetryFailure> retryableFailures)
retryableFailures = retryableFailures == null ? Set.of() : Set.copyOf(retryableFailures);
}
/**
* 双条件裁决:还有剩余次数 且 失败类型可重试,才允许下一次 attempt。
*
* <p>注意 {@code completedAttempts} 是「已完成(失败)的次数」:
* 例如 maxAttempts=1(EvidenceRepair)时,第一次失败后
* {@code 1 < 1} 为 false,永不重试。
*/
public boolean allowsRetry(int completedAttempts, RetryFailure failure) {
return completedAttempts < maxAttempts && retryableFailures.contains(failure);
}