docs(harness): add retry guide and learning roadmap; annotate retry core
This commit is contained in:
@@ -9,6 +9,20 @@ import com.superbiz.agent.harness.core.RunState;
|
||||
import java.util.Objects;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
/**
|
||||
* 统一重试执行器:把「失败分类 + 策略裁决 + attempt 记录」集中到一个循环里。
|
||||
*
|
||||
* <p>为什么重试必须在这里,而不是 SDK 内部:
|
||||
* <ul>
|
||||
* <li>SDK 隐式重试(Spring AI 默认 maxAttempts=10)已通过
|
||||
* {@code spring.ai.retry.max-attempts: 1} 关闭,重试所有权上移到 Harness;</li>
|
||||
* <li>每次 attempt 前都 {@code checkActive}——Run 已终止(取消/预算/超时)时立即停止,
|
||||
* 不会在 Run 死后继续烧预算;</li>
|
||||
* <li>{@code RunAbortedException} / {@code BudgetExceededException} 永不重试,直接透出;
|
||||
* 其他异常先由 {@code classifier} 分类,再由 {@code policy} 裁决是否再试;</li>
|
||||
* <li>每个 attempt 都经 {@code recorder} 记录,Trace 可回放「试了几次、为什么停」。</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class HarnessRetryExecutor {
|
||||
|
||||
private final DiagnosisHarnessCore core;
|
||||
@@ -17,6 +31,15 @@ public final class HarnessRetryExecutor {
|
||||
this.core = Objects.requireNonNull(core, "core must not be null");
|
||||
}
|
||||
|
||||
/**
|
||||
* 执行带重试的操作。
|
||||
*
|
||||
* @param context RunContext(每次 attempt 前检查 active 用)
|
||||
* @param policy 重试策略:maxAttempts + 可重试失败类型
|
||||
* @param operation 一次操作,通常是「模型调用 + 严格解析」
|
||||
* @param classifier 异常 → RetryFailure 分类器
|
||||
* @param recorder 每次 attempt 的收据(写 Trace / 账本)
|
||||
*/
|
||||
public <T> T execute(RunContext context,
|
||||
RetryPolicy policy,
|
||||
RetryOperation<T> operation,
|
||||
@@ -29,32 +52,39 @@ public final class HarnessRetryExecutor {
|
||||
Objects.requireNonNull(recorder, "recorder must not be null");
|
||||
|
||||
for (int attempt = 1; attempt <= policy.maxAttempts(); attempt++) {
|
||||
// 每个 attempt 前先确认 Run 仍可执行;Run 已死则这里直接抛 RunAbortedException
|
||||
core.checkActive(context);
|
||||
try {
|
||||
T result = operation.execute();
|
||||
recorder.accept(RetryAttempt.succeeded(attempt));
|
||||
return result;
|
||||
} catch (RunAbortedException exception) {
|
||||
// Run 已终止:从终态快照区分预算耗尽还是取消,立即透出,绝不重试
|
||||
RetryFailure failure = exception.termination().state() == RunState.BUDGET_EXHAUSTED
|
||||
? RetryFailure.BUDGET_EXHAUSTED
|
||||
: RetryFailure.CANCELLED;
|
||||
recorder.accept(RetryAttempt.failed(attempt, failure));
|
||||
throw new RetryExecutionException(attempt, failure, exception);
|
||||
} catch (BudgetExceededException exception) {
|
||||
// 预算超限:重试只会再烧预算,立即透出,绝不重试
|
||||
recorder.accept(RetryAttempt.failed(attempt, RetryFailure.BUDGET_EXHAUSTED));
|
||||
throw new RetryExecutionException(
|
||||
attempt, RetryFailure.BUDGET_EXHAUSTED, exception);
|
||||
} catch (Exception exception) {
|
||||
// 其他异常:先分类(null → UNKNOWN),再由策略裁决是否允许下一轮 attempt
|
||||
RetryFailure failure = classifier.classify(exception);
|
||||
if (failure == null) {
|
||||
failure = RetryFailure.UNKNOWN;
|
||||
}
|
||||
recorder.accept(RetryAttempt.failed(attempt, failure));
|
||||
if (!policy.allowsRetry(attempt, failure)) {
|
||||
// 裁决失败:要么次数用尽,要么失败类型不可重试(如业务拒绝/无证据)
|
||||
throw new RetryExecutionException(attempt, failure, exception);
|
||||
}
|
||||
// 允许 → 继续下一轮循环
|
||||
}
|
||||
}
|
||||
// 理论上不可达:policy.maxAttempts >= 1,且循环内要么 return 要么 throw
|
||||
throw new IllegalStateException("retry loop exited without a result");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,6 +3,16 @@ package com.superbiz.agent.harness.retry;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* 每个组件独立的重试策略集合(不可变)。
|
||||
*
|
||||
* <p>五个组件各有自己的 {@code maxAttempts + retryableFailures},
|
||||
* 因为「是否允许重试」取决于调用者知道的信息:
|
||||
* <ul>
|
||||
* <li>Agent / 业务 Tool 可能有副作用或多轮上下文,不重试;</li>
|
||||
* <li>Router / SemanticGuard 是单轮无副作用的技术判定,允许一次技术重试。</li>
|
||||
* </ul>
|
||||
*/
|
||||
public record HarnessRetryPolicies(
|
||||
RetryPolicy intentRouter,
|
||||
RetryPolicy diagnosisAgent,
|
||||
@@ -18,6 +28,17 @@ public record HarnessRetryPolicies(
|
||||
Objects.requireNonNull(evidenceRepair, "evidenceRepair must not be null");
|
||||
}
|
||||
|
||||
/**
|
||||
* 严格默认策略:
|
||||
*
|
||||
* <pre>
|
||||
* intentRouter : 2 次(超时 / 传输 / 非法输出可重试)——路由判据单轮无副作用
|
||||
* semanticGuard : 2 次(超时 / 传输 / 解析 / schema 可重试)——语义审查单轮无副作用
|
||||
* diagnosisAgent : 1 次——多轮 ReAct,失败会破坏循环上下文,不重试
|
||||
* toolCall : 1 次——业务 Tool 可能有副作用,重试会重复副作用
|
||||
* evidenceRepair : 1 次——只修引用,失败直接走 Fallback,不重试
|
||||
* </pre>
|
||||
*/
|
||||
public static HarnessRetryPolicies strict() {
|
||||
RetryPolicy oneAttempt = new RetryPolicy(1, Set.of());
|
||||
return new HarnessRetryPolicies(
|
||||
|
||||
@@ -1,5 +1,18 @@
|
||||
package com.superbiz.agent.harness.retry;
|
||||
|
||||
/**
|
||||
* 单次重试 attempt 的不可变收据(记录):第几次、成败、失败类型。
|
||||
*
|
||||
* <p>由 {@link HarnessRetryExecutor} 在每次尝试后产生,经调用方的 recorder
|
||||
* ({@code Consumer<RetryAttempt>})写入 Trace(routingAttempt / semanticAttempt /
|
||||
* evidenceRepairAttempt 等事件),让「试了几次、每次什么失败」完全可回放。
|
||||
*
|
||||
* <p>与 {@link RetryExecutionException} 互补:RetryAttempt 是每一步的脚印(过程),
|
||||
* RetryExecutionException 是最终定格(attempts 总数 + 最后失败类型)。
|
||||
*
|
||||
* <p>构造校验保证记录必然自洽:成功不能带失败类型、失败必须带失败类型,
|
||||
* 避免把自相矛盾的脏记录写进 Trace。
|
||||
*/
|
||||
public record RetryAttempt(int attemptNumber, boolean success, RetryFailure failure) {
|
||||
|
||||
public RetryAttempt {
|
||||
@@ -14,10 +27,12 @@ public record RetryAttempt(int attemptNumber, boolean success, RetryFailure fail
|
||||
}
|
||||
}
|
||||
|
||||
/** 成功收据:failure 固定为 null(构造校验保证)。 */
|
||||
public static RetryAttempt succeeded(int attemptNumber) {
|
||||
return new RetryAttempt(attemptNumber, true, null);
|
||||
}
|
||||
|
||||
/** 失败收据:必须携带失败类型,供 Trace 和策略裁决参考。 */
|
||||
public static RetryAttempt failed(int attemptNumber, RetryFailure failure) {
|
||||
return new RetryAttempt(attemptNumber, false, failure);
|
||||
}
|
||||
|
||||
@@ -11,6 +11,8 @@ package com.superbiz.agent.harness.retry;
|
||||
*/
|
||||
public enum RetryFailure {
|
||||
|
||||
// ===== 可重试组:技术类失败,重试可能成功 =====
|
||||
|
||||
/** 单次 attempt 或总超时。 */
|
||||
TIMEOUT,
|
||||
|
||||
@@ -26,6 +28,8 @@ public enum RetryFailure {
|
||||
/** 结构符合 JSON 但 schema/字段约束失败。 */
|
||||
SCHEMA_INVALID,
|
||||
|
||||
// ===== 绝不重试组:业务/系统事实,重试不会改变结果 =====
|
||||
|
||||
/** 业务上判定无可用证据(若某组件使用该分类)。 */
|
||||
NO_EVIDENCE,
|
||||
|
||||
|
||||
@@ -2,6 +2,17 @@ package com.superbiz.agent.harness.retry;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* 重试策略:不可变值对象,表达「最多试几次 + 哪些失败类型可重试」。
|
||||
*
|
||||
* <p>不用布尔 {@code retry=true},而用 {@code maxAttempts + retryableFailures} 组合,
|
||||
* 因为重试必须同时回答两个问题:
|
||||
* <ul>
|
||||
* <li>还能不能再试(次数维度:{@code completedAttempts < maxAttempts});</li>
|
||||
* <li>这次失败值不值得试(类型维度:失败是否在可重试集合里)。</li>
|
||||
* </ul>
|
||||
* {@code maxAttempts} 限定为 1 或 2,防止配置膨胀成不可控的隐式重试。
|
||||
*/
|
||||
public record RetryPolicy(int maxAttempts, Set<RetryFailure> retryableFailures) {
|
||||
|
||||
public RetryPolicy {
|
||||
@@ -11,6 +22,13 @@ public record RetryPolicy(int maxAttempts, Set<RetryFailure> retryableFailures)
|
||||
retryableFailures = retryableFailures == null ? Set.of() : Set.copyOf(retryableFailures);
|
||||
}
|
||||
|
||||
/**
|
||||
* 双条件裁决:还有剩余次数 且 失败类型可重试,才允许下一次 attempt。
|
||||
*
|
||||
* <p>注意 {@code completedAttempts} 是「已完成(失败)的次数」:
|
||||
* 例如 maxAttempts=1(EvidenceRepair)时,第一次失败后
|
||||
* {@code 1 < 1} 为 false,永不重试。
|
||||
*/
|
||||
public boolean allowsRetry(int completedAttempts, RetryFailure failure) {
|
||||
return completedAttempts < maxAttempts && retryableFailures.contains(failure);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user