- Annotate DiagnosisProgressTracker, HarnessToolInterceptor, DiagnosisProgressProjector, DiagnosisReleaseUseCase, ToolBoundary, ToolBoundaryResult, CanonicalToolInvocation - Add progress code learning note: interceptor gates, tracker state machine, canonical lifecycle, execution gate, projection/release pipeline - Update learning roadmap: progress marked as deeply learned, next is tool domain
283 lines
13 KiB
Java
283 lines
13 KiB
Java
package com.superbiz.agent.harness.progress;
|
||
|
||
import com.superbiz.agent.harness.contract.EvidenceStatus;
|
||
|
||
import java.util.ArrayList;
|
||
import java.util.LinkedHashSet;
|
||
import java.util.List;
|
||
import java.util.Set;
|
||
|
||
/**
|
||
* Progress 层的核心状态机:判定「继续收集证据是否还有价值」。
|
||
*
|
||
* <p>与 RunBudget 的分工(双停止机制):
|
||
* <ul>
|
||
* <li>RunBudget 管「能不能花」——模型次数 / Tool 次数 / Token / bytes 等硬资源上限;</li>
|
||
* <li>本 Tracker 管「继续查有没有价值」——连续 NO_GAIN、重复 scope、协议违规都会推动
|
||
* 收集状态走向 SATURATED,进而在硬预算之前让 Agent 受控停止。</li>
|
||
* </ul>
|
||
*
|
||
* <p>设计要点:
|
||
* <ul>
|
||
* <li>只保存做停止决策需要的最小状态(identity + 计数),不保存 request / raw /
|
||
* agent result,避免出现第二份 Tool 真相(完整事实在 Canonical Store);</li>
|
||
* <li>所有状态读写 synchronized,是 RunContext 中的线程安全单一所有者;</li>
|
||
* <li>Tool 提供客观结果,模型判断语义增益(GAINED/NO_GAIN),但最终停止权归 Harness。</li>
|
||
* </ul>
|
||
*/
|
||
public final class DiagnosisProgressTracker {
|
||
|
||
/** 连续 NO_GAIN 达到该阈值 → SATURATED + INFORMATION_SATURATED(默认 2)。 */
|
||
private final int stopAfterConsecutiveNoGain;
|
||
/** 连续 progress 协议违规达到该阈值 → SATURATED + PROGRESS_PROTOCOL_VIOLATED(默认 2)。 */
|
||
private final int stopAfterConsecutiveProgressProtocolViolations;
|
||
/** 已完成调用的去重集合:toolName + normalizedScope,backend 执行前判重。 */
|
||
private final Set<ToolScopeIdentity> completedScopes = new LinkedHashSet<>();
|
||
/** 已完成调用的 identity 列表(无 payload),供结束时 Projector 回读 canonical。 */
|
||
private final List<CompletedToolCall> completedToolCalls = new ArrayList<>();
|
||
/** 连续无增益次数;GAINED 清零。 */
|
||
private int consecutiveNoGain;
|
||
/** 连续协议违规次数;一次合法评价(或无 pending 的合法调用)后清零。 */
|
||
private int consecutiveProgressProtocolViolations;
|
||
/** 收集状态机:COLLECTING(可继续收集)→ SATURATED(已饱和,只能停止)。 */
|
||
private DiagnosisCollectionState collectionState = DiagnosisCollectionState.COLLECTING;
|
||
/** 停止原因:INFORMATION_SATURATED / BUDGET_LIMIT_REACHED / PROGRESS_PROTOCOL_VIOLATED。 */
|
||
private DiagnosisStopReason stopReason;
|
||
/** 等待模型评价的 tool_call_id;同一时刻最多一个 pending。 */
|
||
private String pendingToolCallId;
|
||
/** 一次 STOP_REQUIRED 指令是否已交付(claimStopInstruction 只成功一次)。 */
|
||
private boolean stopInstructionDelivered;
|
||
|
||
/**
|
||
* 单参数构造:连续 NO_GAIN 阈值显式指定,协议违规阈值使用默认值 2。
|
||
*/
|
||
public DiagnosisProgressTracker(int stopAfterConsecutiveNoGain) {
|
||
this(stopAfterConsecutiveNoGain, 2);
|
||
}
|
||
|
||
/**
|
||
* 全参构造:两个连续停止阈值都必须为正数(不允许 0 或负数)。
|
||
*
|
||
* @param stopAfterConsecutiveNoGain 连续 NO_GAIN 达到该次数即饱和
|
||
* @param stopAfterConsecutiveProgressProtocolViolations 连续协议违规达到该次数即饱和
|
||
*/
|
||
public DiagnosisProgressTracker(
|
||
int stopAfterConsecutiveNoGain,
|
||
int stopAfterConsecutiveProgressProtocolViolations) {
|
||
if (stopAfterConsecutiveNoGain <= 0) {
|
||
throw new IllegalArgumentException("stopAfterConsecutiveNoGain must be positive");
|
||
}
|
||
if (stopAfterConsecutiveProgressProtocolViolations <= 0) {
|
||
throw new IllegalArgumentException(
|
||
"stopAfterConsecutiveProgressProtocolViolations must be positive");
|
||
}
|
||
this.stopAfterConsecutiveNoGain = stopAfterConsecutiveNoGain;
|
||
this.stopAfterConsecutiveProgressProtocolViolations =
|
||
stopAfterConsecutiveProgressProtocolViolations;
|
||
}
|
||
|
||
/**
|
||
* 应用模型在下次 Tool Call 中回传的对上一轮观察的评价。
|
||
*
|
||
* <p>三类协议违规会被拒绝并抛 {@link ProgressProtocolViolationException}:
|
||
* <ul>
|
||
* <li>{@link ProgressProtocolViolationType#UNEXPECTED_PREVIOUS_OBSERVATION}——没有 pending
|
||
* 时却带了 previous_observation(如首次调用);</li>
|
||
* <li>{@link ProgressProtocolViolationType#MISSING_PREVIOUS_OBSERVATION}——有 pending 却
|
||
* 没带评价;</li>
|
||
* <li>{@link ProgressProtocolViolationType#OUT_OF_ORDER_PREVIOUS_OBSERVATION}——带的
|
||
* tool_call_id 与 pending 不符(乱序/指向未知调用)。</li>
|
||
* </ul>
|
||
*
|
||
* <p>校验通过后才清协议违规计数并应用 GAINED/NO_GAIN。协议错误与无增益是两件事:
|
||
* 前者 Tool 根本没执行,后者 Tool 执行了但没推进诊断,因此必须分开统计。
|
||
*/
|
||
public synchronized void applyPreviousObservation(PreviousObservation observation) {
|
||
if (pendingToolCallId == null) {
|
||
if (observation != null) {
|
||
throw new ProgressProtocolViolationException(
|
||
ProgressProtocolViolationType.UNEXPECTED_PREVIOUS_OBSERVATION,
|
||
"No Tool observation is pending evaluation",
|
||
"previous_observation",
|
||
null);
|
||
}
|
||
// 没有 pending 且没带评价:正常(如首次调用),顺带清协议违规计数
|
||
clearProtocolViolations();
|
||
return;
|
||
}
|
||
if (observation == null) {
|
||
throw new ProgressProtocolViolationException(
|
||
ProgressProtocolViolationType.MISSING_PREVIOUS_OBSERVATION,
|
||
"Previous Tool observation must be evaluated",
|
||
"previous_observation",
|
||
pendingToolCallId);
|
||
}
|
||
if (!pendingToolCallId.equals(observation.toolCallId())) {
|
||
throw new ProgressProtocolViolationException(
|
||
ProgressProtocolViolationType.OUT_OF_ORDER_PREVIOUS_OBSERVATION,
|
||
"Previous Tool observation ID is out of order",
|
||
"previous_observation.tool_call_id",
|
||
pendingToolCallId);
|
||
}
|
||
// 校验通过:清空 pending,评价生效
|
||
pendingToolCallId = null;
|
||
clearProtocolViolations();
|
||
applyGain(observation.informationGain());
|
||
}
|
||
|
||
/**
|
||
* 重复检测:toolName + normalizedScope 是否已被本 Run 完成过(backend 执行前调用)。
|
||
*/
|
||
public synchronized boolean isDuplicate(String toolName, String normalizedScope) {
|
||
return completedScopes.contains(new ToolScopeIdentity(toolName, normalizedScope));
|
||
}
|
||
|
||
/**
|
||
* 记录一次被判重的调用:Harness 直接判定为 NO_GAIN(backend 未被调用)。
|
||
*/
|
||
public synchronized void recordDuplicateScope() {
|
||
clearProtocolViolations();
|
||
applyGain(InformationGain.NO_GAIN);
|
||
}
|
||
|
||
/**
|
||
* 记录一次成功的 Tool 完成。
|
||
*
|
||
* <ul>
|
||
* <li>SATURATED 后禁止再记录完成(饱和即停止收集);</li>
|
||
* <li>只接受 {@link EvidenceStatus#EVIDENCE_FOUND} 或 {@link EvidenceStatus#NO_EVIDENCE};
|
||
* 失败走技术失败流程,不进入进度统计;</li>
|
||
* <li>重复 scope 抛 IllegalStateException(应在此之前被 isDuplicate 拦截);</li>
|
||
* <li>{@link EvidenceStatus#NO_EVIDENCE}:空结果由 Harness 直接判 NO_GAIN,不需要模型评价;</li>
|
||
* <li>{@link EvidenceStatus#EVIDENCE_FOUND}:设置 pendingToolCallId,等模型在下次
|
||
* Tool Call 的 previous_observation 中评价语义增益。</li>
|
||
* </ul>
|
||
*/
|
||
public synchronized void recordCompleted(CompletedToolCall call, EvidenceStatus evidenceStatus) {
|
||
if (collectionState == DiagnosisCollectionState.SATURATED) {
|
||
throw new IllegalStateException("Cannot record Tool completion after saturation");
|
||
}
|
||
if (evidenceStatus != EvidenceStatus.EVIDENCE_FOUND
|
||
&& evidenceStatus != EvidenceStatus.NO_EVIDENCE) {
|
||
throw new IllegalArgumentException("Completed Tool requires a successful evidence status");
|
||
}
|
||
ToolScopeIdentity scope = new ToolScopeIdentity(call.toolName(), call.normalizedScope());
|
||
if (!completedScopes.add(scope)) {
|
||
throw new IllegalStateException("Completed Tool scope was already recorded");
|
||
}
|
||
completedToolCalls.add(call);
|
||
if (evidenceStatus == EvidenceStatus.NO_EVIDENCE) {
|
||
// 空结果无需模型评价:立即累计 NO_GAIN
|
||
clearProtocolViolations();
|
||
applyGain(InformationGain.NO_GAIN);
|
||
} else {
|
||
// 非空结果:挂起等待模型在下一轮评价语义增益
|
||
pendingToolCallId = call.toolCallId();
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 记录一次 progress 协议违规,返回最新快照。
|
||
*
|
||
* <p>协议违规(缺评价/乱序/非法 Envelope)不计入 NO_GAIN——那是 Tool 执行了却没增益,
|
||
* 而违规时 backend 从未执行。连续违规达到独立阈值后进入 SATURATED。
|
||
*/
|
||
public synchronized DiagnosisProgressSnapshotState recordProgressProtocolViolation() {
|
||
if (collectionState == DiagnosisCollectionState.SATURATED) {
|
||
// 已饱和:不再累计,直接返回当前快照
|
||
return snapshot();
|
||
}
|
||
consecutiveProgressProtocolViolations++;
|
||
if (consecutiveProgressProtocolViolations
|
||
>= stopAfterConsecutiveProgressProtocolViolations) {
|
||
collectionState = DiagnosisCollectionState.SATURATED;
|
||
stopReason = DiagnosisStopReason.PROGRESS_PROTOCOL_VIOLATED;
|
||
}
|
||
return snapshot();
|
||
}
|
||
|
||
/**
|
||
* 领取一次停止指令(STOP_REQUIRED)。
|
||
*
|
||
* <p>只有 SATURATED 且尚未交付过时返回 true——给模型一次合法完成机会(输出 Draft),
|
||
* 而不是立即抛错;之后模型仍请求 Tool 时由上层抛
|
||
* {@code DiagnosisCollectionStoppedException} 穿出框架 ReAct loop。
|
||
*/
|
||
public synchronized boolean claimStopInstruction() {
|
||
if (collectionState != DiagnosisCollectionState.SATURATED) {
|
||
return false;
|
||
}
|
||
if (stopInstructionDelivered) {
|
||
return false;
|
||
}
|
||
stopInstructionDelivered = true;
|
||
return true;
|
||
}
|
||
|
||
/**
|
||
* 标记预算触顶(由 RunBudget 侧调用)。
|
||
*
|
||
* <p>只在尚无 stopReason 时设置 BUDGET_LIMIT_REACHED,不覆盖已有的
|
||
* INFORMATION_SATURATED / PROGRESS_PROTOCOL_VIOLATED——三种停止原因必须分开,
|
||
* 信息饱和不能伪装成预算耗尽。
|
||
*/
|
||
public synchronized void markBudgetLimitReached() {
|
||
if (stopReason == null) {
|
||
stopReason = DiagnosisStopReason.BUDGET_LIMIT_REACHED;
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 返回内部控制快照(计数、pending、停止指令状态与已完成调用列表)。
|
||
*/
|
||
public synchronized DiagnosisProgressSnapshotState snapshot() {
|
||
return new DiagnosisProgressSnapshotState(
|
||
consecutiveNoGain,
|
||
consecutiveProgressProtocolViolations,
|
||
collectionState,
|
||
stopReason,
|
||
pendingToolCallId,
|
||
stopInstructionDelivered,
|
||
completedToolCalls);
|
||
}
|
||
|
||
public int stopAfterConsecutiveNoGain() {
|
||
return stopAfterConsecutiveNoGain;
|
||
}
|
||
|
||
public int stopAfterConsecutiveProgressProtocolViolations() {
|
||
return stopAfterConsecutiveProgressProtocolViolations;
|
||
}
|
||
|
||
/**
|
||
* 应用单次增益判定(核心状态迁移):
|
||
* <ul>
|
||
* <li>GAINED:清零连续 NO_GAIN——一次早期空查不能使后续有效取证被过早停止;</li>
|
||
* <li>NO_GAIN:累加,达到阈值 → SATURATED + INFORMATION_SATURATED。</li>
|
||
* </ul>
|
||
* 饱和后禁止再次应用(停止权只行使一次)。
|
||
*/
|
||
private void applyGain(InformationGain gain) {
|
||
if (collectionState == DiagnosisCollectionState.SATURATED) {
|
||
throw new IllegalStateException("Collection is already saturated");
|
||
}
|
||
if (gain == InformationGain.GAINED) {
|
||
consecutiveNoGain = 0;
|
||
return;
|
||
}
|
||
consecutiveNoGain++;
|
||
if (consecutiveNoGain >= stopAfterConsecutiveNoGain) {
|
||
collectionState = DiagnosisCollectionState.SATURATED;
|
||
stopReason = DiagnosisStopReason.INFORMATION_SATURATED;
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 清空协议违规计数:一次合法评价(或没有 pending 的合法调用)都会重置,
|
||
* 避免历史违规累积导致误饱和(协议违规只按「连续」计数)。
|
||
*/
|
||
private void clearProtocolViolations() {
|
||
consecutiveProgressProtocolViolations = 0;
|
||
}
|
||
}
|