package com.superbiz.agent.harness.progress; import com.superbiz.agent.harness.contract.EvidenceStatus; import java.util.ArrayList; import java.util.LinkedHashSet; import java.util.List; import java.util.Set; /** * Progress 层的核心状态机:判定「继续收集证据是否还有价值」。 * *
与 RunBudget 的分工(双停止机制): *
设计要点: *
三类协议违规会被拒绝并抛 {@link ProgressProtocolViolationException}: *
校验通过后才清协议违规计数并应用 GAINED/NO_GAIN。协议错误与无增益是两件事: * 前者 Tool 根本没执行,后者 Tool 执行了但没推进诊断,因此必须分开统计。 */ public synchronized void applyPreviousObservation(PreviousObservation observation) { if (pendingToolCallId == null) { if (observation != null) { throw new ProgressProtocolViolationException( ProgressProtocolViolationType.UNEXPECTED_PREVIOUS_OBSERVATION, "No Tool observation is pending evaluation", "previous_observation", null); } // 没有 pending 且没带评价:正常(如首次调用),顺带清协议违规计数 clearProtocolViolations(); return; } if (observation == null) { throw new ProgressProtocolViolationException( ProgressProtocolViolationType.MISSING_PREVIOUS_OBSERVATION, "Previous Tool observation must be evaluated", "previous_observation", pendingToolCallId); } if (!pendingToolCallId.equals(observation.toolCallId())) { throw new ProgressProtocolViolationException( ProgressProtocolViolationType.OUT_OF_ORDER_PREVIOUS_OBSERVATION, "Previous Tool observation ID is out of order", "previous_observation.tool_call_id", pendingToolCallId); } // 校验通过:清空 pending,评价生效 pendingToolCallId = null; clearProtocolViolations(); applyGain(observation.informationGain()); } /** * 重复检测:toolName + normalizedScope 是否已被本 Run 完成过(backend 执行前调用)。 */ public synchronized boolean isDuplicate(String toolName, String normalizedScope) { return completedScopes.contains(new ToolScopeIdentity(toolName, normalizedScope)); } /** * 记录一次被判重的调用:Harness 直接判定为 NO_GAIN(backend 未被调用)。 */ public synchronized void recordDuplicateScope() { clearProtocolViolations(); applyGain(InformationGain.NO_GAIN); } /** * 记录一次成功的 Tool 完成。 * *
协议违规(缺评价/乱序/非法 Envelope)不计入 NO_GAIN——那是 Tool 执行了却没增益, * 而违规时 backend 从未执行。连续违规达到独立阈值后进入 SATURATED。 */ public synchronized DiagnosisProgressSnapshotState recordProgressProtocolViolation() { if (collectionState == DiagnosisCollectionState.SATURATED) { // 已饱和:不再累计,直接返回当前快照 return snapshot(); } consecutiveProgressProtocolViolations++; if (consecutiveProgressProtocolViolations >= stopAfterConsecutiveProgressProtocolViolations) { collectionState = DiagnosisCollectionState.SATURATED; stopReason = DiagnosisStopReason.PROGRESS_PROTOCOL_VIOLATED; } return snapshot(); } /** * 领取一次停止指令(STOP_REQUIRED)。 * *
只有 SATURATED 且尚未交付过时返回 true——给模型一次合法完成机会(输出 Draft), * 而不是立即抛错;之后模型仍请求 Tool 时由上层抛 * {@code DiagnosisCollectionStoppedException} 穿出框架 ReAct loop。 */ public synchronized boolean claimStopInstruction() { if (collectionState != DiagnosisCollectionState.SATURATED) { return false; } if (stopInstructionDelivered) { return false; } stopInstructionDelivered = true; return true; } /** * 标记预算触顶(由 RunBudget 侧调用)。 * *
只在尚无 stopReason 时设置 BUDGET_LIMIT_REACHED,不覆盖已有的 * INFORMATION_SATURATED / PROGRESS_PROTOCOL_VIOLATED——三种停止原因必须分开, * 信息饱和不能伪装成预算耗尽。 */ public synchronized void markBudgetLimitReached() { if (stopReason == null) { stopReason = DiagnosisStopReason.BUDGET_LIMIT_REACHED; } } /** * 返回内部控制快照(计数、pending、停止指令状态与已完成调用列表)。 */ public synchronized DiagnosisProgressSnapshotState snapshot() { return new DiagnosisProgressSnapshotState( consecutiveNoGain, consecutiveProgressProtocolViolations, collectionState, stopReason, pendingToolCallId, stopInstructionDelivered, completedToolCalls); } public int stopAfterConsecutiveNoGain() { return stopAfterConsecutiveNoGain; } public int stopAfterConsecutiveProgressProtocolViolations() { return stopAfterConsecutiveProgressProtocolViolations; } /** * 应用单次增益判定(核心状态迁移): *