feat(agent): add executor evidence output contract
This commit is contained in:
@@ -101,6 +101,8 @@ public class DiagnosisTraceEvaluator {
|
||||
failedChecks.add("reject output does not use degraded template");
|
||||
}
|
||||
|
||||
failedChecks.addAll(validateExecutorStructuredOutput(trace));
|
||||
|
||||
Integer toolCallCount = trace.getToolInvocations() == null ? 0 : trace.getToolInvocations().size();
|
||||
Integer durationMs = trace.getSession() == null ? null : trace.getSession().getTotalDurationMs();
|
||||
|
||||
@@ -174,6 +176,32 @@ public class DiagnosisTraceEvaluator {
|
||||
return value == null ? null : String.valueOf(value);
|
||||
}
|
||||
|
||||
private List<String> validateExecutorStructuredOutput(DiagnosisTraceResponse trace) {
|
||||
Object structuredOutput = nestedValue(trace, "verifier_evaluation", "executor_structured_output");
|
||||
if (!(structuredOutput instanceof Map<?, ?> output)) {
|
||||
return List.of();
|
||||
}
|
||||
Object claims = output.get("claims");
|
||||
if (!(claims instanceof List<?> claimList)) {
|
||||
return List.of("executor structured output missing claims array");
|
||||
}
|
||||
|
||||
List<String> failedChecks = new ArrayList<>();
|
||||
for (Object item : claimList) {
|
||||
if (!(item instanceof Map<?, ?> claim)) {
|
||||
failedChecks.add("executor structured claim is not an object");
|
||||
continue;
|
||||
}
|
||||
Object claimIdValue = claim.get("claim_id");
|
||||
String claimId = claimIdValue == null ? "unknown" : String.valueOf(claimIdValue);
|
||||
Object bindings = claim.get("evidence_bindings");
|
||||
if (!(bindings instanceof List<?> bindingList) || bindingList.isEmpty()) {
|
||||
failedChecks.add("executor confirmed claim missing evidence bindings: " + claimId);
|
||||
}
|
||||
}
|
||||
return failedChecks;
|
||||
}
|
||||
|
||||
private Object nestedValue(DiagnosisTraceResponse trace, String firstKey, String secondKey) {
|
||||
if (trace.getSession() == null || trace.getSession().getSelfEvaluation() == null) {
|
||||
return null;
|
||||
|
||||
@@ -5,6 +5,8 @@ import com.alibaba.cloud.ai.graph.agent.hook.HookPosition;
|
||||
import com.alibaba.cloud.ai.graph.agent.hook.HookPositions;
|
||||
import com.alibaba.cloud.ai.graph.agent.hook.messages.AgentCommand;
|
||||
import com.alibaba.cloud.ai.graph.agent.hook.messages.MessagesModelHook;
|
||||
import com.fasterxml.jackson.core.type.TypeReference;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.superbiz.agent.service.ToolTraceSummaryService;
|
||||
import com.superbiz.agent.util.SessionContextHolder;
|
||||
@@ -27,6 +29,8 @@ public class VerifierInputHook extends MessagesModelHook {
|
||||
|
||||
private final ToolTraceSummaryService toolTraceSummaryService;
|
||||
private final ObjectMapper objectMapper = new ObjectMapper();
|
||||
private static final TypeReference<Map<String, Object>> MAP_TYPE = new TypeReference<>() {
|
||||
};
|
||||
|
||||
public VerifierInputHook(ToolTraceSummaryService toolTraceSummaryService) {
|
||||
this.toolTraceSummaryService = toolTraceSummaryService;
|
||||
@@ -48,6 +52,10 @@ public class VerifierInputHook extends MessagesModelHook {
|
||||
executorFinalAnswer = extractLastAssistantText(previousMessages);
|
||||
}
|
||||
|
||||
ExecutorOutputParseResult parseResult = parseExecutorOutput(executorFinalAnswer);
|
||||
VerifierContextHolder.setExecutorStructuredOutput(parseResult.structuredOutput());
|
||||
VerifierContextHolder.setExecutorOutputParseStatus(parseResult.status());
|
||||
|
||||
List<Map<String, Object>> toolTraceSummary =
|
||||
toolTraceSummaryService.buildVerifierTraceSummary(sessionId, executorFinalAnswer);
|
||||
VerifierContextHolder.setToolTraceSummary(toolTraceSummary);
|
||||
@@ -55,6 +63,8 @@ public class VerifierInputHook extends MessagesModelHook {
|
||||
Map<String, Object> verifierInput = new LinkedHashMap<>();
|
||||
verifierInput.put("original_query", VerifierContextHolder.getOriginalQuery());
|
||||
verifierInput.put("executor_final_answer", executorFinalAnswer);
|
||||
verifierInput.put("executor_structured_output", parseResult.structuredOutput());
|
||||
verifierInput.put("executor_output_parse_status", parseResult.status());
|
||||
verifierInput.put("tool_trace_summary", toolTraceSummary);
|
||||
verifierInput.put("retry_context", VerifierContextHolder.getRetryContext());
|
||||
|
||||
@@ -66,6 +76,59 @@ public class VerifierInputHook extends MessagesModelHook {
|
||||
}
|
||||
}
|
||||
|
||||
private ExecutorOutputParseResult parseExecutorOutput(String executorFinalAnswer) {
|
||||
if (executorFinalAnswer == null || executorFinalAnswer.isBlank()) {
|
||||
return new ExecutorOutputParseResult(null, status("missing", "executor_final_answer is blank"));
|
||||
}
|
||||
|
||||
String sanitized = sanitizeJsonPayload(executorFinalAnswer);
|
||||
if (!looksJsonLike(sanitized)) {
|
||||
return new ExecutorOutputParseResult(null, status("missing", "executor output is not JSON"));
|
||||
}
|
||||
|
||||
try {
|
||||
JsonNode root = objectMapper.readTree(sanitized);
|
||||
if (!root.isObject() || !root.path("claims").isArray()) {
|
||||
return new ExecutorOutputParseResult(null, status("malformed",
|
||||
"executor output JSON does not match evidence-attribution contract"));
|
||||
}
|
||||
Map<String, Object> structuredOutput = objectMapper.convertValue(root, MAP_TYPE);
|
||||
return new ExecutorOutputParseResult(structuredOutput, status("valid", "parsed executor evidence contract"));
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to parse executor structured output", e);
|
||||
return new ExecutorOutputParseResult(null, status("malformed", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
private String sanitizeJsonPayload(String raw) {
|
||||
String trimmed = raw.trim();
|
||||
int fenceStart = trimmed.indexOf("```");
|
||||
if (fenceStart >= 0) {
|
||||
int firstNewline = trimmed.indexOf('\n', fenceStart);
|
||||
int lastFence = trimmed.indexOf("```", firstNewline + 1);
|
||||
if (firstNewline >= 0 && lastFence > firstNewline) {
|
||||
return trimmed.substring(firstNewline + 1, lastFence).trim();
|
||||
}
|
||||
}
|
||||
int objectStart = trimmed.indexOf('{');
|
||||
int objectEnd = trimmed.lastIndexOf('}');
|
||||
if (objectStart >= 0 && objectEnd > objectStart) {
|
||||
return trimmed.substring(objectStart, objectEnd + 1).trim();
|
||||
}
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
private boolean looksJsonLike(String text) {
|
||||
return text.startsWith("{") && text.endsWith("}");
|
||||
}
|
||||
|
||||
private Map<String, Object> status(String status, String detail) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("status", status);
|
||||
result.put("detail", detail == null ? "" : detail);
|
||||
return result;
|
||||
}
|
||||
|
||||
private String extractLastAssistantText(List<Message> previousMessages) {
|
||||
for (int i = previousMessages.size() - 1; i >= 0; i--) {
|
||||
if (previousMessages.get(i) instanceof AssistantMessage assistantMessage) {
|
||||
@@ -102,4 +165,10 @@ public class VerifierInputHook extends MessagesModelHook {
|
||||
}
|
||||
return message.toString();
|
||||
}
|
||||
|
||||
private record ExecutorOutputParseResult(
|
||||
Map<String, Object> structuredOutput,
|
||||
Map<String, Object> status
|
||||
) {
|
||||
}
|
||||
}
|
||||
|
||||
@@ -444,7 +444,8 @@ public class ChatService {
|
||||
}
|
||||
|
||||
if ("PASS".equals(finalDecision.verdict())) {
|
||||
answer = answer == null || answer.isBlank() ? "抱歉,多 Agent 分析未能生成有效结论。" : answer;
|
||||
answer = extractUserFacingAnswer(answer)
|
||||
.orElse(answer == null || answer.isBlank() ? "抱歉,多 Agent 分析未能生成有效结论。" : answer);
|
||||
persistVerifierEvaluation(session, finalDecision, round);
|
||||
break;
|
||||
}
|
||||
@@ -744,6 +745,10 @@ public class ChatService {
|
||||
verifierEvaluation.put("rationale", decision.rationale());
|
||||
verifierEvaluation.put("round", round);
|
||||
verifierEvaluation.put("traceability_version", "v1");
|
||||
verifierEvaluation.put("executor_output_parse_status",
|
||||
Optional.ofNullable(VerifierContextHolder.getExecutorOutputParseStatus())
|
||||
.orElse(Map.of("status", "missing", "detail", "executor parse status unavailable")));
|
||||
verifierEvaluation.put("executor_structured_output", VerifierContextHolder.getExecutorStructuredOutput());
|
||||
verifierEvaluation.put("tool_trace_summary",
|
||||
Optional.ofNullable(VerifierContextHolder.getToolTraceSummary()).orElse(List.of()));
|
||||
|
||||
@@ -795,6 +800,22 @@ public class ChatService {
|
||||
return output.toString();
|
||||
}
|
||||
|
||||
private Optional<String> extractUserFacingAnswer(String executorAnswer) {
|
||||
if (executorAnswer == null || executorAnswer.isBlank()) {
|
||||
return Optional.empty();
|
||||
}
|
||||
try {
|
||||
JsonNode root = objectMapper.readTree(sanitizeJsonPayload(executorAnswer));
|
||||
String userFacingAnswer = root.path("user_facing_answer").asText("");
|
||||
if (!userFacingAnswer.isBlank()) {
|
||||
return Optional.of(userFacingAnswer);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
logger.debug("Executor answer is not structured JSON, keep raw answer");
|
||||
}
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
private String buildDegradedOutput(VerifierDecision decision) {
|
||||
StringBuilder output = new StringBuilder(DEGRADED_PREFIX);
|
||||
|
||||
|
||||
@@ -11,6 +11,8 @@ public final class VerifierContextHolder {
|
||||
private static final ThreadLocal<String> ORIGINAL_QUERY = new ThreadLocal<>();
|
||||
private static final ThreadLocal<String> RETRY_CONTEXT = new ThreadLocal<>();
|
||||
private static final ThreadLocal<String> EXECUTOR_FINAL_ANSWER = new ThreadLocal<>();
|
||||
private static final ThreadLocal<Map<String, Object>> EXECUTOR_STRUCTURED_OUTPUT = new ThreadLocal<>();
|
||||
private static final ThreadLocal<Map<String, Object>> EXECUTOR_OUTPUT_PARSE_STATUS = new ThreadLocal<>();
|
||||
private static final ThreadLocal<List<Map<String, Object>>> TOOL_TRACE_SUMMARY = new ThreadLocal<>();
|
||||
|
||||
private VerifierContextHolder() {
|
||||
@@ -40,6 +42,22 @@ public final class VerifierContextHolder {
|
||||
return EXECUTOR_FINAL_ANSWER.get();
|
||||
}
|
||||
|
||||
public static void setExecutorStructuredOutput(Map<String, Object> executorStructuredOutput) {
|
||||
EXECUTOR_STRUCTURED_OUTPUT.set(executorStructuredOutput);
|
||||
}
|
||||
|
||||
public static Map<String, Object> getExecutorStructuredOutput() {
|
||||
return EXECUTOR_STRUCTURED_OUTPUT.get();
|
||||
}
|
||||
|
||||
public static void setExecutorOutputParseStatus(Map<String, Object> executorOutputParseStatus) {
|
||||
EXECUTOR_OUTPUT_PARSE_STATUS.set(executorOutputParseStatus);
|
||||
}
|
||||
|
||||
public static Map<String, Object> getExecutorOutputParseStatus() {
|
||||
return EXECUTOR_OUTPUT_PARSE_STATUS.get();
|
||||
}
|
||||
|
||||
public static void setToolTraceSummary(List<Map<String, Object>> toolTraceSummary) {
|
||||
TOOL_TRACE_SUMMARY.set(toolTraceSummary);
|
||||
}
|
||||
@@ -52,6 +70,8 @@ public final class VerifierContextHolder {
|
||||
ORIGINAL_QUERY.remove();
|
||||
RETRY_CONTEXT.remove();
|
||||
EXECUTOR_FINAL_ANSWER.remove();
|
||||
EXECUTOR_STRUCTURED_OUTPUT.remove();
|
||||
EXECUTOR_OUTPUT_PARSE_STATUS.remove();
|
||||
TOOL_TRACE_SUMMARY.remove();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,15 +1,17 @@
|
||||
你是任务执行器。执行 Planner 分配给你的具体步骤,并及时反馈结果。
|
||||
|
||||
## 职责
|
||||
- 按步骤执行具体的查询任务
|
||||
- 需要外部信息时调用工具,但须遵守下方的检索约束
|
||||
- 不要凭记忆回答,必须基于工具返回的真实数据
|
||||
- 执行完成后,综合所有结果给出完整的答案
|
||||
- 按步骤执行具体的查询任务。
|
||||
- 需要外部信息时调用工具,但必须遵守下方的检索约束。
|
||||
- 严禁凭记忆回答,必须基于本轮工具返回的真实数据。
|
||||
- 执行完成后,输出严格的证据归因 JSON,供 Verifier 校验。
|
||||
|
||||
## 规则
|
||||
- 按顺序执行,不可跳过步骤
|
||||
- 不要凭记忆回答,必须基于工具返回的真实数据
|
||||
- 执行完成后,综合所有结果给出完整的答案
|
||||
- 按顺序执行,不可跳过步骤。
|
||||
- 所有事实性结论必须来自本轮 evidence tools 的返回。
|
||||
- runbook、skill、历史案例、知识库中的通用模式只能作为排查指导或建议动作,不能直接写成本次事故的已确认事实。
|
||||
- 如果检索内容不足以支撑结论,必须显式声明证据不足,严禁补全事故故事。
|
||||
- 不要使用“通常情况下”“根据经验”“很可能已经发生”等无证据推断词来伪装事实。
|
||||
|
||||
## 检索约束
|
||||
|
||||
@@ -20,22 +22,99 @@
|
||||
|
||||
### 2. 重复了该怎么办
|
||||
如果当前想检索的内容与【已检索上下文】语义相似:
|
||||
- 禁止换关键词重新检索
|
||||
- 直接基于已有事实回答
|
||||
- 禁止换关键词重新检索。
|
||||
- 直接基于已有事实回答。
|
||||
- 如果信息不足,先明确指出缺少什么具体维度
|
||||
(如:"缺少 HikariCP 具体配置参数"、"缺少连接池耗尽的日志样例"),
|
||||
再针对该维度进行一次定向补充检索——而非盲目换词重查
|
||||
(如:“缺少 HikariCP 具体配置参数”、“缺少连接池耗尽的日志样例”),
|
||||
再针对该维度进行一次定向补充检索,而非盲目换词重查。
|
||||
|
||||
### 3. 合法出口:允许信息不全时给出结论
|
||||
如果你认为已有信息足以回答核心问题,即使细节不全,
|
||||
也请直接给出结论并说明局限性(如:"基于已有信息,连接池配置建议如下,
|
||||
但具体参数值需结合实际负载调整")。
|
||||
**不查全不会被追责,重复检索才会被惩罚。**
|
||||
### 3. 合法出口:允许信息不全时给出有限结论
|
||||
如果已有信息足以回答核心问题,即使细节不全,也可以给出有限结论。
|
||||
但你只能把有证据支撑的内容放入 `claims`。
|
||||
缺失的细节必须写入 `missing_info`,可疑但未证实的方向必须写入 `hypotheses`。
|
||||
**不查全不会被追责,重复检索或编造细节才会被惩罚。**
|
||||
|
||||
### 4. 利用质量信号判断
|
||||
- relevanceLevel=PRECISE → 信息精准,直接使用,不再检索
|
||||
- relevanceLevel=HIGHLY_RELEVANT + 域已在 retrievedDomainsThisSession → 禁止再次调用
|
||||
- relevanceLevel=REFERENCE → 先指出缺什么维度,再定向补充一次
|
||||
- completenessHint 是知识库给你的天花板信号,信任它
|
||||
- lookup_knowledge 的事实证据以 evidenceBlocks 和 contextPack.packedText 为准,不要假设 L0 hint 本身就是事实证据
|
||||
- retrievalTrace 只用于理解检索路径和降级原因,不能单独作为诊断事实
|
||||
- relevanceLevel=PRECISE → 信息精准,直接使用,不再检索。
|
||||
- relevanceLevel=HIGHLY_RELEVANT + 域已在 retrievedDomainsThisSession → 禁止再次调用。
|
||||
- relevanceLevel=REFERENCE → 先指出缺什么维度,再定向补充一次。
|
||||
- completenessHint 是知识库给你的天花板信号,信任它。
|
||||
- lookup_knowledge 的事实证据以 evidenceBlocks 和 contextPack.packedText 为准,不要假设 L0 hint 本身就是事实证据。
|
||||
- retrievalTrace 只用于理解检索路径和降级原因,不能单独作为诊断事实。
|
||||
|
||||
## 证据归因要求
|
||||
|
||||
### confirmed claims
|
||||
`claims` 只允许放已证实或有明确间接支撑的事实断言。
|
||||
每条 claim 必须带证据绑定。
|
||||
|
||||
支持等级:
|
||||
- `direct`:工具返回中有直接事实。
|
||||
- `indirect`:工具返回可支撑方向,但没有直接陈述完整结论。
|
||||
- `none`:不能放入 `claims`,应放入 `hypotheses`、`recommended_actions` 或 `missing_info`。
|
||||
|
||||
### hypotheses
|
||||
`hypotheses` 用来放合理怀疑但未被工具证实的方向。
|
||||
例如:工具只显示连接池耗尽,但没有泄漏日志,则“可能存在连接泄漏”只能是 hypothesis。
|
||||
|
||||
### recommended_actions
|
||||
`recommended_actions` 用来放下一步排查或修复动作。
|
||||
建议可以来自 runbook/skill,但必须说明 reason,不能写成“已确认根因”。
|
||||
|
||||
### missing_info
|
||||
`missing_info` 用来列出无法确认结论所缺少的具体证据。
|
||||
|
||||
## 最终输出格式(严格契约)
|
||||
|
||||
你必须输出且只能输出一个 JSON 对象,不要输出 Markdown,不要输出代码块,不要输出 JSON 之外的解释文字。
|
||||
所有用户可读内容必须使用中文。
|
||||
|
||||
```json
|
||||
{
|
||||
"answer_version": "executor_evidence_v1",
|
||||
"diagnosis_summary": "1-2句话总结,仅包含有证据支撑的事实和证据边界",
|
||||
"claims": [
|
||||
{
|
||||
"claim_id": "claim-1",
|
||||
"claim_type": "root_cause",
|
||||
"claim_text": "事实断言或有限结论",
|
||||
"support_level": "direct",
|
||||
"evidence_bindings": [
|
||||
{
|
||||
"source_type": "tool_trace",
|
||||
"source_id": "工具返回中的 evidence block id、trace_ref 或可定位标识",
|
||||
"tool_name": "lookup_knowledge/query_logs/query_metrics/read_skill 等",
|
||||
"source_invocation_ids": [],
|
||||
"evidence_excerpt": "从工具返回中摘取的原话、指标值、日志片段或关键数据"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"hypotheses": [
|
||||
{
|
||||
"hypothesis_text": "未被证实但值得排查的方向",
|
||||
"basis": "它基于哪些已知证据或为什么只是推测",
|
||||
"needed_evidence": ["需要补充的证据"]
|
||||
}
|
||||
],
|
||||
"recommended_actions": [
|
||||
{
|
||||
"action_text": "建议动作",
|
||||
"reason": "为什么建议做这个动作",
|
||||
"evidence_bindings": []
|
||||
}
|
||||
],
|
||||
"missing_info": [
|
||||
"导致无法确认完整根因的证据缺口"
|
||||
],
|
||||
"user_facing_answer": "面向用户的中文回答。必须与 claims/hypotheses/recommended_actions/missing_info 一致,不得额外加入未绑定证据的确认式事实。"
|
||||
}
|
||||
```
|
||||
|
||||
## 输出校验
|
||||
- `claims[*].support_level` 只能是 `direct` 或 `indirect`。
|
||||
- `claims[*].evidence_bindings` 不能为空。
|
||||
- `evidence_excerpt` 必须来自工具返回,不允许编造。
|
||||
- 如果没有任何可确认事实,`claims` 返回空数组,并在 `missing_info` 说明缺少什么。
|
||||
- `user_facing_answer` 不得出现 `claims` 中没有、且又被写成确认结论的事实。
|
||||
- 不要把其它服务、其它历史案例、其它会话的事实迁移为当前会话事实。
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
|
||||
- `original_query`:用户原始问题
|
||||
- `executor_final_answer`:本轮 Executor 最终答案
|
||||
- `executor_structured_output`:如果 Executor 输出了合法证据归因 JSON,这里会提供解析后的对象。结构包含 `claims`、`hypotheses`、`recommended_actions`、`missing_info`、`user_facing_answer`
|
||||
- `executor_output_parse_status`:Executor 输出解析状态,包含 `status` 和 `detail`。`status` 可能是 `valid` / `missing` / `malformed`
|
||||
- `tool_trace_summary`:基于真实工具调用整理出的证据索引。每一项都带有:
|
||||
- `trace_ref`
|
||||
- `tool_name`
|
||||
@@ -23,7 +25,18 @@
|
||||
## 任务步骤
|
||||
|
||||
### 步骤一:提取关键事实
|
||||
优先提取并校验 `executor_final_answer` 里的全部实质性结论。关键事实至少包括:
|
||||
如果 `executor_output_parse_status.status="valid"` 且 `executor_structured_output.claims` 存在:
|
||||
- 优先逐条校验 `executor_structured_output.claims`
|
||||
- 每个 claim 至少形成一条 `facts_checked`
|
||||
- 必须检查 claim 的 `evidence_bindings` 是否能对应到 `tool_trace_summary` 中真实存在的 trace、tool 或 source_invocation_ids
|
||||
- 如果 claim 声称 direct/indirect 支撑,但 evidence binding 不存在、无法定位、或 excerpt 与工具摘要不匹配,不得判为 `direct_evidence`
|
||||
|
||||
然后必须扫描 `executor_structured_output.user_facing_answer`:
|
||||
- 如果其中出现 confirmed-sounding facts(确认式事实、根因、指标值、错误码、服务名、修复结论)
|
||||
- 且这些事实没有出现在 `executor_structured_output.claims`
|
||||
- 必须额外加入 `facts_checked` 并按工具证据校验
|
||||
|
||||
如果 structured output 缺失或 malformed,则回退到旧逻辑:提取并校验 `executor_final_answer` 里的全部实质性结论。关键事实至少包括:
|
||||
- 每一个根因结论
|
||||
- 每一个错误码、接口、组件归属或语义判断
|
||||
- 每一个明确的修复建议、参数建议、排查步骤
|
||||
@@ -49,6 +62,15 @@
|
||||
- `no_evidence`
|
||||
- `contradicted`
|
||||
|
||||
结构化 claim 的校验规则:
|
||||
- claim 有真实 evidence binding,且工具摘要直接包含该事实 → `direct_evidence`
|
||||
- claim 有真实 evidence binding,但工具摘要只能支持方向或背景 → `indirect_support`
|
||||
- claim 无法绑定真实 trace、invocation 或 excerpt → `no_evidence`
|
||||
- claim 与工具摘要冲突,或编造了不存在的关键实体、服务、错误码、指标值 → `contradicted`
|
||||
|
||||
`hypotheses` 和 `missing_info` 默认不是 confirmed facts,不应因为它们承认缺证据而惩罚。
|
||||
但如果 `user_facing_answer` 把 hypothesis 写成确认结论,必须按 confirmed fact 校验。
|
||||
|
||||
### 步骤三:补齐 evidence_refs
|
||||
`evidence_refs` 必须是数组,数组元素必须引用 `tool_trace_summary` 中真实存在的证据项。每个元素包含:
|
||||
- `trace_ref`
|
||||
|
||||
Reference in New Issue
Block a user