feat(agent): support no-evidence references

This commit is contained in:
aruo
2026-07-08 23:30:00 +08:00
parent 7b8c75e571
commit 9a84b3de34
9 changed files with 1040 additions and 28 deletions
@@ -156,7 +156,8 @@ public class ExecutorGatekeeperService {
SEVERITY_LOW_CONFID);
continue;
}
validateEvidenceBinding(binding, validInvocations, target, claim.get("claim_id"), result);
validateEvidenceBinding(binding, validInvocations, target,
claim.get("claim_id"), claim.get("claim_type"), result);
}
}
@@ -181,7 +182,7 @@ public class ExecutorGatekeeperService {
SEVERITY_LOW_CONFID);
continue;
}
validateEvidenceBinding(binding, validInvocations, target, action.get("action_id"), result);
validateEvidenceBinding(binding, validInvocations, target, action.get("action_id"), null, result);
}
}
}
@@ -190,6 +191,7 @@ public class ExecutorGatekeeperService {
Map<Long, ToolInvocation> validInvocations,
String target,
Object ownerId,
Object ownerType,
GatekeeperResult result) {
Map<String, Object> checked = new LinkedHashMap<>();
checked.put("claim_id", ownerId == null ? "" : String.valueOf(ownerId));
@@ -197,18 +199,6 @@ public class ExecutorGatekeeperService {
checked.put("source_invocation_id", binding.get("source_invocation_id"));
checked.put("raw_path", stringValue(binding.get("raw_path")));
Long id = singleInvocationId(binding);
if (id == null) {
checked.put("status", STATUS_FAIL);
checked.put("rule", RULE_INVOCATION_REF);
checked.put("message", "source_invocation_id is required");
result.checked(checked);
result.fail(RULE_INVOCATION_REF, target + ".source_invocation_id",
"source_invocation_id is required", SEVERITY_LOW_CONFID);
return;
}
checked.put("source_invocation_id", id);
String claimedToolName = stringValue(binding.get("tool_name"));
if (claimedToolName.isBlank()) {
checked.put("status", STATUS_FAIL);
@@ -219,6 +209,46 @@ public class ExecutorGatekeeperService {
return;
}
String rawPath = stringValue(binding.get("raw_path"));
if ("negative_observation".equals(stringValue(ownerType)) && !rawPath.isBlank()
&& !"$.no_evidence".equals(rawPath)) {
checked.put("status", STATUS_FAIL);
checked.put("rule", RULE_RAW_PATH);
checked.put("message", "negative_observation must only bind $.no_evidence references");
result.checked(checked);
result.fail(RULE_RAW_PATH, target + ".raw_path",
"negative_observation must only bind $.no_evidence references", SEVERITY_REJECT);
return;
}
Long id = singleInvocationId(binding);
if (id == null) {
id = uniqueInvocationIdByToolRawPathAndExcerpt(validInvocations, claimedToolName, rawPath,
stringValue(binding.get("evidence_excerpt")));
if (id == null) {
id = uniqueInvocationIdByToolAndRawPath(validInvocations, claimedToolName, rawPath);
}
if (id != null) {
result.warn(Map.of(
"rule", "evidence.invocation_auto_backfill_by_raw_path",
"message", "source_invocation_id was auto-filled from the unique evidence reference candidate",
"tool_name", claimedToolName,
"raw_path", rawPath,
"source_invocation_id", id
));
}
}
if (id == null) {
checked.put("status", STATUS_FAIL);
checked.put("rule", RULE_INVOCATION_REF);
checked.put("message", "source_invocation_id is required");
result.checked(checked);
result.fail(RULE_INVOCATION_REF, target + ".source_invocation_id",
"source_invocation_id is required", SEVERITY_LOW_CONFID);
return;
}
checked.put("source_invocation_id", id);
ToolInvocation invocation = validInvocations.get(id);
if (invocation == null) {
checked.put("status", STATUS_FAIL);
@@ -240,7 +270,6 @@ public class ExecutorGatekeeperService {
return;
}
String rawPath = stringValue(binding.get("raw_path"));
if (rawPath.isBlank()) {
checked.put("status", STATUS_FAIL);
checked.put("rule", RULE_RAW_PATH);
@@ -298,6 +327,82 @@ public class ExecutorGatekeeperService {
result.checked(checked);
}
private Long uniqueInvocationIdByToolAndRawPath(Map<Long, ToolInvocation> validInvocations,
String toolName,
String rawPath) {
if (toolName == null || toolName.isBlank() || rawPath == null || rawPath.isBlank()) {
return null;
}
Long matchedId = null;
for (Map.Entry<Long, ToolInvocation> entry : validInvocations.entrySet()) {
ToolInvocation invocation = entry.getValue();
if (!Objects.equals(toolName, invocation.getToolName())) {
continue;
}
if (!evidenceRefsByRawPath(invocation.getRetrievalDetails()).containsKey(rawPath)) {
continue;
}
if (matchedId != null) {
return null;
}
matchedId = entry.getKey();
}
return matchedId;
}
private Long uniqueInvocationIdByToolRawPathAndExcerpt(Map<Long, ToolInvocation> validInvocations,
String toolName,
String rawPath,
String excerpt) {
if (toolName == null || toolName.isBlank()
|| rawPath == null || rawPath.isBlank()
|| excerpt == null || excerpt.isBlank()) {
return null;
}
Long matchedId = null;
for (Map.Entry<Long, ToolInvocation> entry : validInvocations.entrySet()) {
ToolInvocation invocation = entry.getValue();
if (!Objects.equals(toolName, invocation.getToolName())) {
continue;
}
String matchedText = evidenceRefsByRawPath(invocation.getRetrievalDetails()).get(rawPath);
if (matchedText == null || !isBackfillCandidateSupported(rawPath, excerpt, matchedText)) {
continue;
}
if (matchedId != null) {
return null;
}
matchedId = entry.getKey();
}
return matchedId;
}
private boolean isBackfillCandidateSupported(String rawPath, String excerpt, String matchedText) {
if ("$.no_evidence".equals(rawPath)) {
String excerptQuery = semicolonField(excerpt, "query");
String matchedQuery = semicolonField(matchedText, "query");
if (!excerptQuery.isBlank() && !matchedQuery.isBlank()
&& !normalized(excerptQuery).equals(normalized(matchedQuery))) {
return false;
}
}
return isExcerptSupported(excerpt, matchedText);
}
private String semicolonField(String text, String field) {
if (text == null || text.isBlank() || field == null || field.isBlank()) {
return "";
}
String prefix = field + "=";
for (String part : text.split(";")) {
String trimmed = part.trim();
if (trimmed.regionMatches(true, 0, prefix, 0, prefix.length())) {
return trimmed.substring(prefix.length()).trim();
}
}
return "";
}
private Long singleInvocationId(Map<?, ?> binding) {
Long singular = asLong(binding.get("source_invocation_id"));
if (singular != null) {
@@ -88,7 +88,7 @@ public class ToolInvocationRecorder {
if (extraDetails != null && !extraDetails.isEmpty()) {
details.putAll(extraDetails);
}
List<Map<String, Object>> evidenceRefs = extractEvidenceRefs(toolName, output, details);
List<Map<String, Object>> evidenceRefs = extractEvidenceRefs(toolName, inputParams, output, details);
if (!evidenceRefs.isEmpty()) {
details.put("evidence_refs", evidenceRefs);
}
@@ -145,6 +145,12 @@ public class ToolInvocationRecorder {
}
details.put("evidence_blocks", record.evidenceBlocks() == null ? List.of() : record.evidenceBlocks());
List<Map<String, Object>> evidenceRefs = evidenceRefsFromEvidenceBlocks(record.evidenceBlocks());
if (evidenceRefs.isEmpty()
&& EVIDENCE_STATUS_NO_EVIDENCE.equals(normalizeEvidenceStatus(record.success(), record.evidenceStatus()))) {
Map<String, Object> input = new LinkedHashMap<>();
input.put("query", record.query());
evidenceRefs = List.of(noEvidenceRef("lookup_knowledge", input, null, details));
}
if (!evidenceRefs.isEmpty()) {
details.put("evidence_refs", evidenceRefs);
}
@@ -195,7 +201,10 @@ public class ToolInvocationRecorder {
return success ? EVIDENCE_STATUS_SUPPORTED : EVIDENCE_STATUS_FAILED;
}
private List<Map<String, Object>> extractEvidenceRefs(String toolName, String output, Map<String, Object> details) {
private List<Map<String, Object>> extractEvidenceRefs(String toolName,
Map<String, Object> inputParams,
String output,
Map<String, Object> details) {
if (output == null || output.isBlank()) {
return List.of();
}
@@ -205,11 +214,14 @@ public class ToolInvocationRecorder {
try {
JsonNode root = objectMapper.readTree(output);
boolean noEvidence = EVIDENCE_STATUS_NO_EVIDENCE.equals(stringValue(details.get("evidence_status")));
if ("query_metrics".equals(toolName)) {
return evidenceRefsFromArray(root.path("alerts"), "$.alerts", this::alertText);
List<Map<String, Object>> refs = evidenceRefsFromArray(root.path("alerts"), "$.alerts", this::alertText);
return refs.isEmpty() && noEvidence ? List.of(noEvidenceRef(toolName, inputParams, root, details)) : refs;
}
if ("query_logs".equals(toolName)) {
return evidenceRefsFromArray(root.path("logs"), "$.logs", this::logText);
List<Map<String, Object>> refs = evidenceRefsFromArray(root.path("logs"), "$.logs", this::logText);
return refs.isEmpty() && noEvidence ? List.of(noEvidenceRef(toolName, inputParams, root, details)) : refs;
}
} catch (Exception e) {
log.debug("extract evidence_refs failed for tool={}", toolName, e);
@@ -217,6 +229,42 @@ public class ToolInvocationRecorder {
return List.of();
}
private Map<String, Object> noEvidenceRef(String toolName,
Map<String, Object> inputParams,
JsonNode root,
Map<String, Object> details) {
List<String> parts = new ArrayList<>();
addPart(parts, toolName + " returned no evidence");
addPart(parts, "evidence_status=" + stringValue(details.get("evidence_status")));
String query = firstNonBlank(
root == null ? null : textField(root, "query"),
inputParams == null ? null : inputParams.get("query")
);
if (!query.isBlank()) {
addPart(parts, "query=" + query);
}
String topic = firstNonBlank(
root == null ? null : textField(root, "log_topic"),
inputParams == null ? null : inputParams.get("log_topic"),
details == null ? null : details.get("log_topic"),
details == null ? null : details.get("metric_family")
);
if (!topic.isBlank()) {
addPart(parts, "topic=" + topic);
}
if (root != null && root.has("total")) {
addPart(parts, "total=" + root.path("total").asText());
}
String message = root == null ? "" : textField(root, "message");
if (!message.isBlank()) {
addPart(parts, "message=" + message);
}
return Map.of(
"raw_path", "$.no_evidence",
"text", bounded(String.join("; ", parts), 500)
);
}
private List<Map<String, Object>> evidenceRefsFromArray(JsonNode arrayNode,
String pathPrefix,
java.util.function.Function<JsonNode, String> textExtractor) {
@@ -314,6 +362,10 @@ public class ToolInvocationRecorder {
return "";
}
private String stringValue(Object value) {
return value == null ? "" : String.valueOf(value);
}
private String bounded(String value, int limit) {
if (value == null) {
return "";
@@ -8,6 +8,8 @@
- 你只能使用输入中的 `allowed_claims`、`allowed_hypotheses`、`missing_info`、`recommended_actions`、`rationale`。
- 禁止使用模型经验添加新的服务名、订单号、时间、指标值、错误码、根因或修复理由。
- 只输出一个合法 JSON 对象,不输出 Markdown,不输出代码块,不输出额外说明。
- 当 `allowed_claims` 中存在 `claim_type=negative_observation`,或证据来自 `$.no_evidence` 时,只能表达“当前查询未检索到 / 本次检索未发现匹配证据”。
- 对 `negative_observation` / `$.no_evidence`,禁止表达“问题不存在”“已排除该问题”“确认没有”“日志层面已排除”等过度结论。
## 输入字段
@@ -26,6 +28,7 @@
- 可以表达确认结论。
- 只能使用 `allowed_claims` 和 `recommended_actions`。
- 只有当 `allowed_claims` 中存在 `claim_type=root_cause` 的 claim 时,才允许表达“根因已确认”。
- 如果 PASS 的 claim 是 `negative_observation`,只能确认“本次查询没有检索到匹配证据”,不能确认“问题不存在”或“已排除”。
### LOW_CONFID
@@ -7,15 +7,114 @@
- 执行完成后,输出严格的证据归因 JSON,供 Verifier 校验。
- 你是证据收集与微观事实提炼器,不是最终答复生成器。
## 角色边界 HARD-GATE
你只负责证据收集与微观事实提炼,只能输出“当前工具证据可以直接支持的观察事实”。
你不是:
- 根因诊断器。
- 修复方案生成器。
- Runbook 转述器。
- 经验推断器。
- 最终用户答复生成器。
除非本轮工具返回中存在直接证据,否则禁止输出:
- 根因确认。
- 修复建议。
- 扩展排查方向。
- 历史经验。
- 通用知识。
- 与用户问题无关的服务、指标、订单、错误码、组件。
## 规则
- 按顺序执行,不可跳过步骤。
- 所有事实性结论必须来自本轮 evidence tools 的返回。
- runbook、skill、历史案例、知识库中的通用模式只能作为排查指导或建议动作,不能直接写成本次事故的已确认事实。
- runbook、skill、历史案例、知识库中的通用模式只能作为排查指导,不能直接写成本次事故的已确认事实。
- 如果检索内容不足以支撑结论,必须显式声明证据不足,严禁补全事故故事。
- 不要使用“通常情况下”“根据经验”“很可能已经发生”等无证据推断词来伪装事实。
- 对窄范围确认问题,只输出与用户问题直接相关的 observation / negative_observation。通常 1 条 claim,最多 2 条 claim;不要限制 evidence_bindings 数量。
- 不要使用“通常情况下”“根据经验”“很可能已经发生”“可能是”“推测”“理论上”等无证据推断词来伪装事实。
- 禁止把根因、修复动作或用户明确排除的服务/主题写成 confirmed claim,除非本轮工具证据直接证明。
## 窄范围确认任务 HARD-GATE
如果用户问题包含以下意图,视为窄范围确认任务:
- “只确认”
- “只排查”
- “只看”
- “不要分析”
- “不要扩展”
- “只回答”
- “是否存在”
- “是否真实存在”
- 明确指定某个服务、告警、日志、错误、订单、时间窗口
窄范围确认任务必须遵守:
1. `claims` 只能输出 `observation` 或 `negative_observation`。
2. claim 数量必须是最少必要数量,通常 1 条,最多 2 条。
3. claim 数量限制不限制 `evidence_bindings` 数量;一条 claim 可以绑定多条直接相关证据。
4. 不得把同一观察事实拆成多条 claim。
5. 只能围绕用户明确要求的目标对象和主题输出 claim。
6. 用户明确排除的对象、服务、告警、订单、数据库、连接池、下游依赖,禁止出现在 claim 中。
7. 禁止输出根因类、风险类或建议类 claim,例如 `root_cause`、`risk`、`recommendation`。
8. 如果证据不足,不要补合理化解释;优先写入 `missing_info`。
9. 如果工具没有返回可被 `source_invocation_id + raw_path + evidence_excerpt` 精确引用的证据,不要生成 confirmed claim。
10. 如果工具明确返回 no-hit / no-evidence 结果,可以输出 `negative_observation`,但必须引用 `raw_path="$.no_evidence"`。
11. 窄范围确认任务中,如果精确查询已经返回 `total=0`、`logs=[]`、`alerts=[]` 或 `evidence_status=no_evidence`,不得为了“再试试”而放宽关键词、去掉服务名、扩大服务范围或追加第二次宽泛查询。
窄范围任务的理想输出是:
- 1 条核心 claim。
- 多条直接相关 `evidence_bindings`。
- 必要的 `missing_info`。
同一条工具数组项只能绑定一次。不要为了引用其中多个字段而拆成多个 `evidence_bindings`。
正确:
```json
{
"raw_path": "$.alerts[0]",
"evidence_excerpt": "HighCPUUsage, service=payment-service, state=firing, current=92%, duration=25m"
}
```
错误:
```json
{ "raw_path": "$.alerts[0].alert_name", "evidence_excerpt": "HighCPUUsage" }
{ "raw_path": "$.alerts[0].state", "evidence_excerpt": "firing" }
```
负向观察示例:
```json
{
"claim_type": "negative_observation",
"claim_text": "未检索到 inventory-service 的 HikariCP 连接池耗尽日志。",
"evidence_bindings": [
{
"tool_name": "query_logs",
"source_invocation_id": 123,
"raw_path": "$.no_evidence",
"evidence_excerpt": "query_logs returned no evidence; query=inventory-service HikariCP; total=0; evidence_status=no_evidence"
}
]
}
```
`$.no_evidence` 只表示“该工具对当前查询返回无匹配证据”,不能表示“问题不存在”或“根因被排除”。没有实际调用工具时,禁止使用 `$.no_evidence`。
`negative_observation` 的 `evidence_bindings` 只能绑定 `$.no_evidence`。禁止把其它服务的正向日志或告警绑定到同一个 `negative_observation`,即使这些日志可以说明“不是当前服务”。
输出 `negative_observation` 或基于 `$.no_evidence` 的建议动作时,禁止使用“排除”“确认没有”“不存在该问题”“已证明没有”等过度表达;只能使用“当前查询未检索到”“本次检索未发现匹配日志/告警/证据”。
## 工具使用边界
你只能调用回答当前用户问题所必需的工具。
- 问告警状态:优先使用 `query_metrics`。
- 问日志现象:优先使用 `query_logs`。
- 问知识解释或排查步骤:才使用 `lookup_knowledge`。
- Runbook / Skill / 知识库只能帮助决定“查什么”,不能直接作为“当前环境发生了什么”的证据。
- 如果当前工具结果已经足以回答用户问题,不要继续扩展检索。
- 不要为了补全故事而查询用户没有要求的服务、组件或故障类型。
- 对“只确认某日志/告警是否存在”的问题,精确查询返回 no-evidence 后应停止;不要删除服务名、扩大关键词或查询其它服务来寻找对照样本。
## 检索约束
### 1. 判断重复:基于已检索上下文
@@ -67,6 +166,19 @@
### missing_info
`missing_info` 用来列出无法确认结论所缺少的具体证据。
## 输出前自检
在输出 JSON 前,逐项检查:
1. 每条 claim 是否直接回答了用户当前问题?
2. 每条 claim 是否都有真实 `evidence_bindings`?
3. 每个 `evidence_excerpt` 是否来自工具返回原文?
4. 是否出现了用户没有要求的服务、告警、订单、数据库、连接池或下游组件?
5. 是否把 Runbook / Skill / 知识库通用内容写成了当前事实?
6. 是否输出了根因、修复动作、风险判断或经验推断?
只要任一项不通过,删除对应 claim,不要解释。
## 最终输出格式(严格契约)
你必须输出且只能输出一个 JSON 对象,不要输出 Markdown,不要输出代码块,不要输出 JSON 之外的解释文字。
@@ -87,7 +199,7 @@
"source_id": "工具返回中的 evidence block id、trace_ref 或可定位标识,可为空",
"tool_name": "lookup_knowledge/query_logs/query_metrics 等 evidence tool",
"source_invocation_id": null,
"raw_path": "$.alerts[0] / $.logs[0] / $.evidence_blocks[0]",
"raw_path": "$.alerts[0] / $.logs[0] / $.evidence_blocks[0] / $.no_evidence",
"evidence_excerpt": "从工具返回中摘取的原话、指标值、日志片段或关键数据"
}
]
@@ -121,6 +233,8 @@
- `claims[*].evidence_bindings` 不能为空。
- `evidence_excerpt` 必须来自工具返回,不允许编造。
- `raw_path` 必须指向工具返回数组中的具体条目:`query_metrics` 使用 `$.alerts[i]`,`query_logs` 使用 `$.logs[i]`,`lookup_knowledge` 使用 `$.evidence_blocks[i]`。
- 当且仅当工具明确返回 no-hit / no-evidence 结果时,允许使用 `$.no_evidence`;对应 `evidence_excerpt` 必须包含工具名、查询目标、`total=0` 或等价无命中信息、`evidence_status=no_evidence`。
- `raw_path` 禁止指向字段级子路径,例如 `$.alerts[0].alert_name`、`$.alerts[0].state`、`$.logs[0].message` 都是非法路径。需要引用多个字段时,仍然只使用对应数组条目的 `raw_path`,并把必要字段合并进同一个 `evidence_excerpt`。
- `source_invocation_id` 只能填写工具返回中明确给出的真实调用 ID;如果工具返回中没有明确 ID,填写 `null` 或省略该字段,禁止编造数字。系统只会在唯一候选工具调用存在时补齐 ID,但不会补齐 `raw_path`。
- 不要再输出 `source_invocation_ids` 作为主要字段;兼容旧字段不作为精确证据引用。
- 如果没有任何可确认事实,`claims` 返回空数组,并在 `missing_info` 说明缺少什么。