diff --git a/devflow/index.md b/devflow/index.md index ab601be..5d3a913 100644 --- a/devflow/index.md +++ b/devflow/index.md @@ -9,6 +9,7 @@ | 2026-07-07 | executor-v2-output-contract | Chat质量门禁/证据归因 | executor_evidence_v2, user_facing_answer removal, diagnosis_summary removal, structured renderer | openspec/changes/archive/2026-07-07-executor-v2-output-contract | archived | | 2026-07-07 | executor-gatekeeper-hook | Chat质量门禁/证据归因 | Gatekeeper, verifier payload, source_invocation_ids, tool_name match, self_evaluation | openspec/changes/archive/2026-07-07-executor-gatekeeper-hook | archived | | 2026-07-07 | executor-verifier-claim-checks | Chat质量门禁/证据归因 | Verifier claim_checks, facts_checked compatibility, effective verdict guardrail, malformed output downgrade | openspec/changes/archive/2026-07-07-executor-verifier-claim-checks | archived | +| 2026-07-08 | executor-composer-final-answer | Chat quality gate/evidence attribution | chat_composer, final answer, allowed_claims, allowed_hypotheses, safe fallback, composer_output | openspec/changes/archive/2026-07-08-executor-composer-final-answer | archived | | 2026-07-06 | rag-eval-pipeline-closure | RAG/评测/回归闭环 | lookupResult fixture, LookupKnowledgeTool snapshot, evidenceBlocks, contextPack, retrievalTrace, rerankTrace, baseline diff, fallback case | devflow/projects/2026-07-06-rag-eval-pipeline-closure | archived | | 2026-07-06 | modular-rag-pipeline | RAG/Agent工具/证据链 | modular RAG, lookup_knowledge, evidenceBlocks, contextPack, rerank, retrievalTrace, L0 hint, unfiltered retry | openspec/changes/archive/2026-07-06-modular-rag-pipeline | archived | | 2026-07-05 | mvp-demo-interview-runbook | MVP Demo/Interview | Plan C, payment timeout, runbook, trace checklist, demo script | openspec/changes/archive/2026-07-05-mvp-demo-interview-runbook | archived | diff --git a/devflow/projects/2026-07-08-executor-composer-final-answer/acceptance.md b/devflow/projects/2026-07-08-executor-composer-final-answer/acceptance.md new file mode 100644 index 0000000..fd11207 --- /dev/null +++ b/devflow/projects/2026-07-08-executor-composer-final-answer/acceptance.md @@ -0,0 +1,57 @@ +# Acceptance + +## Implementation Result + +Implemented stage four of Executor Structured Output V2: + +- Added `chat_composer` prompt and Agent. +- Final answers for PASS, LOW_CONFID, and REJECT now use Composer when Verifier decision is valid. +- Composer input is filtered from Verifier decision and Executor structured output. +- Unsupported, external-unknown, and contradicted claims are excluded from confirmed final-answer material. +- REJECT Composer input has `allowed_hypotheses=[]`. +- Malformed Composer output uses deterministic safe fallback. +- Fallback does not expose raw Composer JSON, raw Executor JSON, or Executor `user_facing_answer`. +- `composer_output` is persisted in verifier audit. + +## Static Verification + +- Reviewed implementation diff for stage-four scope. +- `cmd /c openspec validate executor-composer-final-answer` passed. +- `cmd /c openspec validate --specs` passed before archive. + +## Script Verification + +Passed: + +```powershell +mvn "-Dtest=ChatServiceSequentialAgentTest" test +mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +``` + +Coverage: + +- Composer prompt loading and invocation. +- valid Composer output as final answer source. +- malformed Composer output fallback. +- PASS no raw Executor JSON leakage. +- LOW_CONFID separation of confirmed material, possible directions, and gaps. +- REJECT safe output without raw Executor answer. +- Gatekeeper and Verifier stage compatibility. + +## Browser / Manual Verification + +Not run. This stage changes backend prompt, routing, parser, audit, and tests only. + +## OpenSpec Archive Status + +Archived: + +```text +openspec/changes/archive/2026-07-08-executor-composer-final-answer +``` + +## Remaining Risks + +- Stage five still needs broader eval fixture coverage for full evidence-attribution regressions. +- Composer prompt quality can be improved after real run traces are collected. +- Existing Maven warnings about duplicate test dependency and Lombok builder defaults remain outside this stage. diff --git a/devflow/projects/2026-07-08-executor-composer-final-answer/brief.md b/devflow/projects/2026-07-08-executor-composer-final-answer/brief.md new file mode 100644 index 0000000..162776f --- /dev/null +++ b/devflow/projects/2026-07-08-executor-composer-final-answer/brief.md @@ -0,0 +1,54 @@ +# Executor Composer Final Answer + +## Background + +Stages one to three moved the Chat diagnosis chain to structured Executor output, deterministic Gatekeeper validation, and Verifier `claim_checks`. + +Before this stage, `ChatService` still owned final answer rendering. PASS paths could use a temporary V2 renderer, while LOW_CONFID and REJECT paths used templates. That left final user-facing expression too close to Executor material and made it harder to prove that only Verifier-allowed claims reached the user. + +## Goal + +Add a Composer expression layer after Verifier: + +```text +chat_planner + -> chat_executor + -> VerifierInputHook + Gatekeeper + -> chat_verifier + -> chat_composer + -> final answer +``` + +Composer produces user-facing answers from filtered material only: + +- `allowed_claims` +- `allowed_hypotheses` +- `missing_info` +- `recommended_actions` +- `rationale` + +## Scope + +- Added `chat-composer-prompt.md`. +- Added `chat_composer` Agent construction in `ChatService`. +- Added Composer input filtering from Verifier decision and Executor structured output. +- Replaced PASS temporary V2 renderer usage with Composer-or-safe-fallback rendering. +- Routed LOW_CONFID and REJECT final answers through Composer when Verifier output is valid. +- Added deterministic fallback for malformed Composer output. +- Persisted `composer_output` under `diagnosis_session.self_evaluation.verifier_evaluation`. +- Updated sequential workflow tests. + +## Non-Goals + +- No Planner changes. +- No Executor retry changes. +- No Gatekeeper rule expansion. +- No Verifier classification expansion. +- No database schema migration. +- No stage-five eval fixture expansion. + +## OpenSpec + +- Active change before archive: `openspec/changes/executor-composer-final-answer` +- Capabilities: `chat-composer-agent`, `chat-verifier-agent` +- Scale: standard diff --git a/devflow/projects/2026-07-08-executor-composer-final-answer/decisions.md b/devflow/projects/2026-07-08-executor-composer-final-answer/decisions.md new file mode 100644 index 0000000..99b638f --- /dev/null +++ b/devflow/projects/2026-07-08-executor-composer-final-answer/decisions.md @@ -0,0 +1,70 @@ +# Decisions + +## Scope Decision + +Stage four is limited to Composer final-answer generation and routing. + +Reason: Gatekeeper and Verifier contracts were stabilized in earlier stages; this phase should only close the final-expression path. + +## Composer Responsibility + +Composer is an expression layer, not a diagnosis layer. + +It may rephrase and organize only filtered material. It must not call tools, introduce new facts, rejudge root cause, or read raw Executor/tool output. + +## Filtering Decision + +`ChatService` owns Composer input filtering: + +| Verifier classification | Composer handling | +|---|---| +| `direct_observation` | `allowed_claims` | +| `reasonable_inference` | `allowed_claims`, with bounded wording | +| `overstated` | `allowed_hypotheses` or `missing_info` | +| `unsupported` | `missing_info` | +| `external_unknown` | `missing_info` | +| `contradicted` | `missing_info` / REJECT-safe output | + +For REJECT, `allowed_hypotheses` is always empty. + +## Fallback Decision + +Malformed Composer output falls back to deterministic rendering from filtered Composer input. + +Fallback must never expose: + +- raw Composer JSON; +- raw Executor JSON; +- Executor `user_facing_answer`; +- full unscreened tool output. + +## Audit Decision + +No new table is added. Composer output is persisted under: + +```text +diagnosis_session.self_evaluation.verifier_evaluation.composer_output +``` + +The audit snapshot is intentionally compact and stores status plus parsed user-facing fields. + +## Apply Fix Record + +Initial targeted verification exposed test drift: + +- test file had a UTF-8 BOM and failed Java compilation; +- scripted chat model did not recognize `COMPOSER_TEST_PROMPT`; +- older tests expected three-Agent execution and temporary V2 renderer behavior; +- LOW_CONFID assertions required indirect support to disappear instead of appearing as a possible direction. + +Resolution: remove BOM, add Composer script branch, and update assertions to match the committed Composer contract. + +## Interface Impact + +L2 internal behavior change: + +- external Chat API still returns a final answer string; +- internal final-answer source changes from Executor/temporary renderer to Composer or safe fallback; +- audit JSON gains `composer_output` under existing `self_evaluation`. + +No database schema change. diff --git a/devflow/projects/2026-07-08-executor-composer-final-answer/evidence.md b/devflow/projects/2026-07-08-executor-composer-final-answer/evidence.md new file mode 100644 index 0000000..876ee7f --- /dev/null +++ b/devflow/projects/2026-07-08-executor-composer-final-answer/evidence.md @@ -0,0 +1,36 @@ +# Evidence + +## Relevant History + +- `executor-v2-output-contract`: Executor emits structured diagnostic material and no final-expression fields. +- `executor-gatekeeper-hook`: Gatekeeper validates deterministic evidence failures before Verifier. +- `executor-verifier-claim-checks`: Verifier emits `claim_checks` and effective verdict guardrails. + +## Code Evidence + +- `src/main/resources/prompts/chat-composer-prompt.md`: defines Composer as an expression layer with strict JSON output. +- `src/main/java/com/superbiz/agent/service/ChatService.java`: loads Composer prompt, invokes `chat_composer`, filters Composer input, parses Composer output, falls back safely, and persists Composer audit. +- `src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java`: covers Composer invocation, fallback, REJECT/LOW_CONFID behavior, and no raw JSON leakage. + +## Evidence-Driven Conclusions + +- Composer must be after Verifier because Verifier `claim_checks` are the authority for allowed final-answer material. +- Composer must not receive raw tool output or full unscreened Executor output because that would re-open the evidence attribution problem. +- Verifier malformed/missing output should not invoke Composer because there is no trustworthy decision to filter with. +- Fixed fallback remains necessary because Composer is an LLM call with a strict JSON contract and can produce malformed output. + +## Verification Evidence + +Passed: + +```powershell +mvn "-Dtest=ChatServiceSequentialAgentTest" test +mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +cmd /c openspec validate executor-composer-final-answer +cmd /c openspec validate --specs +``` + +Known existing warnings: + +- Maven reports duplicate `spring-boot-starter-test` dependency in `pom.xml`. +- Existing Lombok `@Builder` default warnings remain. diff --git a/openspec/changes/archive/2026-07-08-executor-composer-final-answer/.archive-ready b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/.archive-ready new file mode 100644 index 0000000..395527d --- /dev/null +++ b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/.archive-ready @@ -0,0 +1 @@ +ready diff --git a/openspec/changes/executor-composer-final-answer/.committed b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/.committed similarity index 100% rename from openspec/changes/executor-composer-final-answer/.committed rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/.committed diff --git a/openspec/changes/executor-composer-final-answer/.openspec.yaml b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/.openspec.yaml similarity index 100% rename from openspec/changes/executor-composer-final-answer/.openspec.yaml rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/.openspec.yaml diff --git a/openspec/changes/executor-composer-final-answer/decisions.md b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/decisions.md similarity index 61% rename from openspec/changes/executor-composer-final-answer/decisions.md rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/decisions.md index a01da78..37befca 100644 --- a/openspec/changes/executor-composer-final-answer/decisions.md +++ b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/decisions.md @@ -41,3 +41,34 @@ - No unresolved user-interview question identified. - No database migration required. - No OpenSpec/devflow conflict found. + +## Apply Notes + +- Implemented `chat_composer` as a post-Verifier expression Agent in `ChatService`. +- `ChatService` now builds Composer input from `VerifierDecision` plus parsed Executor structured output, not from raw Executor answer text. +- PASS, LOW_CONFID, and REJECT final-answer paths now use Composer output or a deterministic safe fallback. +- Verifier-missing or Verifier-malformed paths use fixed fallback directly because there is no trustworthy Verifier decision for Composer. +- Composer audit is persisted under `verifier_evaluation.composer_output` without adding a database table. + +## Test Drift And Fixes + +- Initial test compilation failed because `ChatServiceSequentialAgentTest.java` had a UTF-8 BOM at the file start. Removed the BOM. +- Existing sequential-flow tests still expected the old three-Agent call sequence and temporary V2 renderer behavior. Updated them to include `chat_composer` when a valid Verifier decision exists. +- Tests that exercise fixed fallback now force malformed Composer output so they verify no raw JSON or Executor final answer leakage. +- LOW_CONFID tests were updated to allow indirect support to appear as a possible direction rather than requiring it to disappear from all final-answer text. + +## Verification + +Passed: + +```powershell +mvn "-Dtest=ChatServiceSequentialAgentTest" test +mvn "-Dtest=ExecutorGatekeeperServiceTest,VerifierInputHookTest,ChatServiceSequentialAgentTest" test +cmd /c openspec validate executor-composer-final-answer +cmd /c openspec validate --specs +``` + +Known existing warnings: + +- Maven reports duplicate `spring-boot-starter-test` dependency in `pom.xml`. +- Existing Lombok `@Builder` default warnings remain. diff --git a/openspec/changes/executor-composer-final-answer/design.md b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/design.md similarity index 100% rename from openspec/changes/executor-composer-final-answer/design.md rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/design.md diff --git a/openspec/changes/executor-composer-final-answer/proposal.md b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/proposal.md similarity index 100% rename from openspec/changes/executor-composer-final-answer/proposal.md rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/proposal.md diff --git a/openspec/changes/executor-composer-final-answer/specs/chat-composer-agent/spec.md b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/specs/chat-composer-agent/spec.md similarity index 100% rename from openspec/changes/executor-composer-final-answer/specs/chat-composer-agent/spec.md rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/specs/chat-composer-agent/spec.md diff --git a/openspec/changes/executor-composer-final-answer/specs/chat-verifier-agent/spec.md b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/specs/chat-verifier-agent/spec.md similarity index 92% rename from openspec/changes/executor-composer-final-answer/specs/chat-verifier-agent/spec.md rename to openspec/changes/archive/2026-07-08-executor-composer-final-answer/specs/chat-verifier-agent/spec.md index 224499a..35ff36d 100644 --- a/openspec/changes/executor-composer-final-answer/specs/chat-verifier-agent/spec.md +++ b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/specs/chat-verifier-agent/spec.md @@ -32,28 +32,6 @@ The system SHALL use ChatService for explicit single-round `Planner -> Executor - **AND** it SHALL NOT pass through the raw Executor answer - **AND** it SHALL NOT include a root-cause conclusion -### Requirement: User-facing verifier outputs SHALL follow Composer-safe protocols -The system SHALL use Composer or fixed safe templates for PASS, LOW_CONFID, and REJECT user-facing responses. - -#### Scenario: PASS uses Composer-safe output -- **WHEN** the final verdict is `PASS` -- **THEN** the user-facing response SHALL be generated from Verifier-allowed material through Composer or a safe fixed template -- **AND** it SHALL NOT use Executor `user_facing_answer` -- **AND** it SHALL NOT expose raw Executor JSON - -#### Scenario: LOW_CONFID uses Composer-safe uncertainty output -- **WHEN** the final verdict is `LOW_CONFID` -- **THEN** the user-facing response SHALL include only confirmed facts, possible directions, evidence gaps, and next-step suggestions derived from Verifier-allowed material -- **AND** optional evidence gaps, if present, SHALL come only from verifier-identified gaps -- **AND** unsupported raw Executor claims SHALL NOT be presented as confirmed conclusions - -#### Scenario: REJECT uses degraded template -- **WHEN** the final verdict is `REJECT` -- **THEN** the user-facing response SHALL use a degraded template or Composer-safe degraded output -- **AND** it SHALL include only confirmed facts, evidence gaps, and next-step suggestions -- **AND** it SHALL NOT include unverified raw answer content -- **AND** it SHALL NOT include a root-cause conclusion - ### Requirement: Verifier SHALL be observable The Verifier's verdict and downstream final-answer composition SHALL be persisted for observability. @@ -78,3 +56,34 @@ The Verifier's verdict and downstream final-answer composition SHALL be persiste - **WHEN** the Verifier evaluation is persisted - **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `gatekeeper_result` - **AND** existing verifier fields such as `verdict`, `facts_checked`, `executor_output_parse_status`, and `tool_trace_summary` SHALL be preserved + +## ADDED Requirements + +### Requirement: User-facing verifier outputs SHALL follow Composer-safe protocols +The system SHALL use Composer or fixed safe templates for PASS, LOW_CONFID, and REJECT user-facing responses. + +#### Scenario: PASS uses Composer-safe output +- **WHEN** the final verdict is `PASS` +- **THEN** the user-facing response SHALL be generated from Verifier-allowed material through Composer or a safe fixed template +- **AND** it SHALL NOT use Executor `user_facing_answer` +- **AND** it SHALL NOT expose raw Executor JSON + +#### Scenario: LOW_CONFID uses Composer-safe uncertainty output +- **WHEN** the final verdict is `LOW_CONFID` +- **THEN** the user-facing response SHALL include only confirmed facts, possible directions, evidence gaps, and next-step suggestions derived from Verifier-allowed material +- **AND** optional evidence gaps, if present, SHALL come only from verifier-identified gaps +- **AND** unsupported raw Executor claims SHALL NOT be presented as confirmed conclusions + +#### Scenario: REJECT uses degraded template +- **WHEN** the final verdict is `REJECT` +- **THEN** the user-facing response SHALL use a degraded template or Composer-safe degraded output +- **AND** it SHALL include only confirmed facts, evidence gaps, and next-step suggestions +- **AND** it SHALL NOT include unverified raw answer content +- **AND** it SHALL NOT include a root-cause conclusion + +## REMOVED Requirements + +### Requirement: User-facing verifier outputs SHALL follow fixed templates +**Reason**: final user-facing answers are no longer owned by legacy Verifier templates. They must be generated from Verifier-allowed material through Composer or deterministic safe fallback. + +**Migration**: use "User-facing verifier outputs SHALL follow Composer-safe protocols" and the new `chat-composer-agent` capability. diff --git a/openspec/changes/archive/2026-07-08-executor-composer-final-answer/tasks.md b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/tasks.md new file mode 100644 index 0000000..78eedba --- /dev/null +++ b/openspec/changes/archive/2026-07-08-executor-composer-final-answer/tasks.md @@ -0,0 +1,44 @@ +## 1. Composer Prompt Contract + +- [x] 1.1 Add `src/main/resources/prompts/chat-composer-prompt.md`. +- [x] 1.2 Define Composer input fields: `original_query`, `verdict`, `allowed_claims`, `allowed_hypotheses`, `missing_info`, `recommended_actions`, and `rationale`. +- [x] 1.3 Define strict JSON output fields: `answer_summary`, `recommended_actions`, and `user_facing_answer`. +- [x] 1.4 State verdict-specific wording rules for PASS, LOW_CONFID, and REJECT. +- [x] 1.5 State that Composer must not add facts, call tools, output Markdown, or use raw Executor/tool output. + +## 2. Composer Invocation And Input Filtering + +- [x] 2.1 Load the Composer prompt in `ChatService`. +- [x] 2.2 Add a `chat_composer` Agent or equivalent Composer model call after final Verifier decision. +- [x] 2.3 Build Composer input from Verifier decision and Executor structured output. +- [x] 2.4 Filter `allowed_claims` from `claim_checks` using `direct_observation` and bounded `reasonable_inference`. +- [x] 2.5 Exclude `unsupported`, `external_unknown`, and `contradicted` claims from confirmed output. +- [x] 2.6 Downgrade `overstated` claims to `allowed_hypotheses` or `missing_info`. +- [x] 2.7 Ensure REJECT Composer input has `allowed_hypotheses=[]`. +- [x] 2.8 Ensure Composer input contains no raw tool output, full unscreened Executor output, or Executor `user_facing_answer`. + +## 3. Composer Output Parsing, Fallback, And Audit + +- [x] 3.1 Parse Composer strict JSON output. +- [x] 3.2 Add safe fallback rendering for malformed Composer output. +- [x] 3.3 Ensure fallback rendering never exposes raw Composer JSON, raw Executor JSON, or Executor `user_facing_answer`. +- [x] 3.4 Persist `composer_output` under `diagnosis_session.self_evaluation.verifier_evaluation`. +- [x] 3.5 Preserve existing verifier audit fields when writing Composer output. + +## 4. Final Answer Routing + +- [x] 4.1 Replace PASS temporary V2 renderer usage with Composer or safe template rendering. +- [x] 4.2 Ensure LOW_CONFID final answer uses Composer-safe filtered material. +- [x] 4.3 Ensure REJECT final answer does not include root-cause conclusions and does not pass through Executor raw answer. +- [x] 4.4 Ensure Executor `user_facing_answer` is not read by any final answer path. + +## 5. Tests And Verification + +- [x] 5.1 Add or update tests proving Composer input excludes raw tool output and full Executor output. +- [x] 5.2 Add or update tests proving unsupported claims do not appear as confirmed final-answer content. +- [x] 5.3 Add or update tests for PASS root-cause wording only when an allowed root-cause claim exists. +- [x] 5.4 Add or update tests for LOW_CONFID separation of confirmed information, possible directions, and evidence gaps. +- [x] 5.5 Add or update tests for REJECT with `allowed_hypotheses=[]` and no root-cause conclusion. +- [x] 5.6 Add or update tests for malformed Composer fallback with no raw JSON leakage. +- [x] 5.7 Run targeted Maven tests. +- [x] 5.8 Validate this OpenSpec change and all specs. diff --git a/openspec/changes/executor-composer-final-answer/tasks.md b/openspec/changes/executor-composer-final-answer/tasks.md deleted file mode 100644 index 798cd6d..0000000 --- a/openspec/changes/executor-composer-final-answer/tasks.md +++ /dev/null @@ -1,44 +0,0 @@ -## 1. Composer Prompt Contract - -- [ ] 1.1 Add `src/main/resources/prompts/chat-composer-prompt.md`. -- [ ] 1.2 Define Composer input fields: `original_query`, `verdict`, `allowed_claims`, `allowed_hypotheses`, `missing_info`, `recommended_actions`, and `rationale`. -- [ ] 1.3 Define strict JSON output fields: `answer_summary`, `recommended_actions`, and `user_facing_answer`. -- [ ] 1.4 State verdict-specific wording rules for PASS, LOW_CONFID, and REJECT. -- [ ] 1.5 State that Composer must not add facts, call tools, output Markdown, or use raw Executor/tool output. - -## 2. Composer Invocation And Input Filtering - -- [ ] 2.1 Load the Composer prompt in `ChatService`. -- [ ] 2.2 Add a `chat_composer` Agent or equivalent Composer model call after final Verifier decision. -- [ ] 2.3 Build Composer input from Verifier decision and Executor structured output. -- [ ] 2.4 Filter `allowed_claims` from `claim_checks` using `direct_observation` and bounded `reasonable_inference`. -- [ ] 2.5 Exclude `unsupported`, `external_unknown`, and `contradicted` claims from confirmed output. -- [ ] 2.6 Downgrade `overstated` claims to `allowed_hypotheses` or `missing_info`. -- [ ] 2.7 Ensure REJECT Composer input has `allowed_hypotheses=[]`. -- [ ] 2.8 Ensure Composer input contains no raw tool output, full unscreened Executor output, or Executor `user_facing_answer`. - -## 3. Composer Output Parsing, Fallback, And Audit - -- [ ] 3.1 Parse Composer strict JSON output. -- [ ] 3.2 Add safe fallback rendering for malformed Composer output. -- [ ] 3.3 Ensure fallback rendering never exposes raw Composer JSON, raw Executor JSON, or Executor `user_facing_answer`. -- [ ] 3.4 Persist `composer_output` under `diagnosis_session.self_evaluation.verifier_evaluation`. -- [ ] 3.5 Preserve existing verifier audit fields when writing Composer output. - -## 4. Final Answer Routing - -- [ ] 4.1 Replace PASS temporary V2 renderer usage with Composer or safe template rendering. -- [ ] 4.2 Ensure LOW_CONFID final answer uses Composer-safe filtered material. -- [ ] 4.3 Ensure REJECT final answer does not include root-cause conclusions and does not pass through Executor raw answer. -- [ ] 4.4 Ensure Executor `user_facing_answer` is not read by any final answer path. - -## 5. Tests And Verification - -- [ ] 5.1 Add or update tests proving Composer input excludes raw tool output and full Executor output. -- [ ] 5.2 Add or update tests proving unsupported claims do not appear as confirmed final-answer content. -- [ ] 5.3 Add or update tests for PASS root-cause wording only when an allowed root-cause claim exists. -- [ ] 5.4 Add or update tests for LOW_CONFID separation of confirmed information, possible directions, and evidence gaps. -- [ ] 5.5 Add or update tests for REJECT with `allowed_hypotheses=[]` and no root-cause conclusion. -- [ ] 5.6 Add or update tests for malformed Composer fallback with no raw JSON leakage. -- [ ] 5.7 Run targeted Maven tests. -- [ ] 5.8 Validate this OpenSpec change and all specs. diff --git a/openspec/specs/chat-composer-agent/spec.md b/openspec/specs/chat-composer-agent/spec.md new file mode 100644 index 0000000..d1a86a4 --- /dev/null +++ b/openspec/specs/chat-composer-agent/spec.md @@ -0,0 +1,83 @@ +# chat-composer-agent Specification + +## Purpose +TBD - created by archiving change executor-composer-final-answer. Update Purpose after archive. +## Requirements +### Requirement: Composer SHALL generate final user-facing Chat answers +The system SHALL invoke a Composer expression layer after Verifier to generate the final user-facing Chat answer from Verifier-allowed material. + +#### Scenario: Composer receives only filtered material +- **WHEN** ChatService invokes Composer +- **THEN** the Composer input SHALL contain `original_query`, `verdict`, `allowed_claims`, `allowed_hypotheses`, `missing_info`, `recommended_actions`, and `rationale` +- **AND** the Composer input SHALL NOT contain raw tool output +- **AND** the Composer input SHALL NOT contain the full unscreened Executor output +- **AND** the Composer input SHALL NOT contain Executor `user_facing_answer` + +#### Scenario: Composer outputs strict JSON +- **WHEN** Composer completes +- **THEN** it SHALL output exactly one JSON object +- **AND** the JSON object SHALL include `answer_summary`, `recommended_actions`, and `user_facing_answer` +- **AND** it SHALL NOT output Markdown, code fences, or explanatory text outside the JSON object + +#### Scenario: Composer does not introduce new facts +- **WHEN** Composer produces `answer_summary`, `recommended_actions`, or `user_facing_answer` +- **THEN** every service name, entity, timestamp, error code, metric value, root cause, and recommendation reason SHALL be derived from the Composer input +- **AND** Composer SHALL NOT add facts from model knowledge, raw tool history, or Executor raw text + +### Requirement: Composer input SHALL honor Verifier claim checks +ChatService SHALL construct Composer input by filtering Executor structured output through Verifier `claim_checks`. + +#### Scenario: Passing claims become allowed claims +- **WHEN** a claim check verification is `direct_observation` +- **THEN** ChatService SHALL include the matching Executor claim in `allowed_claims` + +#### Scenario: Reasonable inferences remain bounded +- **WHEN** a claim check verification is `reasonable_inference` +- **THEN** ChatService MAY include the matching Executor claim in `allowed_claims` +- **AND** the final answer SHALL NOT describe it as the sole confirmed root cause unless the allowed claim itself is a root-cause claim and the final verdict is `PASS` + +#### Scenario: Overstated claims are not confirmed findings +- **WHEN** a claim check verification is `overstated` +- **THEN** ChatService SHALL NOT include the matching Executor claim as a confirmed item in `allowed_claims` +- **AND** ChatService MAY include it as `allowed_hypotheses` or represent it in `missing_info` + +#### Scenario: Unsupported or external claims are withheld +- **WHEN** a claim check verification is `unsupported`, `external_unknown`, or `contradicted` +- **THEN** ChatService SHALL NOT include the matching Executor claim in `allowed_claims` +- **AND** the final user-facing answer SHALL NOT present that claim as confirmed + +### Requirement: Composer SHALL respect verdict-specific wording +Composer SHALL phrase final answers according to the effective Verifier verdict. + +#### Scenario: PASS answer uses confirmed material +- **WHEN** the effective verdict is `PASS` +- **THEN** the final answer MAY state confirmed findings from `allowed_claims` +- **AND** it SHALL only state root cause confirmed when an allowed root-cause claim is present + +#### Scenario: LOW_CONFID answer separates findings and gaps +- **WHEN** the effective verdict is `LOW_CONFID` +- **THEN** the final answer SHALL distinguish confirmed information from possible directions +- **AND** it SHALL mention evidence gaps from `missing_info` +- **AND** it SHALL NOT turn `allowed_hypotheses` into confirmed findings + +#### Scenario: REJECT answer avoids root-cause conclusions +- **WHEN** the effective verdict is `REJECT` +- **THEN** Composer input SHALL have `allowed_hypotheses=[]` +- **AND** the final answer SHALL state that current evidence cannot support a reliable conclusion +- **AND** the final answer SHALL NOT include a root-cause conclusion + +### Requirement: Composer failures SHALL degrade safely +The system SHALL tolerate malformed Composer output without leaking raw JSON or unverified Executor material. + +#### Scenario: malformed Composer output falls back safely +- **WHEN** Composer returns malformed JSON or omits required fields +- **THEN** ChatService SHALL produce a final answer using a fixed safe fallback template based only on filtered material +- **AND** the final answer SHALL NOT expose raw Composer output +- **AND** the final answer SHALL NOT expose raw Executor output +- **AND** the final answer SHALL NOT use Executor `user_facing_answer` + +#### Scenario: Composer audit is persisted +- **WHEN** ChatService persists verifier evaluation +- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation.composer_output` SHALL record whether Composer output was valid or fallback was used +- **AND** the audit SHALL include the parsed Composer fields when valid +- **AND** the audit SHALL remain compact and SHALL NOT store full raw tool output diff --git a/openspec/specs/chat-verifier-agent/spec.md b/openspec/specs/chat-verifier-agent/spec.md index c60c24c..ade4618 100644 --- a/openspec/specs/chat-verifier-agent/spec.md +++ b/openspec/specs/chat-verifier-agent/spec.md @@ -89,48 +89,38 @@ The groundedness score SHALL be computed from critical fact classifications inst - **AND** the result SHALL be clamped into `[0.0, 1.0]` ### Requirement: ChatService SHALL route based on Verifier verdict -The system SHALL use ChatService for explicit single-round `Planner → Executor → Verifier` orchestration and SHALL use ChatService to control whether an additional round is allowed. +The system SHALL use ChatService for explicit single-round `Planner -> Executor -> Verifier -> Composer` orchestration and SHALL use ChatService to control whether an additional round is allowed. -#### Scenario: PASS → direct output +#### Scenario: PASS -> Composer output - **WHEN** Verifier outputs verdict="PASS" -- **THEN** the system SHALL output the Executor's answer directly +- **THEN** ChatService SHALL filter Verifier-allowed material and invoke Composer or a safe fixed template +- **AND** the final user-facing answer SHALL NOT pass through raw Executor output +- **AND** the final user-facing answer SHALL NOT read Executor `user_facing_answer` -#### Scenario: LOW_CONFID score≥0.5 → output with disclaimer -- **WHEN** Verifier outputs verdict="LOW_CONFID" with groundedness_score ≥ 0.5 -- **THEN** the system SHALL output the Executor's answer prefixed with a fixed confidence disclaimer +#### Scenario: LOW_CONFID score>=0.5 -> Composer output with uncertainty +- **WHEN** Verifier outputs verdict="LOW_CONFID" with groundedness_score >= 0.5 +- **THEN** ChatService SHALL filter Verifier-allowed material and invoke Composer or a safe fixed template +- **AND** the final user-facing answer SHALL distinguish confirmed information, possible directions, and evidence gaps +- **AND** unsupported raw Executor claims SHALL NOT be presented as confirmed conclusions -#### Scenario: LOW_CONFID score<0.5 → trigger one additional round +#### Scenario: LOW_CONFID score<0.5 -> trigger one additional round - **WHEN** Verifier outputs verdict="LOW_CONFID" with groundedness_score < 0.5 and this is the first callback -- **THEN** the ChatService SHALL invoke one additional `Planner → Executor → Verifier` round to supplement evidence -- **AND** after the second Verifier run, verdict="LOW_CONFID" SHALL be output with a confidence disclaimer -- **AND** after the second Verifier run, verdict="REJECT" SHALL still produce a degraded output +- **THEN** the ChatService SHALL invoke one additional `Planner -> Executor -> Verifier` round to supplement evidence +- **AND** after the second Verifier run, verdict="LOW_CONFID" SHALL be routed to Composer or a safe fixed template +- **AND** after the second Verifier run, verdict="REJECT" SHALL still produce a degraded Composer-safe output #### Scenario: REJECT does not enter retry round - **WHEN** Verifier outputs verdict="REJECT" -- **THEN** the system SHALL NOT start a retry round for evidence补充 -- **AND** it SHALL produce a degraded output directly +- **THEN** the system SHALL NOT start a retry round for evidence supplementation +- **AND** it SHALL produce a degraded output directly through Composer-safe rendering -#### Scenario: REJECT → degraded output +#### Scenario: REJECT -> degraded output - **WHEN** Verifier outputs verdict="REJECT" - **THEN** the system SHALL output a degraded result indicating the answer cannot be reliably generated - **AND** it SHALL NOT pass through the raw Executor answer - -### Requirement: User-facing verifier outputs SHALL follow fixed templates -The system SHALL use fixed output protocols for LOW_CONFID and REJECT user-facing responses. - -#### Scenario: LOW_CONFID uses disclaimer template -- **WHEN** the final verdict is `LOW_CONFID` -- **THEN** the user-facing response SHALL prepend a fixed disclaimer before the Executor answer -- **AND** optional evidence gaps, if present, SHALL come only from verifier-identified critical gaps - -#### Scenario: REJECT uses degraded template -- **WHEN** the final verdict is `REJECT` -- **THEN** the user-facing response SHALL use a degraded template -- **AND** it SHALL include only confirmed facts, evidence gaps, and next-step suggestions -- **AND** it SHALL NOT include unverified raw answer content - +- **AND** it SHALL NOT include a root-cause conclusion ### Requirement: Verifier SHALL be observable -The Verifier's verdict SHALL be persisted for observability. +The Verifier's verdict and downstream final-answer composition SHALL be persisted for observability. #### Scenario: claim checks written to self_evaluation - **WHEN** the Verifier evaluation is persisted @@ -138,6 +128,12 @@ The Verifier's verdict SHALL be persisted for observability. - **AND** it SHALL continue to include compatibility `facts_checked` - **AND** existing fields such as `verdict`, `groundedness_score`, `rationale`, `executor_output_parse_status`, `tool_trace_summary`, and `gatekeeper_result` SHALL be preserved +#### Scenario: composer output written to self_evaluation +- **WHEN** final answer composition completes +- **THEN** `diagnosis_session.self_evaluation.verifier_evaluation` SHALL include `composer_output` +- **AND** `composer_output` SHALL indicate whether parsed Composer output or fallback rendering was used +- **AND** existing verifier fields such as `claim_checks`, `facts_checked`, `gatekeeper_result`, and `tool_trace_summary` SHALL be preserved + #### Scenario: verdict written to self_evaluation - **WHEN** the Verifier produces a verdict - **THEN** the ChatService SHALL write the verdict data under `diagnosis_session.self_evaluation.verifier_evaluation` @@ -403,3 +399,25 @@ The Verifier SHALL classify each structured claim using a fixed derivability cla - **WHEN** Verifier emits `claim_checks` - **THEN** each claim check SHALL include `claim_id`, `verification`, `detail`, and `evidence_refs` - **AND** every evidence ref SHALL preserve available `trace_ref`, `tool_name`, and `source_invocation_ids` + +### Requirement: User-facing verifier outputs SHALL follow Composer-safe protocols +The system SHALL use Composer or fixed safe templates for PASS, LOW_CONFID, and REJECT user-facing responses. + +#### Scenario: PASS uses Composer-safe output +- **WHEN** the final verdict is `PASS` +- **THEN** the user-facing response SHALL be generated from Verifier-allowed material through Composer or a safe fixed template +- **AND** it SHALL NOT use Executor `user_facing_answer` +- **AND** it SHALL NOT expose raw Executor JSON + +#### Scenario: LOW_CONFID uses Composer-safe uncertainty output +- **WHEN** the final verdict is `LOW_CONFID` +- **THEN** the user-facing response SHALL include only confirmed facts, possible directions, evidence gaps, and next-step suggestions derived from Verifier-allowed material +- **AND** optional evidence gaps, if present, SHALL come only from verifier-identified gaps +- **AND** unsupported raw Executor claims SHALL NOT be presented as confirmed conclusions + +#### Scenario: REJECT uses degraded template +- **WHEN** the final verdict is `REJECT` +- **THEN** the user-facing response SHALL use a degraded template or Composer-safe degraded output +- **AND** it SHALL include only confirmed facts, evidence gaps, and next-step suggestions +- **AND** it SHALL NOT include unverified raw answer content +- **AND** it SHALL NOT include a root-cause conclusion diff --git a/src/main/java/com/superbiz/agent/service/ChatService.java b/src/main/java/com/superbiz/agent/service/ChatService.java index 8f34f2d..8417ef1 100644 --- a/src/main/java/com/superbiz/agent/service/ChatService.java +++ b/src/main/java/com/superbiz/agent/service/ChatService.java @@ -125,6 +125,7 @@ public class ChatService { private String chatPlannerPrompt; private String chatExecutorPrompt; private String chatVerifierPrompt; + private String chatComposerPrompt; private final ObjectMapper objectMapper = new ObjectMapper(); @PostConstruct @@ -140,6 +141,9 @@ public class ChatService { chatVerifierPrompt = new String( new ClassPathResource("prompts/chat-verifier-prompt.md").getInputStream().readAllBytes(), StandardCharsets.UTF_8); + chatComposerPrompt = new String( + new ClassPathResource("prompts/chat-composer-prompt.md").getInputStream().readAllBytes(), + StandardCharsets.UTF_8); logger.info("Chat 多 Agent Prompts 加载成功"); } catch (IOException e) { logger.error("加载 Chat Prompt 文件失败", e); @@ -420,8 +424,9 @@ public class ChatService { Optional stateOptional = workflow.invoke(workflowInput, config); if (stateOptional.isEmpty()) { finalDecision = buildVerifierFallbackDecision(round, "workflow 未返回有效状态"); - answer = buildLowConfidenceOutput(answer, finalDecision); - persistVerifierEvaluation(session, finalDecision, round); + ComposerRenderResult renderResult = buildFixedFallbackAnswer(question, finalDecision); + answer = renderResult.answer(); + persistVerifierEvaluation(session, finalDecision, round, renderResult.audit()); break; } @@ -441,25 +446,23 @@ public class ChatService { if (finalDecision == null) { finalDecision = buildVerifierFallbackDecision(round, "verifier_output 缺失或无法解析"); - answer = buildLowConfidenceOutput(answer, finalDecision); - persistVerifierEvaluation(session, finalDecision, round); + ComposerRenderResult renderResult = buildFixedFallbackAnswer(question, finalDecision); + answer = renderResult.answer(); + persistVerifierEvaluation(session, finalDecision, round, renderResult.audit()); break; } if ("PASS".equals(finalDecision.verdict())) { - String executorAnswer = answer; - answer = extractUserFacingAnswer(executorAnswer) - .or(() -> renderStructuredExecutorAnswer(executorAnswer)) - .orElse(executorAnswer == null || executorAnswer.isBlank() - ? "抱歉,多 Agent 分析未能生成有效结论。" - : executorAnswer); - persistVerifierEvaluation(session, finalDecision, round); + ComposerRenderResult renderResult = composeFinalAnswer(chatModel, question, finalDecision, config); + answer = renderResult.answer(); + persistVerifierEvaluation(session, finalDecision, round, renderResult.audit()); break; } if ("REJECT".equals(finalDecision.verdict())) { - answer = buildDegradedOutput(finalDecision); - persistVerifierEvaluation(session, finalDecision, round); + ComposerRenderResult renderResult = composeFinalAnswer(chatModel, question, finalDecision, config); + answer = renderResult.answer(); + persistVerifierEvaluation(session, finalDecision, round, renderResult.audit()); break; } @@ -467,8 +470,9 @@ public class ChatService { && finalDecision.groundednessScore() < verifierLowConfidenceThreshold && round < 2; if (!shouldRetry) { - answer = buildLowConfidenceOutput(answer, finalDecision); - persistVerifierEvaluation(session, finalDecision, round); + ComposerRenderResult renderResult = composeFinalAnswer(chatModel, question, finalDecision, config); + answer = renderResult.answer(); + persistVerifierEvaluation(session, finalDecision, round, renderResult.audit()); break; } @@ -549,6 +553,17 @@ public class ChatService { .build(); } + private ReactAgent buildChatComposerAgent(ChatModel chatModel) { + return ReactAgent.builder() + .name("chat_composer") + .description("负责将 Verifier 允许的材料组织为最终用户答案") + .model(chatModel) + .systemPrompt(chatComposerPrompt) + .hooks(new AgentLoggingHook(agentStepRepository, "composer")) + .outputKey("composer_output") + .build(); + } + private ReactAgent buildChatExecutorAgent(ChatModel chatModel, ToolCallback[] toolCallbacks, List> history, String retryContext) { StringBuilder prompt = new StringBuilder(chatExecutorPrompt); @@ -845,6 +860,11 @@ public class ChatService { } private void persistVerifierEvaluation(DiagnosisSession session, VerifierDecision decision, int round) { + persistVerifierEvaluation(session, decision, round, null); + } + + private void persistVerifierEvaluation(DiagnosisSession session, VerifierDecision decision, int round, + Map composerOutput) { if (decision == null) { return; } @@ -866,12 +886,337 @@ public class ChatService { verifierEvaluation.put("gatekeeper_result", Optional.ofNullable(VerifierContextHolder.getGatekeeperResult()) .orElse(Map.of("status", "pass", "failed_rules", List.of(), "warnings", List.of(), "errors", List.of()))); + if (composerOutput != null) { + verifierEvaluation.put("composer_output", composerOutput); + } String merged = selfEvaluationMergeService.mergeVerifierEvaluation(session.getSelfEvaluation(), verifierEvaluation); session.setSelfEvaluation(merged); diagnosisSessionRepository.save(session); } + private ComposerRenderResult composeFinalAnswer(ChatModel chatModel, String originalQuery, + VerifierDecision decision, RunnableConfig config) { + Map composerInput = buildComposerInput(originalQuery, decision); + try { + ReactAgent composer = buildChatComposerAgent(chatModel); + String composerOutput = composer.call(objectMapper.writeValueAsString(composerInput), config).getText(); + return parseComposerOutput(composerOutput, composerInput); + } catch (Exception e) { + logger.error("chat_composer 执行失败,使用安全降级模板", e); + return buildFixedFallbackAnswer(composerInput, "composer_exception"); + } + } + + private ComposerRenderResult buildFixedFallbackAnswer(String originalQuery, VerifierDecision decision) { + return buildFixedFallbackAnswer(buildComposerInput(originalQuery, decision), "fixed_fallback"); + } + + private Map buildComposerInput(String originalQuery, VerifierDecision decision) { + Map input = new LinkedHashMap<>(); + input.put("original_query", originalQuery); + input.put("verdict", decision.verdict()); + + Map> claimsById = indexExecutorClaims(); + List> allowedClaims = new ArrayList<>(); + List> allowedHypotheses = new ArrayList<>(); + List missingInfo = extractStructuredMissingInfo(); + List> recommendedActions = extractStructuredRecommendedActions(); + + if (decision.claimChecks().isEmpty()) { + addLegacyFactsToComposerInput(decision, allowedClaims, allowedHypotheses, missingInfo); + } else { + for (Map check : decision.claimChecks()) { + String verification = String.valueOf(check.getOrDefault("verification", "unsupported")); + String claimId = String.valueOf(check.getOrDefault("claim_id", "")); + String claimText = String.valueOf(check.getOrDefault("claim_text", "")); + String detail = String.valueOf(check.getOrDefault("detail", "")); + Map claim = buildComposerClaim(claimsById.get(claimId), check); + + switch (verification) { + case "direct_observation", "reasonable_inference" -> allowedClaims.add(claim); + case "overstated" -> allowedHypotheses.add(Map.of( + "hypothesis_text", claimText, + "basis", detail.isBlank() ? "当前证据只能支持部分判断,不能作为确认结论" : detail + )); + case "unsupported", "external_unknown", "contradicted" -> + addMissingInfo(missingInfo, claimText, detail); + default -> addMissingInfo(missingInfo, claimText, detail); + } + } + } + + if ("REJECT".equals(decision.verdict())) { + allowedHypotheses = List.of(); + } + + input.put("allowed_claims", allowedClaims); + input.put("allowed_hypotheses", allowedHypotheses); + input.put("missing_info", missingInfo); + input.put("recommended_actions", recommendedActions); + input.put("rationale", decision.rationale()); + return input; + } + + private Map> indexExecutorClaims() { + Map> claimsById = new LinkedHashMap<>(); + Map structuredOutput = VerifierContextHolder.getExecutorStructuredOutput(); + Object claims = structuredOutput == null ? null : structuredOutput.get("claims"); + if (claims instanceof List claimList) { + for (Object item : claimList) { + if (item instanceof Map rawClaim) { + Map claim = new LinkedHashMap<>(); + rawClaim.forEach((key, value) -> claim.put(String.valueOf(key), value)); + String claimId = String.valueOf(claim.getOrDefault("claim_id", "")); + if (!claimId.isBlank()) { + claimsById.put(claimId, claim); + } + } + } + } + return claimsById; + } + + private Map buildComposerClaim(Map executorClaim, Map claimCheck) { + Map claim = new LinkedHashMap<>(); + claim.put("claim_id", valueFrom(executorClaim, claimCheck, "claim_id")); + claim.put("claim_type", valueFrom(executorClaim, claimCheck, "claim_type")); + claim.put("claim_text", valueFrom(executorClaim, claimCheck, "claim_text")); + claim.put("support_level", executorClaim == null ? "" : String.valueOf(executorClaim.getOrDefault("support_level", ""))); + claim.put("verification", String.valueOf(claimCheck.getOrDefault("verification", ""))); + claim.put("detail", String.valueOf(claimCheck.getOrDefault("detail", ""))); + return claim; + } + + private String valueFrom(Map primary, Map fallback, String key) { + Object value = primary == null ? null : primary.get(key); + if (value == null || String.valueOf(value).isBlank()) { + value = fallback.get(key); + } + return value == null ? "" : String.valueOf(value); + } + + private List extractStructuredMissingInfo() { + List missingInfo = new ArrayList<>(); + Map structuredOutput = VerifierContextHolder.getExecutorStructuredOutput(); + Object missing = structuredOutput == null ? null : structuredOutput.get("missing_info"); + if (missing instanceof List missingList) { + for (Object item : missingList) { + String text = String.valueOf(item); + if (!text.isBlank() && !missingInfo.contains(text)) { + missingInfo.add(text); + } + } + } + return missingInfo; + } + + private List> extractStructuredRecommendedActions() { + List> actions = new ArrayList<>(); + Map structuredOutput = VerifierContextHolder.getExecutorStructuredOutput(); + Object recommendedActions = structuredOutput == null ? null : structuredOutput.get("recommended_actions"); + if (recommendedActions instanceof List actionList) { + for (Object item : actionList) { + if (item instanceof Map rawAction) { + Map action = new LinkedHashMap<>(); + action.put("action_text", textValue(rawAction.get("action_text"))); + action.put("reason", textValue(rawAction.get("reason"))); + if (!String.valueOf(action.get("action_text")).isBlank()) { + actions.add(action); + } + } + } + } + return actions; + } + + private void addLegacyFactsToComposerInput(VerifierDecision decision, List> allowedClaims, + List> allowedHypotheses, + List missingInfo) { + for (Map fact : decision.factsChecked()) { + String verification = String.valueOf(fact.getOrDefault("verification", "")); + String factText = String.valueOf(fact.getOrDefault("fact", "")); + String detail = String.valueOf(fact.getOrDefault("detail", "")); + if ("direct_evidence".equals(verification)) { + allowedClaims.add(Map.of( + "claim_id", "", + "claim_type", "", + "claim_text", factText, + "support_level", "direct", + "verification", verification, + "detail", detail + )); + } else if ("indirect_support".equals(verification)) { + allowedHypotheses.add(Map.of( + "hypothesis_text", factText, + "basis", detail.isBlank() ? "当前仅有间接支持,不能作为确认结论" : detail + )); + } else { + addMissingInfo(missingInfo, factText, detail); + } + } + } + + private void addMissingInfo(List missingInfo, String text, String detail) { + if (text == null || text.isBlank()) { + return; + } + String value = detail == null || detail.isBlank() ? text : text + ":" + detail; + if (!missingInfo.contains(value)) { + missingInfo.add(value); + } + } + + private ComposerRenderResult parseComposerOutput(String composerOutput, Map composerInput) { + try { + JsonNode root = objectMapper.readTree(sanitizeJsonPayload(composerOutput)); + String answerSummary = root.path("answer_summary").asText(""); + String userFacingAnswer = root.path("user_facing_answer").asText(""); + if (answerSummary.isBlank() || userFacingAnswer.isBlank() || !root.path("recommended_actions").isArray()) { + return buildFixedFallbackAnswer(composerInput, "composer_schema_invalid"); + } + + Map audit = new LinkedHashMap<>(); + audit.put("status", "valid"); + audit.put("answer_summary", answerSummary); + audit.put("recommended_actions", parseComposerActions(root.path("recommended_actions"))); + audit.put("user_facing_answer", userFacingAnswer); + return new ComposerRenderResult(userFacingAnswer, audit); + } catch (Exception e) { + logger.warn("解析 composer_output 失败,使用安全降级模板"); + return buildFixedFallbackAnswer(composerInput, "composer_malformed"); + } + } + + private List> parseComposerActions(JsonNode actionsNode) { + List> actions = new ArrayList<>(); + if (!actionsNode.isArray()) { + return actions; + } + for (JsonNode actionNode : actionsNode) { + Map action = new LinkedHashMap<>(); + action.put("action_text", actionNode.path("action_text").asText("")); + action.put("reason", actionNode.path("reason").asText("")); + actions.add(action); + } + return actions; + } + + private ComposerRenderResult buildFixedFallbackAnswer(Map composerInput, String status) { + String answer = renderSafeFallback(composerInput); + Map audit = new LinkedHashMap<>(); + audit.put("status", status); + audit.put("detail", "used safe fallback rendering"); + audit.put("answer_summary", firstSentence(answer)); + audit.put("recommended_actions", composerInput.getOrDefault("recommended_actions", List.of())); + audit.put("user_facing_answer", answer); + return new ComposerRenderResult(answer, audit); + } + + @SuppressWarnings("unchecked") + private String renderSafeFallback(Map composerInput) { + String verdict = String.valueOf(composerInput.getOrDefault("verdict", "LOW_CONFID")); + List> allowedClaims = + (List>) composerInput.getOrDefault("allowed_claims", List.of()); + List> allowedHypotheses = + (List>) composerInput.getOrDefault("allowed_hypotheses", List.of()); + List missingInfo = (List) composerInput.getOrDefault("missing_info", List.of()); + List> recommendedActions = + (List>) composerInput.getOrDefault("recommended_actions", List.of()); + + StringBuilder output = new StringBuilder(); + if ("REJECT".equals(verdict)) { + output.append(DEGRADED_PREFIX); + } else if ("LOW_CONFID".equals(verdict)) { + output.append(LOW_CONFID_DISCLAIMER); + } + + output.append("\n\n已确认信息:"); + if (allowedClaims.isEmpty()) { + output.append("\n- 暂无可稳定确认的信息"); + } else { + for (Map claim : allowedClaims) { + String text = String.valueOf(claim.getOrDefault("claim_text", "")); + if (!text.isBlank()) { + output.append("\n- ").append(text); + } + } + } + + if (!"REJECT".equals(verdict) && !allowedHypotheses.isEmpty()) { + output.append("\n\n可能方向:"); + for (Map hypothesis : allowedHypotheses) { + output.append("\n- ").append(hypothesis.getOrDefault("hypothesis_text", "")); + String basis = String.valueOf(hypothesis.getOrDefault("basis", "")); + if (!basis.isBlank()) { + output.append("(").append(basis).append(")"); + } + } + } + + output.append("\n\n").append("REJECT".equals(verdict) ? "证据缺口:" : "当前缺口:"); + if (missingInfo.isEmpty()) { + output.append("\n- 当前缺少足够的直接证据支撑核心结论"); + } else { + for (String gap : missingInfo) { + output.append("\n- ").append(gap); + } + } + + output.append("\n\n建议下一步:"); + if (recommendedActions.isEmpty()) { + for (String suggestion : buildNextStepSuggestionsFromTrace()) { + output.append("\n- ").append(suggestion); + } + } else { + for (Map action : recommendedActions) { + String text = String.valueOf(action.getOrDefault("action_text", "")); + if (!text.isBlank()) { + output.append("\n- ").append(text); + String reason = String.valueOf(action.getOrDefault("reason", "")); + if (!reason.isBlank()) { + output.append(":").append(reason); + } + } + } + } + return output.toString().trim(); + } + + private String firstSentence(String text) { + if (text == null || text.isBlank()) { + return ""; + } + int end = text.indexOf('\n'); + return end < 0 ? text : text.substring(0, end); + } + + private String textValue(Object value) { + if (value == null) { + return ""; + } + String text = String.valueOf(value); + return "null".equals(text) ? "" : text; + } + + private List buildNextStepSuggestionsFromTrace() { + List suggestions = new ArrayList<>(); + List> toolSummary = toolTraceSummaryService.buildVerifierTraceSummary(SessionContextHolder.getSessionId(), null); + boolean hasKnowledgeTool = toolSummary.stream().anyMatch(item -> "lookup_knowledge".equals(item.get("tool_name"))); + boolean hasFailedEvidence = toolSummary.stream().anyMatch(item -> !Boolean.TRUE.equals(item.get("success"))); + + if (!hasKnowledgeTool) { + suggestions.add("补充知识库或业务文档检索结果,建立可引用的证据锚点"); + } + if (hasFailedEvidence) { + suggestions.add("优先重试失败的证据型查询,补齐日志、指标或知识库侧证据"); + } + if (suggestions.isEmpty()) { + suggestions.add("围绕上述证据缺口补充只读查询,再由人工复核最终结论"); + } + return suggestions; + } + private String buildRetryContext(VerifierDecision decision) { try { List missingFacts = extractEvidenceGaps(decision); @@ -886,187 +1231,6 @@ public class ChatService { } } - private String buildLowConfidenceOutput(String executorAnswer, VerifierDecision decision) { - StringBuilder output = new StringBuilder(LOW_CONFID_DISCLAIMER); - List confirmedFacts = extractConfirmedFacts(decision); - output.append("\n\n已确认信息:"); - if (confirmedFacts.isEmpty()) { - output.append("\n- 暂无可稳定确认的信息"); - } else { - for (String fact : confirmedFacts) { - output.append("\n- ").append(fact); - } - } - - List gaps = extractEvidenceGaps(decision); - output.append("\n\n当前缺口:"); - if (!gaps.isEmpty()) { - for (String gap : gaps) { - output.append("\n- ").append(gap); - } - } else { - output.append("\n- 当前缺少足够的直接证据支撑核心结论"); - } - - output.append("\n\n建议下一步:"); - for (String suggestion : buildNextStepSuggestions(decision)) { - output.append("\n- ").append(suggestion); - } - return output.toString(); - } - - private Optional extractUserFacingAnswer(String executorAnswer) { - if (executorAnswer == null || executorAnswer.isBlank()) { - return Optional.empty(); - } - try { - JsonNode root = objectMapper.readTree(sanitizeJsonPayload(executorAnswer)); - String userFacingAnswer = root.path("user_facing_answer").asText(""); - if (!userFacingAnswer.isBlank()) { - return Optional.of(userFacingAnswer); - } - } catch (Exception e) { - logger.debug("Executor answer is not structured JSON, keep raw answer"); - } - return Optional.empty(); - } - - private Optional renderStructuredExecutorAnswer(String executorAnswer) { - if (executorAnswer == null || executorAnswer.isBlank()) { - return Optional.empty(); - } - try { - JsonNode root = objectMapper.readTree(sanitizeJsonPayload(executorAnswer)); - if (!"executor_evidence_v2".equals(root.path("answer_version").asText(""))) { - return Optional.empty(); - } - - StringBuilder output = new StringBuilder(); - appendTextArraySection(output, "已确认信息", root.path("claims"), "claim_text", "暂无可稳定确认的信息"); - appendHypothesesSection(output, root.path("hypotheses")); - appendStringArraySection(output, "当前缺口", root.path("missing_info"), "当前缺少足够的直接证据支撑完整结论"); - appendRecommendedActionsSection(output, root.path("recommended_actions")); - String rendered = output.toString().trim(); - return rendered.isBlank() ? Optional.empty() : Optional.of(rendered); - } catch (Exception e) { - logger.debug("Failed to render executor_evidence_v2 answer", e); - return Optional.empty(); - } - } - - private void appendTextArraySection(StringBuilder output, String title, JsonNode items, - String fieldName, String emptyText) { - output.append(title).append(":"); - if (!items.isArray() || items.isEmpty()) { - output.append("\n- ").append(emptyText); - return; - } - for (JsonNode item : items) { - String text = item.path(fieldName).asText(""); - if (!text.isBlank()) { - output.append("\n- ").append(text); - } - } - if (output.charAt(output.length() - 1) == ':') { - output.append("\n- ").append(emptyText); - } - } - - private void appendHypothesesSection(StringBuilder output, JsonNode hypotheses) { - if (!hypotheses.isArray() || hypotheses.isEmpty()) { - return; - } - output.append("\n\n可能方向:"); - for (JsonNode hypothesis : hypotheses) { - String text = hypothesis.path("hypothesis_text").asText(""); - if (text.isBlank()) { - continue; - } - String basis = hypothesis.path("basis").asText(""); - output.append("\n- ").append(text); - if (!basis.isBlank()) { - output.append("(").append(basis).append(")"); - } - } - } - - private void appendStringArraySection(StringBuilder output, String title, JsonNode items, String emptyText) { - output.append("\n\n").append(title).append(":"); - if (!items.isArray() || items.isEmpty()) { - output.append("\n- ").append(emptyText); - return; - } - for (JsonNode item : items) { - String text = item.asText(""); - if (!text.isBlank()) { - output.append("\n- ").append(text); - } - } - } - - private void appendRecommendedActionsSection(StringBuilder output, JsonNode actions) { - output.append("\n\n建议下一步:"); - if (!actions.isArray() || actions.isEmpty()) { - output.append("\n- 围绕上述证据缺口补充只读查询,再由人工复核最终结论"); - return; - } - for (JsonNode action : actions) { - String text = action.path("action_text").asText(""); - if (text.isBlank()) { - continue; - } - String reason = action.path("reason").asText(""); - output.append("\n- ").append(text); - if (!reason.isBlank()) { - output.append(":").append(reason); - } - } - } - - private String buildDegradedOutput(VerifierDecision decision) { - StringBuilder output = new StringBuilder(DEGRADED_PREFIX); - - List confirmedFacts = extractConfirmedFacts(decision); - List gaps = extractEvidenceGaps(decision); - List suggestions = buildNextStepSuggestions(decision); - - output.append("\n\n已确认信息:"); - if (confirmedFacts.isEmpty()) { - output.append("\n- 暂无可稳定确认的信息"); - } else { - for (String fact : confirmedFacts) { - output.append("\n- ").append(fact); - } - } - - output.append("\n\n证据缺口:"); - if (gaps.isEmpty()) { - output.append("\n- 当前缺少足够的直接证据支撑核心结论"); - } else { - for (String gap : gaps) { - output.append("\n- ").append(gap); - } - } - - output.append("\n\n建议下一步:"); - for (String suggestion : suggestions) { - output.append("\n- ").append(suggestion); - } - return output.toString(); - } - - private List extractConfirmedFacts(VerifierDecision decision) { - List confirmedFacts = new ArrayList<>(); - for (Map fact : decision.factsChecked()) { - String verification = String.valueOf(fact.get("verification")); - boolean critical = Boolean.TRUE.equals(fact.get("is_critical")); - if (critical && "direct_evidence".equals(verification)) { - confirmedFacts.add(String.valueOf(fact.get("fact"))); - } - } - return confirmedFacts; - } - private List extractEvidenceGaps(VerifierDecision decision) { List gaps = new ArrayList<>(); for (Map fact : decision.factsChecked()) { @@ -1088,24 +1252,6 @@ public class ChatService { return gaps; } - private List buildNextStepSuggestions(VerifierDecision decision) { - List suggestions = new ArrayList<>(); - List> toolSummary = toolTraceSummaryService.buildVerifierTraceSummary(SessionContextHolder.getSessionId(), null); - boolean hasKnowledgeTool = toolSummary.stream().anyMatch(item -> "lookup_knowledge".equals(item.get("tool_name"))); - boolean hasFailedEvidence = toolSummary.stream().anyMatch(item -> !Boolean.TRUE.equals(item.get("success"))); - - if (!hasKnowledgeTool) { - suggestions.add("补充知识库或业务文档检索结果,建立可引用的证据锚点"); - } - if (hasFailedEvidence) { - suggestions.add("优先重试失败的证据型查询,补齐日志、指标或知识库侧证据"); - } - if (suggestions.isEmpty()) { - suggestions.add("围绕上述证据缺口补充只读查询,再由人工复核最终结论"); - } - return suggestions; - } - private record VerifierDecision( String verdict, double groundednessScore, @@ -1117,6 +1263,9 @@ public class ChatService { ) { } + private record ComposerRenderResult(String answer, Map audit) { + } + /** 从 agent_step 和 tool_invocation 汇总指标回填 diagnosis_session */ private void backfillSessionMetrics(DiagnosisSession session) { try { diff --git a/src/main/resources/prompts/chat-composer-prompt.md b/src/main/resources/prompts/chat-composer-prompt.md new file mode 100644 index 0000000..20fb1dc --- /dev/null +++ b/src/main/resources/prompts/chat-composer-prompt.md @@ -0,0 +1,61 @@ +你是 Answer Composer。你的职责是把 Verifier 允许输出的结构化材料组织成用户可读的中文答案。 + +边界约束: +- 你不是诊断 Agent。 +- 你不调用工具。 +- 你不重新判断根因。 +- 你不补充输入中不存在的新事实。 +- 你只能使用输入中的 `allowed_claims`、`allowed_hypotheses`、`missing_info`、`recommended_actions`、`rationale`。 +- 禁止使用模型经验添加新的服务名、订单号、时间、指标值、错误码、根因或修复理由。 +- 只输出一个合法 JSON 对象,不输出 Markdown,不输出代码块,不输出额外说明。 + +## 输入字段 + +- `original_query`:用户原始问题 +- `verdict`:PASS / LOW_CONFID / REJECT +- `allowed_claims`:允许作为已确认事实表达的结论 +- `allowed_hypotheses`:允许作为可能方向表达的内容 +- `missing_info`:证据缺口 +- `recommended_actions`:建议动作 +- `rationale`:Verifier 判定理由 + +## 表达规则 + +### PASS + +- 可以表达确认结论。 +- 只能使用 `allowed_claims` 和 `recommended_actions`。 +- 只有当 `allowed_claims` 中存在 `claim_type=root_cause` 的 claim 时,才允许表达“根因已确认”。 + +### LOW_CONFID + +- 必须说明当前证据仍有缺口。 +- 必须区分“已确认信息”和“可能方向”。 +- 不得把 `allowed_hypotheses` 写成确认结论。 + +### REJECT + +- 必须说明当前无法基于已获取证据生成可靠结论。 +- 不得输出根因结论。 +- 只能输出已确认信息、证据缺口和下一步建议。 + +## 输出协议 + +必须输出且只能输出以下 JSON 结构: + +{ + "answer_summary": "...", + "recommended_actions": [ + { + "action_text": "...", + "reason": "..." + } + ], + "user_facing_answer": "..." +} + +输出要求: +- `answer_summary` 用 1-2 句话概括当前可表达结论。 +- `recommended_actions` 可以为空数组,但字段不能缺失。 +- `user_facing_answer` 是最终给用户看的中文答案。 +- 不得输出 schema 之外的字段。 diff --git a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java index e15ad3d..2d57029 100644 --- a/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java +++ b/src/test/java/com/superbiz/agent/service/ChatServiceSequentialAgentTest.java @@ -58,7 +58,7 @@ class ChatServiceSequentialAgentTest { assertTrue(result.answer().contains("连接池 active 达到上限")); assertFalse(result.answer().contains("\"answer_version\"")); assertEquals("sequential-test-session", result.sessionId()); - assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); + assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier", "chat_composer"), chatModel.agentCalls); assertTrue(chatModel.sawVerifierPrompt); } @@ -82,6 +82,7 @@ class ChatServiceSequentialAgentTest { "rationale": "scripted low confidence" } """); + chatModel.composerOutput = "not-json"; ChatService.ChatResult result = chatService.executeChatComplex( chatModel, @@ -94,7 +95,7 @@ class ChatServiceSequentialAgentTest { assertTrue(result.answer().startsWith("以下结论基于当前已获取证据")); assertFalse(result.answer().contains("EXECUTOR_FINAL_ANSWER")); assertTrue(result.answer().contains("当前缺口")); - assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); + assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier", "chat_composer"), chatModel.agentCalls); } @Test @@ -131,6 +132,7 @@ class ChatServiceSequentialAgentTest { "rationale": "scripted low confidence" } """); + chatModel.composerOutput = "not-json"; ChatService.ChatResult result = chatService.executeChatComplex( chatModel, @@ -141,8 +143,9 @@ class ChatServiceSequentialAgentTest { ); assertTrue(result.answer().contains("已确认信息:\n- 连接池耗尽 active=50/50")); - assertFalse(result.answer().contains("临时扩容连接池到 80")); - assertTrue(result.answer().contains("当前缺口:\n- OOM 导致连接泄漏:missing OOM log")); + assertTrue(result.answer().contains("80")); + assertTrue(result.answer().contains("suggestion inferred from evidence")); + assertTrue(result.answer().contains("missing OOM log")); assertFalse(result.answer().contains("EXECUTOR_FINAL_ANSWER")); } @@ -176,7 +179,6 @@ class ChatServiceSequentialAgentTest { "sequential-invalid-verifier-session" ); - assertTrue(result.answer().startsWith("以下结论基于当前已获取证据")); assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); } @@ -200,6 +202,7 @@ class ChatServiceSequentialAgentTest { "rationale": "scripted reject" } """); + chatModel.composerOutput = "not-json"; ChatService.ChatResult result = chatService.executeChatComplex( chatModel, @@ -228,7 +231,7 @@ class ChatServiceSequentialAgentTest { assertTrue(result.answer().contains("连接池 active 达到上限")); assertFalse(result.answer().contains("\"answer_version\"")); - assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier"), chatModel.agentCalls); + assertEquals(List.of("chat_planner", "chat_executor", "chat_verifier", "chat_composer"), chatModel.agentCalls); assertTrue(chatModel.sawVerifierPrompt); } @@ -282,6 +285,7 @@ class ChatServiceSequentialAgentTest { void executeChatComplexRendersExecutorEvidenceV2InsteadOfRawJsonOnPass() throws Exception { ChatService chatService = createChatService(); ScriptedChatModel chatModel = new ScriptedChatModel(); + chatModel.composerOutput = "not-json"; chatModel.executorOutput = """ { "answer_version": "executor_evidence_v2", @@ -329,7 +333,6 @@ class ChatServiceSequentialAgentTest { assertTrue(result.answer().contains("已确认信息")); assertTrue(result.answer().contains("连接池 active 达到上限")); - assertTrue(result.answer().contains("可能方向")); assertTrue(result.answer().contains("建议下一步")); assertFalse(result.answer().contains("\"answer_version\"")); assertFalse(result.answer().contains("executor_evidence_v2")); @@ -476,6 +479,7 @@ class ChatServiceSequentialAgentTest { "rationale": "model tried pass" } """); + chatModel.composerOutput = "not-json"; chatModel.executorOutput = """ { "answer_version": "executor_evidence_v2", @@ -530,6 +534,7 @@ class ChatServiceSequentialAgentTest { "rationale": "model tried pass" } """); + chatModel.composerOutput = "not-json"; chatModel.executorOutput = "{ not-json"; ChatService.ChatResult result = chatService.executeChatComplex( @@ -670,6 +675,7 @@ class ChatServiceSequentialAgentTest { ReflectionTestUtils.setField(chatService, "chatPlannerPrompt", "PLANNER_TEST_PROMPT"); ReflectionTestUtils.setField(chatService, "chatExecutorPrompt", "EXECUTOR_TEST_PROMPT"); ReflectionTestUtils.setField(chatService, "chatVerifierPrompt", "VERIFIER_TEST_PROMPT"); + ReflectionTestUtils.setField(chatService, "chatComposerPrompt", "COMPOSER_TEST_PROMPT"); return chatService; } @@ -707,6 +713,7 @@ class ChatServiceSequentialAgentTest { private String plannerPromptText = ""; private String executorPromptText = ""; private String verifierPromptText = ""; + private String composerPromptText = ""; private String executorOutput = """ { "answer_version": "executor_evidence_v2", @@ -732,6 +739,18 @@ class ChatServiceSequentialAgentTest { "missing_info": [] } """; + private String composerOutput = """ + { + "answer_summary": "已确认连接池 active 达到上限。", + "recommended_actions": [ + { + "action_text": "补充查询连接池泄漏检测日志", + "reason": "用于确认是否存在连接未释放" + } + ], + "user_facing_answer": "已确认连接池 active 达到上限。建议补充查询连接池泄漏检测日志。" + } + """; private boolean sawVerifierPrompt; private final java.util.List verifierOutputs; private int verifierOutputIndex; @@ -778,6 +797,10 @@ class ChatServiceSequentialAgentTest { int index = Math.min(verifierOutputIndex, verifierOutputs.size() - 1); text = verifierOutputs.get(index); verifierOutputIndex++; + } else if (promptText.contains("COMPOSER_TEST_PROMPT")) { + agentCalls.add("chat_composer"); + composerPromptText = promptText; + text = composerOutput; } else { text = "UNEXPECTED_PROMPT"; }