diff --git a/openspec/changes/archive/2026-07-06-trace-ui-workbench/design.md b/openspec/changes/archive/2026-07-06-trace-ui-workbench/design.md new file mode 100644 index 0000000..2abe517 --- /dev/null +++ b/openspec/changes/archive/2026-07-06-trace-ui-workbench/design.md @@ -0,0 +1,70 @@ +# Trace UI workbench design + +## Subject and audience + +Subject: a single-session diagnosis trace audit workbench. + +Audience: backend and AIOps engineers reviewing an MVP diagnosis run after a +demo, incident drill, or regression check. + +Single job: turn one trace response into an inspectable ledger that exposes +agent flow, evidence, verifier judgement, and skill boundaries without reading +raw JSON first. + +## Frontend approach + +- Static `trace.html`, `trace.css`, and `trace.js`. +- No dependency on a package manager, bundler, external fonts, or remote icons. +- Fetches `/api/diagnosis/{sessionId}/trace` and renders client-side. +- Accepts `?sessionId=...` and keeps the loaded session id in the URL. + +## Visual direction + +Palette: + +- `#111827` ink rail for the trace frame. +- `#f7f3ea` warm ledger surface. +- `#1f7a6b` evidence green. +- `#b5472f` rejection red. +- `#c58a19` warning amber. +- `#5b6472` operational gray. + +Type: + +- UI/body: `Inter, ui-sans-serif, system-ui`. +- Data and labels: `ui-monospace, SFMono-Regular, Consolas`. + +Layout: + +```text ++--------------------------------------------------------------------+ +| Session id input | Load | status / verdict / counts strip | ++------------------+----------------------+--------------------------+ +| Agent trace rail | Evidence tool ledger | Inspector | +| ordered steps | filters + calls | verifier + RAG + skill | ++------------------+----------------------+--------------------------+ +``` + +Signature element: a trace rail that treats each agent step as a ledger entry +with ordered markers, duration, token count, and expandable raw model excerpts. + +## Data mapping + +- `data.session`: summary strip, final answer, self-evaluation source. +- `data.steps`: ordered trace rail. +- `data.toolInvocations`: evidence ledger and selected tool inspector. +- `data.session.selfEvaluation.verifier_evaluation`: verdict, groundedness, + facts table, and verifier `tool_trace_summary`. +- `data.toolInvocations[*].retrievalDetails`: RAG query transform, retrieval + trace, context pack, rerank trace, and evidence blocks. +- `steps[*].modelOutput`: best-effort extraction of `selected_skill`. +- `steps[*].modelInput/modelOutput` and tool names: best-effort skill boundary + checks for `read_skill`. + +## Accessibility and states + +- Keyboard-focusable controls. +- Loading, missing session id, API error, empty steps, empty tools, and missing + verifier states. +- Responsive three-pane desktop layout that stacks on narrow screens. +- Respects `prefers-reduced-motion`. diff --git a/openspec/changes/archive/2026-07-06-trace-ui-workbench/proposal.md b/openspec/changes/archive/2026-07-06-trace-ui-workbench/proposal.md new file mode 100644 index 0000000..4976916 --- /dev/null +++ b/openspec/changes/archive/2026-07-06-trace-ui-workbench/proposal.md @@ -0,0 +1,37 @@ +# Trace UI workbench + +## Why + +The MVP already persists diagnosis trace data and exposes it through +`GET /api/diagnosis/{sessionId}/trace`, but reviewers still need to inspect raw +JSON to answer basic audit questions: + +- Which agents ran, in what order, and with what persisted counts? +- Which evidence tools executed, how many times, and did they succeed? +- Which facts did the Verifier check, and which evidence refs support them? +- Did Planner only select skill metadata while Executor handled skill loading? +- What RAG retrieval details were used by `lookup_knowledge`? + +This slows down demo review and makes trace quality issues harder to spot. + +## What changes + +- Add a static Trace workbench page served by Spring Boot static resources. +- Load a diagnosis trace by session id through the existing read-only Trace API. +- Render session summary, agent timeline, evidence tool ledger, verifier facts, + skill boundary checks, and RAG retrieval details. +- Add a navigation entry from the existing chat page to the Trace workbench. + +## Non-goals + +- No new backend endpoint. +- No mutation of diagnosis sessions, agent steps, tool invocations, or feedback. +- No new frontend framework or build pipeline. +- No change to the persisted trace schema. + +## Impact + +- Frontend-only runtime surface under `src/main/resources/static`. +- Uses the existing `Result` API response contract. +- Works with existing trace records, including sessions that lack verifier or RAG + detail fields. diff --git a/openspec/changes/archive/2026-07-06-trace-ui-workbench/specs/mvp-demo-trace-acceptance/spec.md b/openspec/changes/archive/2026-07-06-trace-ui-workbench/specs/mvp-demo-trace-acceptance/spec.md new file mode 100644 index 0000000..a3bd19f --- /dev/null +++ b/openspec/changes/archive/2026-07-06-trace-ui-workbench/specs/mvp-demo-trace-acceptance/spec.md @@ -0,0 +1,36 @@ +## ADDED Requirements + +### Requirement: MVP demo SHALL provide a browser trace workbench +The MVP demo SHALL provide a browser-accessible static page for inspecting one +diagnosis trace by session id using the existing read-only Trace API. + +#### Scenario: Existing trace renders in the workbench +- **WHEN** a reviewer opens the Trace workbench with a session id that exists +- **THEN** the page SHALL request `GET /api/diagnosis/{sessionId}/trace` +- **AND** it SHALL render session summary, ordered agent steps, ordered tool + invocations, verifier evaluation, and final answer when present + +#### Scenario: Trace workbench handles missing or failed traces +- **WHEN** the Trace API returns an error or the session id is empty +- **THEN** the page SHALL show a clear error or empty state without mutating any + diagnosis data + +### Requirement: Trace workbench SHALL expose evidence and skill boundaries +The Trace workbench SHALL make tool evidence and skill-loading boundaries visible +without requiring raw JSON inspection first. + +#### Scenario: Tool and verifier evidence are inspectable +- **WHEN** a trace includes tool invocations and verifier facts +- **THEN** the page SHALL show tool invocation counts, success status, tool + filters, verifier facts, and evidence references + +#### Scenario: RAG details are inspectable for lookup knowledge calls +- **WHEN** a `lookup_knowledge` invocation includes retrieval details +- **THEN** the page SHALL show query transform, retrieval trace, context pack, + rerank trace, and evidence block data where available + +#### Scenario: Skill boundary checks are visible +- **WHEN** a trace includes planner, executor, or verifier steps +- **THEN** the page SHALL show best-effort indicators for selected skill, + planner `read_skill` text mentions, executor `read_skill` text mentions, and + verifier `read_skill` text mentions diff --git a/openspec/changes/archive/2026-07-06-trace-ui-workbench/tasks.md b/openspec/changes/archive/2026-07-06-trace-ui-workbench/tasks.md new file mode 100644 index 0000000..37d15c8 --- /dev/null +++ b/openspec/changes/archive/2026-07-06-trace-ui-workbench/tasks.md @@ -0,0 +1,9 @@ +# Tasks + +- [x] 1. Add OpenSpec delta for the Trace UI workbench. +- [x] 2. Add static Trace workbench HTML/CSS/JS. +- [x] 3. Add a chat-page navigation entry to the Trace workbench. +- [x] 4. Verify Java compilation and static page syntax. +- [x] 5. Validate the page against a real trace endpoint when a local service is available. +- [x] 6. Archive the OpenSpec change after validation. +- [x] 7. Commit the completed change. diff --git a/openspec/specs/mvp-demo-trace-acceptance/spec.md b/openspec/specs/mvp-demo-trace-acceptance/spec.md index d416597..eb5a440 100644 --- a/openspec/specs/mvp-demo-trace-acceptance/spec.md +++ b/openspec/specs/mvp-demo-trace-acceptance/spec.md @@ -1,9 +1,7 @@ ## Purpose Provide a repeatable MVP demo flow that can run a chat diagnosis, expose its persisted execution trace, and submit feedback for the same session id. - ## Requirements - ### Requirement: Diagnosis trace can be queried by session id The system SHALL expose a read-only HTTP endpoint `GET /api/diagnosis/{sessionId}/trace` that returns the persisted diagnosis trace for the requested session id. @@ -64,3 +62,39 @@ The MVP demo SHALL document which trace fields to inspect for evidence, verifier #### Scenario: Checklist maps fields to interview claims - **WHEN** a developer reviews a trace response - **THEN** the checklist SHALL map concrete JSON paths to the claims made in the interview walkthrough + +### Requirement: MVP demo SHALL provide a browser trace workbench +The MVP demo SHALL provide a browser-accessible static page for inspecting one +diagnosis trace by session id using the existing read-only Trace API. + +#### Scenario: Existing trace renders in the workbench +- **WHEN** a reviewer opens the Trace workbench with a session id that exists +- **THEN** the page SHALL request `GET /api/diagnosis/{sessionId}/trace` +- **AND** it SHALL render session summary, ordered agent steps, ordered tool + invocations, verifier evaluation, and final answer when present + +#### Scenario: Trace workbench handles missing or failed traces +- **WHEN** the Trace API returns an error or the session id is empty +- **THEN** the page SHALL show a clear error or empty state without mutating any + diagnosis data + +### Requirement: Trace workbench SHALL expose evidence and skill boundaries +The Trace workbench SHALL make tool evidence and skill-loading boundaries visible +without requiring raw JSON inspection first. + +#### Scenario: Tool and verifier evidence are inspectable +- **WHEN** a trace includes tool invocations and verifier facts +- **THEN** the page SHALL show tool invocation counts, success status, tool + filters, verifier facts, and evidence references + +#### Scenario: RAG details are inspectable for lookup knowledge calls +- **WHEN** a `lookup_knowledge` invocation includes retrieval details +- **THEN** the page SHALL show query transform, retrieval trace, context pack, + rerank trace, and evidence block data where available + +#### Scenario: Skill boundary checks are visible +- **WHEN** a trace includes planner, executor, or verifier steps +- **THEN** the page SHALL show best-effort indicators for selected skill, + planner `read_skill` text mentions, executor `read_skill` text mentions, and + verifier `read_skill` text mentions + diff --git a/src/main/resources/static/index.html b/src/main/resources/static/index.html index 161a6a0..8aa46ef 100644 --- a/src/main/resources/static/index.html +++ b/src/main/resources/static/index.html @@ -34,6 +34,14 @@ 文档管理 + + + + + + Trace Workbench + +
近期对话 diff --git a/src/main/resources/static/trace.css b/src/main/resources/static/trace.css new file mode 100644 index 0000000..2b51ada --- /dev/null +++ b/src/main/resources/static/trace.css @@ -0,0 +1,606 @@ +:root { + --ink: #111827; + --ink-soft: #263241; + --paper: #f7f3ea; + --line: #d6cbbb; + --evidence: #1f7a6b; + --evidence-soft: #dfeee9; + --reject: #b5472f; + --reject-soft: #f4dfd8; + --warn: #c58a19; + --warn-soft: #f5ead1; + --steel: #5b6472; + --white: #fffdf8; + --shadow: 0 18px 48px rgba(17, 24, 39, 0.16); +} + +* { + box-sizing: border-box; +} + +html, +body { + min-height: 100%; +} + +body { + margin: 0; + background: var(--paper); + color: var(--ink); + font-family: Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif; + letter-spacing: 0; +} + +button, +input { + font: inherit; +} + +button { + cursor: pointer; +} + +button:focus-visible, +input:focus-visible, +a:focus-visible, +.tool-row:focus-visible { + outline: 3px solid rgba(31, 122, 107, 0.34); + outline-offset: 2px; +} + +svg { + width: 18px; + height: 18px; + fill: none; + stroke: currentColor; + stroke-width: 2; + stroke-linecap: round; + stroke-linejoin: round; +} + +.trace-app { + min-height: 100vh; + display: flex; + flex-direction: column; + padding: 22px; + gap: 14px; +} + +.trace-header { + display: grid; + grid-template-columns: minmax(260px, 1fr) minmax(340px, 560px); + gap: 18px; + align-items: end; + padding: 18px 20px; + background: var(--ink); + color: var(--white); + border-radius: 8px; + box-shadow: var(--shadow); +} + +.trace-back-link { + display: inline-flex; + align-items: center; + gap: 8px; + color: #cbd5e1; + text-decoration: none; + font-size: 13px; + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; +} + +.trace-header h1 { + margin: 8px 0 0; + font-size: clamp(28px, 4vw, 52px); + line-height: 0.95; + font-weight: 760; + letter-spacing: 0; +} + +.session-form { + display: grid; + gap: 8px; +} + +.session-form label, +.summary-cell span, +.pane-heading p, +.tool-meta, +.state-line, +.field-label { + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 12px; + text-transform: uppercase; + color: var(--steel); +} + +.session-form label { + color: #cbd5e1; +} + +.session-control { + display: grid; + grid-template-columns: minmax(0, 1fr) auto; + gap: 8px; +} + +.session-control input { + width: 100%; + min-width: 0; + height: 42px; + padding: 0 12px; + border: 1px solid #4b5563; + border-radius: 6px; + background: #0b1220; + color: var(--white); + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; +} + +.session-control button { + height: 42px; + display: inline-flex; + align-items: center; + justify-content: center; + gap: 8px; + padding: 0 14px; + border: 0; + border-radius: 6px; + color: var(--white); + background: var(--evidence); + font-weight: 700; +} + +.session-control button:disabled { + opacity: 0.62; + cursor: wait; +} + +.state-line { + min-height: 34px; + display: flex; + align-items: center; + padding: 8px 12px; + border: 1px solid var(--line); + border-radius: 6px; + background: var(--white); +} + +.state-line.error { + color: var(--reject); + border-color: #dfb2a5; + background: var(--reject-soft); +} + +.state-line.loading { + color: var(--evidence); + border-color: #a9d2c8; + background: var(--evidence-soft); +} + +.summary-strip { + display: grid; + grid-template-columns: repeat(7, minmax(120px, 1fr)); + border: 1px solid var(--line); + border-radius: 8px; + overflow: hidden; + background: var(--white); +} + +.summary-cell { + min-width: 0; + padding: 13px 14px; + border-right: 1px solid var(--line); +} + +.summary-cell:last-child { + border-right: 0; +} + +.summary-cell strong { + display: block; + margin-top: 5px; + font-size: 21px; + line-height: 1.1; + overflow-wrap: anywhere; +} + +.trace-grid { + min-height: 0; + flex: 1; + display: grid; + grid-template-columns: minmax(300px, 1.05fr) minmax(300px, 0.95fr) minmax(340px, 1.1fr); + gap: 14px; +} + +.trace-pane { + min-height: 620px; + display: flex; + flex-direction: column; + border: 1px solid var(--line); + border-radius: 8px; + background: var(--white); + overflow: hidden; +} + +.pane-heading { + min-height: 74px; + display: flex; + align-items: center; + justify-content: space-between; + gap: 14px; + padding: 14px 16px; + border-bottom: 1px solid var(--line); + background: #fbf7ef; +} + +.pane-heading p, +.pane-heading h2 { + margin: 0; +} + +.pane-heading h2 { + margin-top: 3px; + font-size: 18px; +} + +.pane-count { + min-width: 38px; + min-height: 30px; + display: inline-flex; + align-items: center; + justify-content: center; + padding: 5px 9px; + border-radius: 999px; + background: var(--ink); + color: var(--white); + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 12px; +} + +.trace-rail, +.tool-ledger, +.inspector-stack { + overflow: auto; +} + +.trace-rail { + position: relative; + padding: 18px 16px 24px 34px; +} + +.trace-rail::before { + content: ""; + position: absolute; + left: 23px; + top: 0; + bottom: 0; + width: 3px; + background: var(--ink); +} + +.step-entry { + position: relative; + margin-bottom: 14px; + padding: 12px; + border: 1px solid var(--line); + border-radius: 6px; + background: #fffaf1; +} + +.step-entry::before { + content: attr(data-index); + position: absolute; + left: -30px; + top: 12px; + width: 24px; + height: 24px; + display: grid; + place-items: center; + border: 2px solid var(--ink); + border-radius: 50%; + background: var(--paper); + color: var(--ink); + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 11px; + font-weight: 800; +} + +.step-entry.planner { + border-left: 5px solid var(--warn); +} + +.step-entry.executor { + border-left: 5px solid var(--evidence); +} + +.step-entry.verifier { + border-left: 5px solid var(--reject); +} + +.step-title, +.tool-title { + display: flex; + align-items: flex-start; + justify-content: space-between; + gap: 10px; +} + +.step-title h3, +.tool-title h3 { + margin: 0; + font-size: 16px; + overflow-wrap: anywhere; +} + +.badge { + display: inline-flex; + align-items: center; + justify-content: center; + min-height: 22px; + padding: 3px 8px; + border-radius: 999px; + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 11px; + font-weight: 700; + white-space: nowrap; +} + +.badge.ok { + color: #0b4f44; + background: var(--evidence-soft); +} + +.badge.warn { + color: #754d00; + background: var(--warn-soft); +} + +.badge.bad { + color: #7c2414; + background: var(--reject-soft); +} + +.meta-line { + display: flex; + flex-wrap: wrap; + gap: 8px; + margin-top: 9px; +} + +.meta-pill { + display: inline-flex; + align-items: center; + gap: 5px; + padding: 4px 7px; + border: 1px solid var(--line); + border-radius: 999px; + background: var(--white); + color: var(--steel); + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 11px; +} + +details { + margin-top: 10px; +} + +summary { + color: var(--ink-soft); + font-weight: 700; + cursor: pointer; +} + +pre, +.answer-box { + margin: 8px 0 0; + padding: 10px; + max-height: 280px; + overflow: auto; + border: 1px solid var(--line); + border-radius: 6px; + background: #fdf8ee; + color: #172033; + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 12px; + line-height: 1.55; + white-space: pre-wrap; + overflow-wrap: anywhere; +} + +.tool-filters { + display: flex; + flex-wrap: wrap; + gap: 8px; + padding: 12px 14px; + border-bottom: 1px solid var(--line); + background: #fffaf1; +} + +.tool-filter { + min-height: 30px; + padding: 5px 10px; + border: 1px solid var(--line); + border-radius: 999px; + background: var(--white); + color: var(--ink); + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 12px; +} + +.tool-filter.active { + border-color: var(--ink); + background: var(--ink); + color: var(--white); +} + +.tool-ledger { + padding: 12px 14px 18px; +} + +.tool-row { + width: 100%; + display: block; + margin-bottom: 10px; + padding: 12px; + border: 1px solid var(--line); + border-radius: 6px; + background: var(--white); + color: inherit; + text-align: left; +} + +.tool-row:hover, +.tool-row.selected { + border-color: var(--evidence); + background: #f5fbf8; +} + +.tool-row.failed { + border-color: #e1b4aa; +} + +.tool-meta { + display: flex; + flex-wrap: wrap; + gap: 7px; + margin-top: 8px; + color: var(--steel); + text-transform: none; +} + +.inspector-stack { + padding: 14px; +} + +.inspector-section { + padding: 0 0 16px; + margin-bottom: 16px; + border-bottom: 1px solid var(--line); +} + +.inspector-section:last-child { + margin-bottom: 0; + border-bottom: 0; +} + +.inspector-section h3 { + margin: 0 0 10px; + font-size: 15px; +} + +.boundary-grid { + display: grid; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 8px; +} + +.boundary-item, +.fact-row, +.field-row { + border: 1px solid var(--line); + border-radius: 6px; + background: #fffaf1; +} + +.boundary-item { + padding: 10px; +} + +.boundary-item strong { + display: block; + margin-top: 4px; + font-family: ui-monospace, SFMono-Regular, Consolas, "Liberation Mono", monospace; + font-size: 14px; + overflow-wrap: anywhere; +} + +.fact-row { + padding: 10px; + margin-bottom: 8px; +} + +.fact-row p { + margin: 0 0 8px; + line-height: 1.45; +} + +.field-grid { + display: grid; + gap: 8px; +} + +.field-row { + padding: 10px; +} + +.field-label { + display: block; + margin-bottom: 5px; + color: var(--steel); +} + +.empty-panel { + padding: 14px; + border: 1px dashed var(--line); + border-radius: 6px; + color: var(--steel); + background: #fffaf1; +} + +@media (max-width: 1180px) { + .summary-strip { + grid-template-columns: repeat(4, minmax(120px, 1fr)); + } + + .summary-cell:nth-child(4) { + border-right: 0; + } + + .trace-grid { + grid-template-columns: 1fr; + } + + .trace-pane { + min-height: 420px; + } +} + +@media (max-width: 760px) { + .trace-app { + padding: 12px; + } + + .trace-header { + grid-template-columns: 1fr; + } + + .session-control { + grid-template-columns: 1fr; + } + + .summary-strip { + grid-template-columns: repeat(2, minmax(0, 1fr)); + } + + .summary-cell, + .summary-cell:nth-child(4) { + border-right: 1px solid var(--line); + } + + .summary-cell:nth-child(2n) { + border-right: 0; + } + + .boundary-grid { + grid-template-columns: 1fr; + } +} + +@media (prefers-reduced-motion: reduce) { + *, + *::before, + *::after { + scroll-behavior: auto !important; + transition: none !important; + animation: none !important; + } +} diff --git a/src/main/resources/static/trace.html b/src/main/resources/static/trace.html new file mode 100644 index 0000000..37f18d6 --- /dev/null +++ b/src/main/resources/static/trace.html @@ -0,0 +1,129 @@ + + + + + + Trace Workbench - SuperBizAgent + + + +
+
+
+ + + Chat + +

Trace workbench

+
+ +
+ +
+ + +
+
+
+ +
Enter a session id to load a persisted diagnosis trace.
+ +
+
+ Status + - +
+
+ Verdict + - +
+
+ Groundedness + - +
+
+ Steps + - +
+
+ Tools + - +
+
+ Duration + - +
+
+ Tokens + - +
+
+ +
+
+
+
+

Agent rail

+

Persisted steps

+
+ 0 +
+
+
+ +
+
+
+

Evidence ledger

+

Tool invocations

+
+ 0 +
+
+
+
+ +
+
+
+

Inspector

+

Verifier and RAG

+
+ idle +
+ +
+
+

Skill boundary

+
+
+ +
+

Verifier facts

+
+
+ +
+

Selected tool

+
+
+ +
+

Final answer

+
-
+
+
+
+
+
+ + + + diff --git a/src/main/resources/static/trace.js b/src/main/resources/static/trace.js new file mode 100644 index 0000000..5326002 --- /dev/null +++ b/src/main/resources/static/trace.js @@ -0,0 +1,574 @@ +class TraceWorkbench { + constructor() { + this.trace = null; + this.selectedToolId = null; + this.activeFilter = 'all'; + + this.sessionForm = document.getElementById('sessionForm'); + this.sessionIdInput = document.getElementById('sessionIdInput'); + this.loadButton = document.getElementById('loadButton'); + this.stateLine = document.getElementById('stateLine'); + this.stepsRail = document.getElementById('stepsRail'); + this.toolLedger = document.getElementById('toolLedger'); + this.toolFilters = document.getElementById('toolFilters'); + this.skillBoundary = document.getElementById('skillBoundary'); + this.verifierFacts = document.getElementById('verifierFacts'); + this.toolInspector = document.getElementById('toolInspector'); + this.finalAnswer = document.getElementById('finalAnswer'); + + this.bindEvents(); + this.bootstrapFromUrl(); + } + + bindEvents() { + this.sessionForm.addEventListener('submit', (event) => { + event.preventDefault(); + this.loadTrace(this.sessionIdInput.value.trim()); + }); + } + + bootstrapFromUrl() { + const params = new URLSearchParams(window.location.search); + const sessionId = params.get('sessionId'); + if (sessionId) { + this.sessionIdInput.value = sessionId; + this.loadTrace(sessionId); + return; + } + this.renderEmpty(); + } + + async loadTrace(sessionId) { + if (!sessionId) { + this.setState('Enter a session id before loading.', 'error'); + return; + } + + this.setState(`Loading ${sessionId}...`, 'loading'); + this.loadButton.disabled = true; + + try { + const response = await fetch(`/api/diagnosis/${encodeURIComponent(sessionId)}/trace`); + const data = await this.handleResponse(response); + this.trace = data; + this.selectedToolId = data.toolInvocations && data.toolInvocations.length + ? data.toolInvocations[0].id + : null; + this.activeFilter = 'all'; + this.updateUrl(sessionId); + this.renderTrace(); + this.setState(`Loaded ${sessionId}.`, ''); + } catch (error) { + this.trace = null; + this.renderEmpty(); + this.setState(error.message, 'error'); + } finally { + this.loadButton.disabled = false; + } + } + + async handleResponse(response) { + let payload = null; + try { + payload = await response.json(); + } catch (error) { + throw new Error(`HTTP ${response.status}: response is not JSON`); + } + + if (!response.ok || payload.code !== 200) { + throw new Error(payload.message || `HTTP ${response.status}: trace request failed`); + } + + return payload.data; + } + + updateUrl(sessionId) { + const url = new URL(window.location.href); + url.searchParams.set('sessionId', sessionId); + window.history.replaceState({}, '', url.toString()); + } + + renderTrace() { + const session = this.trace.session || {}; + const verifier = this.getVerifier(); + const summary = this.trace.summary || {}; + const steps = this.trace.steps || []; + const tools = this.trace.toolInvocations || []; + + this.text('summaryStatus', session.status || '-'); + this.text('summaryVerdict', verifier.verdict || '-'); + this.text('summaryScore', this.formatScore(verifier.groundedness_score)); + this.text('summarySteps', `${summary.returnedStepCount ?? steps.length}/${summary.persistedStepCount ?? session.stepCount ?? '-'}`); + this.text('summaryTools', `${summary.returnedToolCallCount ?? tools.length}/${summary.persistedToolCallCount ?? session.toolCallCount ?? '-'}`); + this.text('summaryDuration', this.formatDuration(session.totalDurationMs)); + this.text('summaryTokens', this.formatNumber(session.totalTokenCount)); + this.text('stepCount', steps.length); + this.text('toolCount', tools.length); + this.text('inspectorMode', tools.length ? 'tool' : 'trace'); + + this.renderSteps(steps); + this.renderToolFilters(tools); + this.renderTools(tools); + this.renderSkillBoundary(); + this.renderVerifier(verifier); + this.renderToolInspector(); + this.finalAnswer.textContent = session.answer || 'No final answer persisted.'; + } + + renderEmpty() { + this.text('summaryStatus', '-'); + this.text('summaryVerdict', '-'); + this.text('summaryScore', '-'); + this.text('summarySteps', '-'); + this.text('summaryTools', '-'); + this.text('summaryDuration', '-'); + this.text('summaryTokens', '-'); + this.text('stepCount', '0'); + this.text('toolCount', '0'); + this.text('inspectorMode', 'idle'); + + this.stepsRail.innerHTML = '
No trace loaded.
'; + this.toolFilters.innerHTML = ''; + this.toolLedger.innerHTML = '
No tool invocations loaded.
'; + this.skillBoundary.innerHTML = this.renderBoundaryItems({ + selectedSkill: '-', + plannerReadSkill: 0, + executorReadSkill: 0, + verifierReadSkill: 0 + }); + this.verifierFacts.innerHTML = '
No verifier evaluation loaded.
'; + this.toolInspector.innerHTML = '
Select a tool invocation after loading a trace.
'; + this.finalAnswer.textContent = '-'; + } + + renderSteps(steps) { + if (!steps.length) { + this.stepsRail.innerHTML = '
No persisted agent steps.
'; + return; + } + + this.stepsRail.innerHTML = steps.map((step, index) => { + const agent = (step.agentName || 'agent').toLowerCase(); + const cls = this.agentClass(agent); + return ` +
+
+

${this.escape(step.agentName || 'agent')}

+ ${this.renderBadge(step.hasToolCall ? 'tool' : 'model', step.hasToolCall ? 'warn' : 'ok')} +
+
+ step ${this.escape(step.stepIndex ?? index)} + ${this.escape(this.formatDuration(step.durationMs))} + ${this.escape(this.formatNumber(step.tokenCount))} tokens + ${this.escape(this.formatDate(step.createdAt))} +
+ ${this.renderTextDetail('Model input', step.modelInput)} + ${this.renderTextDetail('Model output', step.modelOutput)} + ${this.renderTextDetail('Thought', step.thought)} +
+ `; + }).join(''); + } + + renderToolFilters(tools) { + if (!tools.length) { + this.toolFilters.innerHTML = ''; + return; + } + + const groups = this.groupByTool(tools); + const filters = [ + { key: 'all', label: `all ${tools.length}` }, + { key: 'success', label: `success ${tools.filter(tool => tool.success !== false).length}` }, + { key: 'failed', label: `failed ${tools.filter(tool => tool.success === false).length}` }, + ...Object.entries(groups).map(([name, count]) => ({ key: `tool:${name}`, label: `${name} ${count}` })) + ]; + + this.toolFilters.innerHTML = filters.map(filter => ` + + `).join(''); + + this.toolFilters.querySelectorAll('button').forEach((button) => { + button.addEventListener('click', () => { + this.activeFilter = button.dataset.filter; + this.renderToolFilters(this.trace.toolInvocations || []); + this.renderTools(this.trace.toolInvocations || []); + }); + }); + } + + renderTools(tools) { + const filtered = tools.filter(tool => this.matchesFilter(tool)); + if (!filtered.length) { + this.toolLedger.innerHTML = '
No tool invocations match this filter.
'; + return; + } + + this.toolLedger.innerHTML = filtered.map((tool, index) => { + const failed = tool.success === false; + const selected = String(tool.id) === String(this.selectedToolId); + return ` + + `; + }).join(''); + + this.toolLedger.querySelectorAll('.tool-row').forEach((row) => { + row.addEventListener('click', () => { + this.selectedToolId = row.dataset.toolId; + this.renderTools(this.trace.toolInvocations || []); + this.renderToolInspector(); + }); + }); + } + + renderSkillBoundary() { + const steps = this.trace.steps || []; + const selectedSkill = this.extractSelectedSkill(steps); + const counts = { + selectedSkill: selectedSkill || '-', + plannerReadSkill: this.countReadSkill(steps, 'planner'), + executorReadSkill: this.countReadSkill(steps, 'executor'), + verifierReadSkill: this.countReadSkill(steps, 'verifier') + }; + this.skillBoundary.innerHTML = this.renderBoundaryItems(counts); + } + + renderBoundaryItems(counts) { + return ` +
+ Selected skill + ${this.escape(counts.selectedSkill)} +
+
+ Planner mentions + ${this.escape(counts.plannerReadSkill)} +
+
+ Executor mentions + ${this.escape(counts.executorReadSkill)} +
+
+ Verifier mentions + ${this.escape(counts.verifierReadSkill)} +
+ `; + } + + renderVerifier(verifier) { + if (!verifier || !Object.keys(verifier).length) { + this.verifierFacts.innerHTML = '
No verifier evaluation in selfEvaluation.
'; + return; + } + + const facts = Array.isArray(verifier.facts_checked) ? verifier.facts_checked : []; + const traceSummary = Array.isArray(verifier.tool_trace_summary) ? verifier.tool_trace_summary : []; + const header = ` +
+ ${this.renderBadge(verifier.verdict || 'verdict -', this.verdictClass(verifier.verdict))} + score ${this.escape(this.formatScore(verifier.groundedness_score))} + facts ${facts.length} + trace refs ${traceSummary.length} +
+ `; + + if (!facts.length) { + this.verifierFacts.innerHTML = header + '
Verifier returned no facts_checked rows.
'; + return; + } + + this.verifierFacts.innerHTML = header + facts.map((fact) => { + const refs = Array.isArray(fact.evidence_refs) ? fact.evidence_refs : []; + return ` +
+

${this.escape(fact.fact || '-')}

+
+ ${this.renderBadge(fact.verification || '-', this.factClass(fact.verification))} + ${fact.is_critical ? 'critical' : 'supporting'} + refs ${refs.length} +
+ ${fact.detail ? `
${this.escape(fact.detail)}
` : ''} + ${refs.length ? `
${this.escape(this.prettyJson(refs))}
` : ''} +
+ `; + }).join(''); + } + + renderToolInspector() { + const tools = this.trace ? (this.trace.toolInvocations || []) : []; + const tool = tools.find(item => String(item.id) === String(this.selectedToolId)); + if (!tool) { + this.toolInspector.innerHTML = '
No tool invocation selected.
'; + return; + } + + const retrieval = tool.retrievalDetails || {}; + const ragFields = [ + ['Input params', tool.inputParams || this.parseJson(tool.inputParamsRaw)], + ['Output preview', tool.outputPreview || '-'], + ['Retrieval layer', tool.retrievalLayer || '-'], + ['L0 matches', tool.l0MatchCount ?? '-'], + ['L1 matches', tool.l1MatchCount ?? '-'], + ['Relevance', tool.relevanceLevel || '-'], + ['Dedup reason', tool.dedupReason || '-'], + ['Query transform', retrieval.query_transform], + ['Retrieval trace', retrieval.retrieval_trace], + ['Context pack', retrieval.context_pack_summary], + ['Rerank trace', retrieval.rerank_trace], + ['Evidence blocks', retrieval.evidence_blocks] + ]; + + this.toolInspector.innerHTML = ` +
+ ${this.renderBadge(tool.success === false ? 'failed' : 'ok', tool.success === false ? 'bad' : 'ok')} + ${this.escape(tool.toolName || 'tool')} + ${this.escape(this.formatDuration(tool.durationMs))} + output ${this.escape(this.formatNumber(tool.outputLength))} + ${tool.truncated ? 'truncated' : ''} +
+ ${tool.errorMessage ? `
${this.escape(tool.errorMessage)}
` : ''} +
+ ${ragFields.map(([label, value]) => this.renderField(label, value)).join('')} +
+ `; + } + + renderField(label, value) { + const empty = value === undefined || value === null || value === ''; + return ` +
+ ${this.escape(label)} +
${this.escape(empty ? '-' : this.stringifyValue(value))}
+
+ `; + } + + renderTextDetail(label, text) { + if (!text) { + return ''; + } + return ` +
+ ${this.escape(label)} +
${this.escape(text)}
+
+ `; + } + + renderBadge(text, cls) { + return `${this.escape(text)}`; + } + + getVerifier() { + const evaluation = (this.trace && this.trace.session && this.trace.session.selfEvaluation) || {}; + if (evaluation.verifier_evaluation) { + return evaluation.verifier_evaluation; + } + if (evaluation.verdict || evaluation.groundedness_score || evaluation.facts_checked) { + return evaluation; + } + return {}; + } + + extractSelectedSkill(steps) { + const plannerText = steps + .filter(step => (step.agentName || '').toLowerCase().includes('planner')) + .map(step => `${step.modelOutput || ''}\n${step.thought || ''}`) + .join('\n'); + const allText = steps + .map(step => `${step.modelInput || ''}\n${step.modelOutput || ''}\n${step.thought || ''}`) + .join('\n'); + const sourceText = `${plannerText}\n${allText}`; + const direct = sourceText.match(/"selected_skill"\s*:\s*"([^"]+)"/); + if (direct) { + return direct[1]; + } + const skillName = sourceText.match(/"skill_name"\s*:\s*"([^"]+)"/); + if (skillName) { + return skillName[1]; + } + const loose = sourceText.match(/selected_skill\s*[:=]\s*([A-Za-z0-9_.-]+)/); + return loose ? loose[1] : ''; + } + + countReadSkill(steps, agentName) { + const text = steps + .filter(step => (step.agentName || '').toLowerCase().includes(agentName)) + .map(step => `${step.modelInput || ''}\n${step.modelOutput || ''}\n${step.thought || ''}`) + .join('\n'); + const matches = text.match(/\bread_skill\b/g); + return matches ? matches.length : 0; + } + + matchesFilter(tool) { + if (this.activeFilter === 'all') { + return true; + } + if (this.activeFilter === 'success') { + return tool.success !== false; + } + if (this.activeFilter === 'failed') { + return tool.success === false; + } + if (this.activeFilter.startsWith('tool:')) { + return tool.toolName === this.activeFilter.slice(5); + } + return true; + } + + groupByTool(tools) { + return tools.reduce((groups, tool) => { + const name = tool.toolName || 'tool'; + groups[name] = (groups[name] || 0) + 1; + return groups; + }, {}); + } + + agentClass(agent) { + if (agent.includes('planner')) { + return 'planner'; + } + if (agent.includes('executor')) { + return 'executor'; + } + if (agent.includes('verifier')) { + return 'verifier'; + } + return ''; + } + + verdictClass(verdict) { + if (!verdict) { + return 'warn'; + } + const value = String(verdict).toUpperCase(); + if (value.includes('PASS')) { + return 'ok'; + } + if (value.includes('REJECT')) { + return 'bad'; + } + return 'warn'; + } + + factClass(verification) { + if (!verification) { + return 'warn'; + } + const value = String(verification).toUpperCase(); + if (value.includes('SUPPORTED') || value.includes('PASS')) { + return 'ok'; + } + if (value.includes('UNSUPPORTED') || value.includes('CONTRADICT')) { + return 'bad'; + } + return 'warn'; + } + + parseJson(raw) { + if (!raw || typeof raw !== 'string') { + return raw; + } + try { + return JSON.parse(raw); + } catch (error) { + return raw; + } + } + + stringifyValue(value) { + if (typeof value === 'string') { + return value; + } + return this.prettyJson(value); + } + + prettyJson(value) { + try { + return JSON.stringify(value, null, 2); + } catch (error) { + return String(value); + } + } + + formatDuration(ms) { + if (ms === undefined || ms === null || ms === '') { + return '-'; + } + if (ms < 1000) { + return `${ms}ms`; + } + const seconds = ms / 1000; + if (seconds < 60) { + return `${seconds.toFixed(1)}s`; + } + const minutes = Math.floor(seconds / 60); + const remaining = Math.round(seconds % 60); + return `${minutes}m ${remaining}s`; + } + + formatNumber(value) { + if (value === undefined || value === null || value === '') { + return '-'; + } + return Number(value).toLocaleString('en-US'); + } + + formatScore(value) { + if (value === undefined || value === null || value === '') { + return '-'; + } + const number = Number(value); + return Number.isFinite(number) ? number.toFixed(2) : String(value); + } + + formatDate(value) { + if (!value) { + return '-'; + } + const date = new Date(value); + if (Number.isNaN(date.getTime())) { + return String(value); + } + return date.toLocaleString('en-US', { hour12: false }); + } + + text(id, value) { + const node = document.getElementById(id); + if (node) { + node.textContent = value; + } + } + + setState(message, statusClass) { + this.stateLine.textContent = message; + this.stateLine.className = `state-line ${statusClass || ''}`.trim(); + } + + escape(value) { + return String(value) + .replace(/&/g, '&') + .replace(//g, '>') + .replace(/"/g, '"') + .replace(/'/g, '''); + } +} + +document.addEventListener('DOMContentLoaded', () => { + new TraceWorkbench(); +});