diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/StateGraphReActAgent.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/StateGraphReActAgent.java index a2d49ba9..905df5f3 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/StateGraphReActAgent.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/StateGraphReActAgent.java @@ -254,7 +254,7 @@ public class StateGraphReActAgent extends BaseAgent implements StructuredStreamC lastEmittedStreamedContent.set(streamed); addWithKindEvent(deltas, streamedContentDelta(isFinalAnswerTurn, output.state().value(NEEDS_TOOL_CALL, false), - output.state().value(CURRENT_ITERATION, 0), + output.state().value(TOOL_CALL_COUNT, 0), streamed)); } @@ -426,7 +426,7 @@ public class StateGraphReActAgent extends BaseAgent implements StructuredStreamC lastEmittedStreamedContent.set(streamed); addWithKindEvent(deltas, streamedContentDelta(isFinalAnswerTurn, output.state().value(NEEDS_TOOL_CALL, false), - output.state().value(CURRENT_ITERATION, 0), + output.state().value(TOOL_CALL_COUNT, 0), streamed)); } @@ -648,15 +648,23 @@ public class StateGraphReActAgent extends BaseAgent implements StructuredStreamC * {@link ContentKind}: the graph is the only layer that definitively knows * whether the completion carried tool calls ({@code NEEDS_TOOL_CALL}) and * whether any tool observation preceded the text this turn - * ({@code CURRENT_ITERATION} — ObservationNode increments it after each - * observed round, so iteration 0 means "no observation yet"). Downstream - * consumers read the tag instead of re-deriving it from stream structure. + * ({@code TOOL_CALL_COUNT} — ObservationNode adds each round's observed + * results to it, so 0 means "no observation yet"). Downstream consumers + * read the tag instead of re-deriving it from stream structure. + * + *

The observation signal MUST be the observation counter, not + * {@code CURRENT_ITERATION}: the latter is an iteration budget + * counter that ObservationNode refunds for progressive-disclosure rounds + * (load_skill / enable_tool) and GoalEvaluationNode resets to 0 on a hard + * continuation. Either path leaves the budget at 0 after real observations + * already landed, which tagged grounded narration as provisional and made + * renderers collapse it. * *

@@ -685,11 +693,11 @@ public class StateGraphReActAgent extends BaseAgent implements StructuredStreamC } static AgentService.StreamDelta streamedContentDelta(boolean isFinalAnswerTurn, boolean carriesToolCalls, - int iteration, String streamed) { + int observationCount, String streamed) { if (isFinalAnswerTurn) { return AgentService.StreamDelta.persistOnly(streamed, null, ContentKind.FINAL_ANSWER); } - ContentKind kind = carriesToolCalls && iteration == 0 + ContentKind kind = carriesToolCalls && observationCount == 0 ? ContentKind.PRE_TOOL_NARRATION : ContentKind.GROUNDED_NARRATION; return AgentService.StreamDelta.segmentOnly(streamed, null, kind); diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/StateGraphReActAgentStreamedContentDeltaTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/StateGraphReActAgentStreamedContentDeltaTest.java index 87f149ff..88e31ebf 100644 --- a/mateclaw-server/src/test/java/vip/mate/agent/graph/StateGraphReActAgentStreamedContentDeltaTest.java +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/StateGraphReActAgentStreamedContentDeltaTest.java @@ -5,6 +5,9 @@ import org.junit.jupiter.api.Test; import vip.mate.agent.AgentService; import vip.mate.agent.ContentKind; +import java.util.ArrayList; +import java.util.List; + import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNull; @@ -28,16 +31,20 @@ import static org.junit.jupiter.api.Assertions.assertTrue; * graph knows definitively whether the completion carried tool calls and whether * any observation preceded the text this turn, so downstream consumers read the * tag instead of re-deriving the category from stream structure. + * + *

The observation signal is the {@code TOOL_CALL_COUNT} observation counter, + * NOT the {@code CURRENT_ITERATION} budget counter — see + * {@code observationsWithoutIterationBudget_taggedGrounded} for why. */ class StateGraphReActAgentStreamedContentDeltaTest { @Test - @DisplayName("iteration 0 with tool calls → segmentOnly + PRE_TOOL_NARRATION — provisional rehearsal") + @DisplayName("zero observations with tool calls → segmentOnly + PRE_TOOL_NARRATION — provisional rehearsal") void preToolNarration_taggedProvisional() { AgentService.StreamDelta d = StateGraphReActAgent.streamedContentDelta( /* isFinalAnswerTurn */ false, /* carriesToolCalls */ true, - /* iteration */ 0, + /* observationCount */ 0, "先加载 skill,然后逐个查询。"); assertTrue(d.persistenceOnly(), @@ -50,10 +57,10 @@ class StateGraphReActAgentStreamedContentDeltaTest { } @Test - @DisplayName("iteration ≥1 narration → segmentOnly + GROUNDED_NARRATION even when the completion issues more tool calls") + @DisplayName("≥1 observation → segmentOnly + GROUNDED_NARRATION even when the completion issues more tool calls") void postObservationNarration_taggedGrounded() { AgentService.StreamDelta d = StateGraphReActAgent.streamedContentDelta( - false, /* carriesToolCalls */ true, /* iteration */ 1, "第一间空闲,继续查下一间。"); + false, /* carriesToolCalls */ true, /* observationCount */ 1, "第一间空闲,继续查下一间。"); assertTrue(d.segmentOnly()); assertEquals(ContentKind.GROUNDED_NARRATION, d.kind(), @@ -61,10 +68,31 @@ class StateGraphReActAgentStreamedContentDeltaTest { } @Test - @DisplayName("iteration 0 without tool calls (non-terminal) → GROUNDED_NARRATION — conservative, never replaced") + @DisplayName("regression: observations landed while the iteration budget still reads 0 → GROUNDED_NARRATION") + void observationsWithoutIterationBudget_taggedGrounded() { + // The false-collapse this pins. Two graph paths leave CURRENT_ITERATION + // at 0 after real observations already happened: + // - ObservationNode refunds the iteration for a progressive-disclosure + // round (load_skill / enable_tool), and + // - GoalEvaluationNode resets it to 0 on a hard continuation. + // Reading the budget counter tagged the next round's grounded narration + // as provisional, so renderers collapsed it. Observed live: a turn whose + // first round was `load_skill` had its second-round narration + // ("我先检查…环境配置和 API 参考文档。") folded behind the "模型在工具执行前 + // 预写的内容" toggle. The observation counter is not refunded or reset. + AgentService.StreamDelta d = StateGraphReActAgent.streamedContentDelta( + false, /* carriesToolCalls */ true, /* observationCount */ 1, + "我先检查腾讯会议技能的环境配置和 API 参考文档。"); + + assertEquals(ContentKind.GROUNDED_NARRATION, d.kind(), + "one observation already landed — narration after it is grounded regardless of the iteration budget"); + } + + @Test + @DisplayName("zero observations without tool calls (non-terminal) → GROUNDED_NARRATION — conservative, never replaced") void noToolCallCompletion_taggedGrounded() { AgentService.StreamDelta d = StateGraphReActAgent.streamedContentDelta( - false, /* carriesToolCalls */ false, /* iteration */ 0, "narrative without tools"); + false, /* carriesToolCalls */ false, /* observationCount */ 0, "narrative without tools"); assertEquals(ContentKind.GROUNDED_NARRATION, d.kind(), "a completion that closed without tool calls has no later observation to defer to — keep it"); @@ -118,7 +146,7 @@ class StateGraphReActAgentStreamedContentDeltaTest { @Test @DisplayName("kind-carrying delta is followed by a segment_kind broadcast event") void kindDeltaEmitsSegmentKindEvent() { - java.util.List deltas = new java.util.ArrayList<>(); + List deltas = new ArrayList<>(); StateGraphReActAgent.addWithKindEvent(deltas, StateGraphReActAgent.streamedContentDelta(false, true, 0, "先查询。")); @@ -132,7 +160,7 @@ class StateGraphReActAgentStreamedContentDeltaTest { @Test @DisplayName("untagged delta emits no segment_kind event") void untaggedDeltaEmitsNoEvent() { - java.util.List deltas = new java.util.ArrayList<>(); + List deltas = new ArrayList<>(); StateGraphReActAgent.addWithKindEvent(deltas, AgentService.StreamDelta.segmentOnly("x", null)); assertEquals(1, deltas.size()); diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java index 29bb5325..2bc7b2c2 100644 --- a/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java @@ -75,6 +75,20 @@ class ObservationNodeRefundTest { assertNull(out.get(ITERATION_REFUND_COUNT)); } + @Test + @DisplayName("退还迭代的轮次仍然递增观察计数(内容分类只能读观察计数,不能读迭代预算)") + void refundedRound_stillAdvancesObservationCount() throws Exception { + // The invariant StateGraphReActAgent.streamedContentDelta relies on: a + // refunded round did real observation work even though it was not + // charged an iteration. Classifying content by CURRENT_ITERATION here + // tagged the NEXT round's grounded narration as pre-tool rehearsal, and + // renderers collapsed it. TOOL_CALL_COUNT is never refunded or reset. + Map out = node().apply(state(0, 0, List.of(result("load_skill")))); + + assertEquals(0, out.get(CURRENT_ITERATION), "iteration budget stays put — the round was refunded"); + assertEquals(1, out.get(TOOL_CALL_COUNT), "the observation still happened and must be counted"); + } + @Test @DisplayName("退还次数达上限后不再退还") void refundCapReached_consumesIteration() throws Exception { diff --git a/mateclaw-ui/src/composables/chat/__tests__/supersede.test.ts b/mateclaw-ui/src/composables/chat/__tests__/supersede.test.ts new file mode 100644 index 00000000..28d18f68 --- /dev/null +++ b/mateclaw-ui/src/composables/chat/__tests__/supersede.test.ts @@ -0,0 +1,120 @@ +import { describe, it, expect } from 'vitest' +import type { MessageSegment } from '@/types' +import { + SUPERSEDED_REASON_PRE_TOOL, + markSuperseded, + supersedesProvisionalNarration, +} from '../supersede' + +function content(over: Partial = {}): MessageSegment { + return { + id: 'seg-1', + type: 'content', + status: 'completed', + text: '我先查一下会议室占用情况:', + timestamp: 0, + ...over, + } as MessageSegment +} + +describe('supersedesProvisionalNarration', () => { + it('replaces a provisional narration once an observation landed after it', () => { + const prev = content({ kind: 'pre_tool_narration' }) + // opened at 0 observations, one tool completed since + expect(supersedesProvisionalNarration(prev, 1, 0)).toBe(true) + }) + + it('keeps narration that no observation followed', () => { + // The reported false-collapse: a phase boundary splits one round's text, so + // a second content span opens with no tool having run in between. Nothing + // replaced the first span — it must stay visible. + const prev = content({ kind: 'pre_tool_narration' }) + expect(supersedesProvisionalNarration(prev, 0, 0)).toBe(false) + }) + + it('keeps narration when the only observations predate it', () => { + // Span opened after two observations; a third span opens with the count + // unchanged — the later text is not a replacement, just more narration. + const prev = content({ kind: 'pre_tool_narration' }) + expect(supersedesProvisionalNarration(prev, 2, 2)).toBe(false) + }) + + it('never touches grounded narration or final answers', () => { + expect(supersedesProvisionalNarration(content({ kind: 'grounded_narration' }), 3, 0)).toBe(false) + expect(supersedesProvisionalNarration(content({ kind: 'final_answer' }), 3, 0)).toBe(false) + }) + + it('never touches an untagged span', () => { + // `segment_kind` has not arrived (or the producer predates the tag) — the + // persisted-metadata pass decides those, not the live rule. + expect(supersedesProvisionalNarration(content(), 3, 0)).toBe(false) + }) + + it('is idempotent — an already-collapsed span is not re-marked', () => { + const prev = content({ kind: 'pre_tool_narration', superseded: true }) + expect(supersedesProvisionalNarration(prev, 5, 0)).toBe(false) + }) + + it('handles the turn-opening span with no predecessor', () => { + expect(supersedesProvisionalNarration(undefined, 1, 0)).toBe(false) + }) + + it('ignores non-content predecessors', () => { + const toolSeg = { id: 'to-1', type: 'tool_call', status: 'completed', kind: 'pre_tool_narration' } as any + expect(supersedesProvisionalNarration(toolSeg, 1, 0)).toBe(false) + }) +}) + +describe('markSuperseded', () => { + it('writes the three annotation fields renderers read', () => { + const seg = content({ kind: 'pre_tool_narration' }) + markSuperseded(seg, 'seg-2') + + expect(seg.superseded).toBe(true) + expect(seg.supersededBySegmentId).toBe('seg-2') + expect(seg.supersededReason).toBe(SUPERSEDED_REASON_PRE_TOOL) + }) +}) + +describe('multi-round timeline', () => { + /** + * Walks the guard across a full ReAct turn, the way useChat drives it: each + * content span records the observation count it opened at, and only the + * immediately preceding span is ever a candidate. + */ + it('collapses only the span an observation actually replaced', () => { + const marks = new Map() + const spans: MessageSegment[] = [] + let observations = 0 + + /** Mirrors the content_delta branch that opens a new content span. */ + const openSpan = (id: string): MessageSegment => { + const prev = spans.at(-1) + if (prev && supersedesProvisionalNarration(prev, observations, marks.get(String(prev.id)) ?? 0)) { + markSuperseded(prev, id) + } + const seg = content({ id, status: 'running' }) + marks.set(id, observations) + spans.push(seg) + return seg + } + + // Round 0: narration written before any tool ran. `segment_kind` lands at + // the end of the round, after the span opened. + const s0 = openSpan('seg-0') + s0.kind = 'pre_tool_narration' + observations++ // load_skill observed — an iteration-refunded round still counts + + // Round 1: narration written with that observation in hand. It replaces s0 + // and is itself tagged provisional (its completion calls tools again). + const s1 = openSpan('seg-1') + s1.kind = 'pre_tool_narration' + + // A phase boundary splits round 1's text — no tool ran in between. + openSpan('seg-2') + + expect(s0.superseded).toBe(true) + expect(s0.supersededBySegmentId).toBe('seg-1') + expect(s1.superseded).toBeUndefined() + }) +}) diff --git a/mateclaw-ui/src/composables/chat/supersede.ts b/mateclaw-ui/src/composables/chat/supersede.ts new file mode 100644 index 00000000..a05c512e --- /dev/null +++ b/mateclaw-ui/src/composables/chat/supersede.ts @@ -0,0 +1,46 @@ +import type { MessageSegment } from '@/types' + +/** + * Live counterpart of the backend's provisional-narration policy. + * + * A `pre_tool_narration` span is text the model wrote in a completion that went + * on to call tools, before any of this turn's observations landed — it may be + * process narration or a rehearsal of a result the tool had not produced yet. + * It is replaced only when a later content span was actually written with an + * observation in hand. Anything else (a phase boundary splitting one round's + * text, a second span inside the same completion, an unrelated earlier span) + * leaves it standing: nothing has superseded it, and collapsing it there hides + * narration the user needs. + */ + +/** Wire value shared with the backend so renderers need no new vocabulary. */ +export const SUPERSEDED_REASON_PRE_TOOL = 'pre_tool_content_replaced_by_post_tool_answer' + +/** + * Whether a content span opening now replaces `prev`. + * + * @param prev the content span immediately preceding the new one, + * or undefined when this is the turn's first + * @param observationCount tool observations completed so far this turn + * @param prevObservationMark observation count when `prev` was opened + */ +export function supersedesProvisionalNarration( + prev: MessageSegment | undefined, + observationCount: number, + prevObservationMark: number, +): boolean { + if (!prev || prev.type !== 'content') return false + // Untagged spans (pre-tag producers, or a span whose `segment_kind` event has + // not arrived yet) are never collapsed live — the persisted-metadata pass + // decides those. + if (prev.kind !== 'pre_tool_narration') return false + if (prev.superseded) return false + return observationCount > prevObservationMark +} + +/** Apply the three annotation fields renderers read. */ +export function markSuperseded(seg: MessageSegment, bySegmentId: string): void { + seg.superseded = true + seg.supersededBySegmentId = bySegmentId + seg.supersededReason = SUPERSEDED_REASON_PRE_TOOL +} diff --git a/mateclaw-ui/src/composables/chat/useChat.ts b/mateclaw-ui/src/composables/chat/useChat.ts index 8a4fa3d9..9f23d9f2 100644 --- a/mateclaw-ui/src/composables/chat/useChat.ts +++ b/mateclaw-ui/src/composables/chat/useChat.ts @@ -13,6 +13,7 @@ import { ref, computed } from 'vue' import { useMessages } from './useMessages' import { useStream } from './useStream' import { useMessageQueue } from './useMessageQueue' +import { supersedesProvisionalNarration, markSuperseded } from './supersede' import { useGoalStore } from '@/stores/useGoalStore' import { useSystemSettingsStore } from '@/stores/useSystemSettingsStore' import { storeToRefs } from 'pinia' @@ -232,6 +233,16 @@ export function useChat(options: UseChatOptions): UseChatReturn { const segIdCounter = { value: 0 } const genSegId = () => `seg-${Date.now()}-${segIdCounter.value++}` + /** + * Tool observations completed so far this turn, and the count each content + * segment was opened at. A provisional narration is only replaced once an + * observation actually landed after it, so the live collapse needs both the + * running total and each span's mark — the same bookkeeping the backend + * tracker does. Turn-scoped: reset with the rest of the streaming state. + */ + let observationCount = 0 + const segmentObservationMark = new Map() + /** * Fine-grained lifecycle stage exposed to the UI for the "connecting → started * → context_prepared → llm_request_sent → streaming" loading bar. Reset on @@ -268,6 +279,8 @@ export function useChat(options: UseChatOptions): UseChatReturn { function resetCurrentTurnState() { currentSegments.value = [] segIdCounter.value = 0 + observationCount = 0 + segmentObservationMark.clear() bufferedText = '' bufferedThinking = '' activeTurnId = `turn-${Date.now()}-${Math.random().toString(36).slice(2, 6)}` @@ -495,18 +508,19 @@ export function useChat(options: UseChatOptions): UseChatReturn { // Close any running thinking segment first const thinkingSeg = segs.findLast((s: MessageSegment) => s.type === 'thinking' && s.status === 'running') if (thinkingSeg) thinkingSeg.status = 'completed' + // The content span that precedes the one about to open — the only + // candidate this span can replace. + const prevContent = segs.findLast((s: MessageSegment) => s.type === 'content') contentSeg = { id: genSegId(), type: 'content', status: 'running', text: '', timestamp: Date.now() } applyIterationTags(contentSeg) - // Later content supersedes any earlier provisional narration of the - // same turn — collapse it in place, live, instead of waiting for the - // persisted-metadata annotations in the done payload. Mirrors the - // backend tracker's rule for kind-tagged segments. - for (const s of segs) { - if (s.type === 'content' && s.kind === 'pre_tool_narration' && !s.superseded) { - s.superseded = true - s.supersededBySegmentId = String(contentSeg.id) - s.supersededReason = 'pre_tool_content_replaced_by_post_tool_answer' - } + segmentObservationMark.set(String(contentSeg.id), observationCount) + // Later content supersedes an earlier provisional narration — collapse + // it in place, live, instead of waiting for the persisted-metadata + // annotations in the done payload. Rule and guards live in + // supersede.ts, mirroring the backend tracker. + const prevMark = prevContent ? (segmentObservationMark.get(String(prevContent.id)) ?? 0) : 0 + if (prevContent && supersedesProvisionalNarration(prevContent, observationCount, prevMark)) { + markSuperseded(prevContent, String(contentSeg.id)) } segs.push(contentSeg) flushSegmentsToMessage() // sync once when a new content segment is created @@ -937,6 +951,11 @@ export function useChat(options: UseChatOptions): UseChatReturn { // Body of tool_call_completed — see handleToolCallStarted. function handleToolCallCompleted(data: any) { if (isStaleEvent(data)) return + // Counted regardless of success: a failed tool still produces an + // observation, and the content written after it is grounded in that + // failure. Counted before the segment work below so a content span opened + // later in this turn sees the higher mark. + observationCount++ if (currentAssistantId.value) { const msg = getMessage(currentAssistantId.value) if (msg) {