diff --git a/mateclaw-server/src/main/java/vip/mate/agent/AgentGraphBuilder.java b/mateclaw-server/src/main/java/vip/mate/agent/AgentGraphBuilder.java index 7bd8c9a8..635dac64 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/AgentGraphBuilder.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/AgentGraphBuilder.java @@ -554,7 +554,7 @@ public class AgentGraphBuilder { if (auditEventService != null) { executor.setAuditEventService(auditEventService); } - PlanGenerationNode planGenerationNode = new PlanGenerationNode(chatModel, planningService, streamingHelper, conversationWindowManager, toolSet); + PlanGenerationNode planGenerationNode = new PlanGenerationNode(chatModel, planningService, streamingHelper, conversationWindowManager, toolSet, goalService, goalProperties); StepExecutionNode stepExecutionNode = new StepExecutionNode(chatModel, toolSet, executor, planningService, streamTracker, reasoningEffort, streamingHelper, conversationWindowManager, skillCatalogRenderer); PlanSummaryNode planSummaryNode = new PlanSummaryNode(chatModel, planningService, streamingHelper); DirectAnswerNode directAnswerNode = new DirectAnswerNode(); @@ -579,6 +579,7 @@ public class AgentGraphBuilder { .addStrategy(PlanStateKeys.CURRENT_STEP_TITLE, KeyStrategy.REPLACE) .addStrategy(PlanStateKeys.CURRENT_STEP_RESULT, KeyStrategy.REPLACE) .addStrategy(PlanStateKeys.COMPLETED_RESULTS, KeyStrategy.APPEND) + .addStrategy(PlanStateKeys.PLAN_REPLAN_COUNT, KeyStrategy.REPLACE) .addStrategy(PlanStateKeys.FINAL_SUMMARY, KeyStrategy.REPLACE) .addStrategy(PlanStateKeys.DIRECT_ANSWER, KeyStrategy.REPLACE) // 工作上下文(REPLACE 策略,每次重新生成) @@ -643,6 +644,7 @@ public class AgentGraphBuilder { .addStrategy(MateClawStateKeys.GOAL_EVALUATED_THIS_RUN, KeyStrategy.REPLACE) .addStrategy(MateClawStateKeys.GOAL_FOLLOWUP_COUNT, KeyStrategy.REPLACE) .addStrategy(MateClawStateKeys.GOAL_ACCOUNTED_LLM_CALL_COUNT, KeyStrategy.REPLACE) + .addStrategy(MateClawStateKeys.GOAL_HARD_CONTINUATION_COUNT, KeyStrategy.REPLACE) // Skill progressive disclosure — pinned skills loaded this // run. Registered in BOTH graphs so the read-merge-write in // ActionNode is not dropped on multi-node merges. @@ -657,6 +659,7 @@ public class AgentGraphBuilder { // ├→ DIRECT_ANSWER_NODE → END // └→ STEP_EXECUTION → (StepProgressDispatcher) // ├→ STEP_EXECUTION (loop) + // ├→ PLAN_GENERATION (re-plan on step failure, bounded by PLAN_REPLAN_COUNT) // └→ PLAN_SUMMARY → (active goal?) // ├→ GOAL_EVALUATION → (followup?) // │ ├→ PLAN_GENERATION (re-plan) @@ -690,11 +693,18 @@ public class AgentGraphBuilder { Map.of( PlanStateKeys.STEP_EXECUTION_NODE, PlanStateKeys.STEP_EXECUTION_NODE, PlanStateKeys.PLAN_SUMMARY_NODE, PlanStateKeys.PLAN_SUMMARY_NODE, + // Step-failure recovery: re-plan the remaining work + // (StepProgressDispatcher returns this on phase=plan_replan). + PlanStateKeys.PLAN_GENERATION_NODE, PlanStateKeys.PLAN_GENERATION_NODE, StateGraph.END, StateGraph.END)) .addConditionalEdges(PlanStateKeys.PLAN_SUMMARY_NODE, AsyncEdgeAction.edge_async(state -> { MateClawStateAccessor a = new MateClawStateAccessor(state); - boolean hasGoal = a.hasActiveGoal(); + // Same-turn activation: fall back to a DB lookup (gated on the + // feature flag) so a goal the agent set THIS turn is evaluated now, + // not only from the next message. See GoalEvaluationNode.resolveActiveGoal. + boolean hasGoal = goalProperties.isEnabled() + && GoalEvaluationNode.resolveActiveGoal(state, goalService).isPresent(); boolean already = a.goalEvaluatedThisRun(); return (hasGoal && !already) ? MateClawStateKeys.GOAL_EVALUATION_NODE @@ -719,7 +729,11 @@ public class AgentGraphBuilder { .addConditionalEdges(PlanStateKeys.DIRECT_ANSWER_NODE, AsyncEdgeAction.edge_async(state -> { MateClawStateAccessor a = new MateClawStateAccessor(state); - boolean hasGoal = a.hasActiveGoal(); + // Same-turn activation: fall back to a DB lookup (gated on the + // feature flag) so a goal the agent set THIS turn is evaluated now, + // not only from the next message. See GoalEvaluationNode.resolveActiveGoal. + boolean hasGoal = goalProperties.isEnabled() + && GoalEvaluationNode.resolveActiveGoal(state, goalService).isPresent(); boolean already = a.goalEvaluatedThisRun(); return (hasGoal && !already) ? MateClawStateKeys.GOAL_EVALUATION_NODE @@ -756,9 +770,17 @@ public class AgentGraphBuilder { * and tool-result chunking. Decoupled from the per-agent value so a small * {@code max_iterations} can never accidentally re-introduce the silent * killer. + *

+ * The base segment budget is further multiplied to cover goal-driven "hard + * continuations" — each grants a fresh full iteration budget after a + * max-iterations turn (see {@code GoalEvaluationNode}). One run can perform + * up to {@link vip.mate.goal.config.GoalProperties#MAX_HARD_CONTINUATIONS_CEILING} of them, so + * the ceiling is sized for {@code (1 + CEILING)} segments to keep the + * recursion guard from tripping before the soft caps do. */ private static int frameworkRecursionLimit() { - return (BaseAgent.MAX_ITERATIONS_HARD_CEILING + 5) * 4 + 100; + int perSegment = (BaseAgent.MAX_ITERATIONS_HARD_CEILING + 5) * 4; + return perSegment * (1 + vip.mate.goal.config.GoalProperties.MAX_HARD_CONTINUATIONS_CEILING) + 100; } CompiledGraph buildReActGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations, String reasoningEffort) { @@ -826,6 +848,7 @@ public class AgentGraphBuilder { .addStrategy(MateClawStateKeys.MESSAGES, KeyStrategy.APPEND) // 迭代控制 .addStrategy(MateClawStateKeys.CURRENT_ITERATION, KeyStrategy.REPLACE) + .addStrategy(MateClawStateKeys.ITERATION_REFUND_COUNT, KeyStrategy.REPLACE) .addStrategy(MateClawStateKeys.MAX_ITERATIONS, KeyStrategy.REPLACE) // 工具调用 .addStrategy(MateClawStateKeys.TOOL_CALLS, KeyStrategy.REPLACE) @@ -907,6 +930,7 @@ public class AgentGraphBuilder { .addStrategy(MateClawStateKeys.GOAL_EVALUATED_THIS_RUN, KeyStrategy.REPLACE) .addStrategy(MateClawStateKeys.GOAL_FOLLOWUP_COUNT, KeyStrategy.REPLACE) .addStrategy(MateClawStateKeys.GOAL_ACCOUNTED_LLM_CALL_COUNT, KeyStrategy.REPLACE) + .addStrategy(MateClawStateKeys.GOAL_HARD_CONTINUATION_COUNT, KeyStrategy.REPLACE) // Skill progressive disclosure — pinned skills loaded this // run. Registered in BOTH graphs so the read-merge-write in // ActionNode is not dropped on multi-node merges. @@ -956,7 +980,11 @@ public class AgentGraphBuilder { .addConditionalEdges(MateClawStateKeys.FINAL_ANSWER_NODE, AsyncEdgeAction.edge_async(state -> { MateClawStateAccessor a = new MateClawStateAccessor(state); - boolean hasGoal = a.hasActiveGoal(); + // Same-turn activation: fall back to a DB lookup (gated on the + // feature flag) so a goal the agent set THIS turn is evaluated now, + // not only from the next message. See GoalEvaluationNode.resolveActiveGoal. + boolean hasGoal = goalProperties.isEnabled() + && GoalEvaluationNode.resolveActiveGoal(state, goalService).isPresent(); boolean already = a.goalEvaluatedThisRun(); return (hasGoal && !already) ? MateClawStateKeys.GOAL_EVALUATION_NODE diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/NodeStreamingChatHelper.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/NodeStreamingChatHelper.java index d1d35757..b378c137 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/NodeStreamingChatHelper.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/NodeStreamingChatHelper.java @@ -362,9 +362,42 @@ public class NodeStreamingChatHelper { // Hard time budget for the primary retry loop (3 min). Prevents // retries from stalling a single conversation turn indefinitely. // Aligned with WikiProcessingService.llmMaxTotalDurationMs. - private static final long MAX_TOTAL_DURATION_MS = 3 * 60 * 1000L; - private static final long BACKOFF_BASE_MS = 3000; - private static final long BACKOFF_CAP_MS = 60_000; + // + // Because the backoff grows exponentially (3s, 6s, 12s, 24s, 48s, then + // capped at 60s), this wall-clock budget — not MAX_RETRIES — is what + // actually bounds a sustained SERVER_ERROR loop: only ~8 of the 10 + // retries fit inside 3 minutes before the elapsed-time check in + // streamCallInternal breaks to the fallback chain. + // + // These three values are instance fields seeded from the DEFAULT_* + // constants (rather than compile-time constants) so tests can shrink + // them to exercise the full retry path in milliseconds instead of + // minutes. Production wiring never overrides them — see + // setRetryTimingForTest. + private static final long DEFAULT_MAX_TOTAL_DURATION_MS = 3 * 60 * 1000L; + private static final long DEFAULT_BACKOFF_BASE_MS = 3000; + private static final long DEFAULT_BACKOFF_CAP_MS = 60_000; + + private long maxTotalDurationMs = DEFAULT_MAX_TOTAL_DURATION_MS; + private long backoffBaseMs = DEFAULT_BACKOFF_BASE_MS; + private long backoffCapMs = DEFAULT_BACKOFF_CAP_MS; + + /** + * Test-only seam to shrink the retry backoff and total-time budget so the + * full {@link #MAX_RETRIES} path (or the time-budget cut-off) can be + * exercised in milliseconds instead of minutes. Package-private and never + * invoked from production wiring, which always keeps the {@code DEFAULT_*} + * timings. + * + * @param backoffBaseMs base backoff for the first retry (doubles each attempt) + * @param backoffCapMs per-attempt backoff ceiling + * @param maxTotalDurationMs hard wall-clock budget for the whole primary retry loop + */ + void setRetryTimingForTest(long backoffBaseMs, long backoffCapMs, long maxTotalDurationMs) { + this.backoffBaseMs = backoffBaseMs; + this.backoffCapMs = backoffCapMs; + this.maxTotalDurationMs = maxTotalDurationMs; + } private static final ObjectMapper TOOL_ARG_JSON_MAPPER = new ObjectMapper(); @@ -592,7 +625,7 @@ public class NodeStreamingChatHelper { // conversation turn indefinitely (e.g., a provider that stays // at 503 for minutes). Aligned with Wiki's maxTotalDurationMs. long elapsedMs = System.currentTimeMillis() - callStartMs; - if (elapsedMs >= MAX_TOTAL_DURATION_MS) { + if (elapsedMs >= maxTotalDurationMs) { log.warn("[{}] Primary retry time budget exhausted ({}ms), handing off to fallback chain", phase, elapsedMs); break; @@ -875,10 +908,10 @@ public class NodeStreamingChatHelper { String conversationId, String phase, boolean broadcast, int attempt) { if (attempt > 0) { - long delay = Math.min(BACKOFF_BASE_MS * (1L << (attempt - 1)), BACKOFF_CAP_MS); + long delay = Math.min(backoffBaseMs * (1L << (attempt - 1)), backoffCapMs); // 加入 jitter 防止雷群效应 delay += ThreadLocalRandom.current().nextLong(0, Math.max(1, delay / 2)); - delay = Math.min(delay, BACKOFF_CAP_MS); + delay = Math.min(delay, backoffCapMs); log.warn("[{}] Retry attempt {}/{} after {}ms for conversation {}", phase, attempt, MAX_RETRIES, delay, conversationId); // 广播给前端:用户可见的重试倒计时 diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/node/GoalEvaluationNode.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/node/GoalEvaluationNode.java index 2f98eb9c..042b35fc 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/node/GoalEvaluationNode.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/node/GoalEvaluationNode.java @@ -78,7 +78,12 @@ public class GoalEvaluationNode implements NodeAction { MateClawStateAccessor accessor = new MateClawStateAccessor(state); - Optional goalOpt = accessor.activeGoal(); + // Resolve the active goal from the turn-start snapshot, falling back to + // a conversation lookup. The fallback is what makes a goal created + // MID-TURN (the agent calls setGoal, which only writes the DB) get + // evaluated on the very turn it was set — otherwise ACTIVE_GOAL is empty + // in this run's state and the goal would sit inert until the next message. + Optional goalOpt = resolveActiveGoal(state, goalService); if (goalOpt.isEmpty()) { return Map.of(); } @@ -94,26 +99,32 @@ public class GoalEvaluationNode implements NodeAction { // chat composable's `message_complete` handler optimistically sets // evaluating=true; without a balancing event the ring would stay // in that state forever after e.g. a max-iterations turn. - Long goalIdForEvents = (goalOpt.get() instanceof GoalEntity ge) ? ge.getId() : null; + Long goalIdForEvents = goalOpt.get().getId(); // ReAct path: FinalAnswerNode wrote a canonical finishReason that // determines whether this turn counts. Plan-Execute usually doesn't // set finishReason on the happy path, so we only enforce these // exit conditions in REACT mode + the universal awaiting_approval // gate that both flavors share. + // A turn that hit the ReAct iteration cap. Continuing it needs a FRESH + // iteration budget (a "hard continuation"), handled in the follow-up + // branch below; capture it here while finishReason is still authoritative. + boolean reactIterationCapReached = flavor == GraphFlavor.REACT + && isIterationCapReached(accessor.finishReason()); + if (flavor == GraphFlavor.REACT) { String fr = accessor.finishReason(); - if (FinishReason.EVIDENCE_INSUFFICIENT.getValue().equals(fr) - || FinishReason.STOPPED.getValue().equals(fr) - || FinishReason.ERROR_FALLBACK.getValue().equals(fr) - || FinishReason.RETURN_DIRECT.getValue().equals(fr) - || FinishReason.MAX_ITERATIONS_REACHED.getValue().equals(fr)) { + if (isHardSkipFinishReason(fr)) { log.debug("[GoalEvaluationNode] skipping evaluation (REACT finishReason={})", fr); return MateClawStateAccessor.output() .goalEvaluatedThisRun(true) .events(List.of(skippedEvent(goalIdForEvents, "react_finish_reason:" + fr))) .build(); } + // MAX_ITERATIONS_REACHED and EVIDENCE_INSUFFICIENT intentionally + // fall through: both mean "answer produced but the goal is likely + // unmet", which is exactly when a corrective follow-up helps. + // Max-iterations additionally needs a fresh budget (see below). } if (accessor.awaitingApproval()) { return MateClawStateAccessor.output() @@ -122,14 +133,7 @@ public class GoalEvaluationNode implements NodeAction { .build(); } - Object goalObj = goalOpt.get(); - if (!(goalObj instanceof GoalEntity goal)) { - log.warn("[GoalEvaluationNode] ACTIVE_GOAL is not a GoalEntity: {}", goalObj.getClass()); - return MateClawStateAccessor.output() - .goalEvaluatedThisRun(true) - .events(List.of(skippedEvent(null, "non_goal_entity"))) - .build(); - } + GoalEntity goal = goalOpt.get(); String terminal = accessor.terminalAnswer(); if (terminal.isEmpty()) { @@ -226,6 +230,9 @@ public class GoalEvaluationNode implements NodeAction { } int followupCountThisRun = accessor.goalFollowupCount(); + int hardContinuationCount = accessor.goalHardContinuationCount(); + int hardCap = Math.min(properties.getMaxHardContinuationsPerRun(), + GoalProperties.MAX_HARD_CONTINUATIONS_CEILING); Optional followup; try { followup = followupService.maybeBuildFollowup(refreshed, result); @@ -241,11 +248,19 @@ public class GoalEvaluationNode implements NodeAction { // active and the cross-message turn / LLM budget (or the user) carries // it on. boolean perRunCapReached = followupCountThisRun >= properties.getMaxFollowupsPerRun(); - if (followup.isPresent() && perRunCapReached) { - log.info("[GoalEvaluationNode] per-run followup cap reached ({}/{}) for goal={}; ending this run", - followupCountThisRun, properties.getMaxFollowupsPerRun(), refreshed.getId()); + // A max-iterations continuation re-runs a FULL fresh ReAct segment + // (iteration budget reset), so it carries a tighter, dedicated cap on + // top of the per-run follow-up cap — and is sized into the graph + // recursion ceiling. hardCap==0 keeps the legacy behaviour (a + // max-iterations turn simply ends the run). + boolean hardCapReached = reactIterationCapReached && hardContinuationCount >= hardCap; + if (followup.isPresent() && (perRunCapReached || hardCapReached)) { + log.info("[GoalEvaluationNode] follow-up suppressed for goal={} " + + "(followups {}/{}, hardContinuations {}/{}, iterationCapReached={}); ending this run", + refreshed.getId(), followupCountThisRun, properties.getMaxFollowupsPerRun(), + hardContinuationCount, hardCap, reactIterationCapReached); } - if (followup.isPresent() && !perRunCapReached) { + if (followup.isPresent() && !perRunCapReached && !hardCapReached) { try { goalService.recordFollowupInjected(refreshed.getId(), followup.get()); } catch (Throwable t) { @@ -283,6 +298,20 @@ public class GoalEvaluationNode implements NodeAction { out.clearFinalAnswer() .clearFinishReason() .messages(List.of((Message) new UserMessage(followup.get()))); + if (reactIterationCapReached) { + // Hard continuation: the run's iteration budget is spent, so + // grant a brand-new ReAct segment. Reset the counter, clear + // the stale limit-exceeded draft/flag (FinalAnswerNode prefers + // the draft over a freshly reasoned answer) and any latched + // error, and advance the dedicated hard-continuation counter. + out.iterationCount(0) + .clearLimitExceededDraft() + .error("") + .goalHardContinuationCount(hardContinuationCount + 1); + log.info("[GoalEvaluationNode] hard continuation {}/{} for goal={} " + + "(fresh ReAct iteration budget after max-iterations turn)", + hardContinuationCount + 1, hardCap, refreshed.getId()); + } } else { // Plan-Execute: wipe the wider mid-pass + terminal state. // WORKING_CONTEXT and PlanStateKeys.GOAL are intentionally @@ -318,6 +347,70 @@ public class GoalEvaluationNode implements NodeAction { .build(); } + /** + * Resolve the active goal for this run: prefer the turn-start + * {@code ACTIVE_GOAL} snapshot; if absent, fall back to a conversation + * lookup so a goal created mid-turn (via {@code setGoal}, which only writes + * the DB) is still evaluated on the turn it was set. + * + *

Shared by {@link #apply} and the {@code FinalAnswer/PlanSummary → + * GoalEvaluation} routing edges so both agree on whether a goal is active. + * The DB fallback costs one indexed lookup per terminal turn whose snapshot + * is empty; callers should additionally gate on {@code properties.isEnabled()} + * to skip it when the feature is off. + */ + public static Optional resolveActiveGoal(OverAllState state, GoalService goalService) { + MateClawStateAccessor a = new MateClawStateAccessor(state); + Optional snapshot = a.activeGoal(); + if (snapshot.isPresent() && snapshot.get() instanceof GoalEntity ge) { + return Optional.of(ge); + } + if (goalService == null) { + return Optional.empty(); + } + String conversationId = a.conversationId(); + if (conversationId == null || conversationId.isBlank()) { + return Optional.empty(); + } + try { + return Optional.ofNullable(goalService.findActiveByConversation(conversationId)); + } catch (Throwable t) { + log.warn("[GoalEvaluationNode] active-goal fallback lookup failed for conversation={}: {}", + conversationId, t.toString()); + return Optional.empty(); + } + } + + /** + * REACT-mode finish reasons that should neither count toward the goal nor + * trigger a continuation: + *
    + *
  • {@code STOPPED} — the user halted the run; don't fight them.
  • + *
  • {@code RETURN_DIRECT} — a tool produced the answer verbatim; this + * is not goal-progress reasoning work to evaluate or continue.
  • + *
  • {@code ERROR_FALLBACK} — a fatal error already failed the turn; + * re-running immediately would just re-fail.
  • + *
+ * Other terminal reasons — notably {@code MAX_ITERATIONS_REACHED} and + * {@code EVIDENCE_INSUFFICIENT} — mean "answer produced but the goal is + * likely unmet", which is exactly when a corrective follow-up helps, so + * they are deliberately NOT skipped. + */ + static boolean isHardSkipFinishReason(String finishReason) { + return FinishReason.STOPPED.getValue().equals(finishReason) + || FinishReason.RETURN_DIRECT.getValue().equals(finishReason) + || FinishReason.ERROR_FALLBACK.getValue().equals(finishReason); + } + + /** + * True when the terminal turn hit the ReAct iteration cap. Continuing such + * a turn requires a fresh iteration budget (a "hard continuation"), because + * the run's shared budget is already exhausted. + */ + static boolean isIterationCapReached(String finishReason) { + return FinishReason.MAX_ITERATIONS_REACHED.getValue().equals(finishReason); + } + /** Stand-in for a missing {@code GraphEventPublisher.custom()} factory. */ private static GraphEventPublisher.GraphEvent goalEvent(String type, Map data) { return new GraphEventPublisher.GraphEvent(type, Map.copyOf(data), System.currentTimeMillis()); diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/node/ObservationNode.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/node/ObservationNode.java index 9755ce93..708670a2 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/node/ObservationNode.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/node/ObservationNode.java @@ -32,6 +32,18 @@ public class ObservationNode implements NodeAction { private final ObservationProcessor observationProcessor; private final vip.mate.channel.web.ChatStreamTracker streamTracker; + /** + * Progressive-disclosure meta-tools that perform setup, not real work. A + * round whose entire batch is one of these is refunded its iteration (see + * {@link MateClawStateKeys#ITERATION_REFUND_COUNT}). Mirrors the authoritative + * set in {@code DefaultToolDisclosureService.ALWAYS_CORE}. + */ + private static final java.util.Set DISCLOSURE_TOOLS = + java.util.Set.of("load_skill", "enable_tool"); + + /** Per-run cap on iteration refunds — keeps a load-skill-only model from looping forever. */ + private static final int MAX_ITERATION_REFUNDS_PER_RUN = 3; + public ObservationNode(ObservationProcessor observationProcessor) { this(observationProcessor, null); } @@ -56,14 +68,29 @@ public class ObservationNode implements NodeAction { int currentIteration = accessor.iterationCount(); int maxIterations = accessor.maxIterations(); - int nextIteration = currentIteration + 1; - - log.info("[ObservationNode] Iteration {}/{}", nextIteration, maxIterations); // 提取最新的工具结果并处理 List toolResults = state.>value(TOOL_RESULTS).orElse(List.of()); + // Iteration refund: a round whose entire batch was progressive-disclosure + // setup (load_skill / enable_tool) did no real work, so don't charge it an + // iteration — otherwise a tight budget loses a step to the load-then-use + // two-step. Bounded by MAX_ITERATION_REFUNDS_PER_RUN so a model that only + // ever loads skills can't dodge the budget forever. + int refundCount = accessor.iterationRefundCount(); + boolean setupOnlyRound = !toolResults.isEmpty() + && toolResults.stream().allMatch(tr -> DISCLOSURE_TOOLS.contains(tr.name())); + boolean refundIteration = setupOnlyRound && refundCount < MAX_ITERATION_REFUNDS_PER_RUN; + int nextIteration = refundIteration ? currentIteration : currentIteration + 1; + + if (refundIteration) { + log.info("[ObservationNode] Iteration refunded (setup-only round, refunds {}/{}); staying at {}/{}", + refundCount + 1, MAX_ITERATION_REFUNDS_PER_RUN, nextIteration, maxIterations); + } else { + log.info("[ObservationNode] Iteration {}/{}", nextIteration, maxIterations); + } + // 将每个工具结果通过 ObservationProcessor 标准化和截断 List processedObservations = toolResults.stream() .map(tr -> observationProcessor.process(tr.name(), tr.responseData())) @@ -121,6 +148,10 @@ public class ObservationNode implements NodeAction { .shouldSummarize(shouldSummarize) .toolCallCount(newToolCallCount); + if (refundIteration) { + builder.iterationRefundCount(refundCount + 1); + } + // Close out the iteration we just observed. We use currentIteration // (not nextIteration) so the index pairs with whatever // iteration_start the ReasoningNode emitted at the top of this turn. diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcher.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcher.java index cb573ae0..0d15a188 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcher.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcher.java @@ -30,6 +30,12 @@ public class StepProgressDispatcher implements EdgeAction { if ("awaiting_approval".equals(currentPhase) || "plan_aborted".equals(currentPhase)) { return StateGraph.END; } + // Step-failure recovery: a failed step requested a re-plan of the + // remaining work. Route back to PlanGeneration instead of aborting; + // PLAN_REPLAN_COUNT (set by StepExecutionNode) bounds the loop. + if ("plan_replan".equals(currentPhase)) { + return PlanStateKeys.PLAN_GENERATION_NODE; + } int currentIndex = state.value(PlanStateKeys.CURRENT_STEP_INDEX, 0); List steps = state.>value(PlanStateKeys.PLAN_STEPS).orElse(List.of()); diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/PlanGenerationNode.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/PlanGenerationNode.java index a2eeb074..3d9facfc 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/PlanGenerationNode.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/PlanGenerationNode.java @@ -16,8 +16,14 @@ import vip.mate.agent.graph.NodeStreamingChatHelper; import vip.mate.agent.graph.plan.state.PlanStateAccessor; import vip.mate.agent.graph.plan.state.PlanStateKeys; import vip.mate.agent.graph.state.MateClawStateKeys; +import vip.mate.agent.context.ChatOrigin; import vip.mate.agent.context.ConversationWindowManager; import vip.mate.agent.context.RuntimeContextInjector; +import vip.mate.goal.config.GoalProperties; +import vip.mate.goal.model.GoalCreateRequest; +import vip.mate.goal.model.GoalCriterion; +import vip.mate.goal.model.GoalEntity; +import vip.mate.goal.service.GoalService; import vip.mate.planning.service.PlanningService; import java.util.ArrayList; @@ -52,6 +58,14 @@ public class PlanGenerationNode implements NodeAction { private final NodeStreamingChatHelper streamingHelper; private final ConversationWindowManager conversationWindowManager; private final AgentToolSet toolSet; + /** Optional — auto-derive a goal from the plan. Null disables the feature (legacy/test). */ + private final GoalService goalService; + private final GoalProperties goalProperties; + + /** Plan steps below this size are trivial tool tasks, not goal-worthy. */ + private static final int MIN_STEPS_FOR_AUTO_GOAL = 2; + /** Cap the auto-derived goal title; the full request rides in the description. */ + private static final int AUTO_GOAL_TITLE_MAX = 80; /** * Structured triage result — field names use @JsonProperty to match the @@ -168,11 +182,21 @@ public class PlanGenerationNode implements NodeAction { NodeStreamingChatHelper streamingHelper, ConversationWindowManager conversationWindowManager, AgentToolSet toolSet) { + this(chatModel, planningService, streamingHelper, conversationWindowManager, toolSet, null, null); + } + + public PlanGenerationNode(ChatModel chatModel, PlanningService planningService, + NodeStreamingChatHelper streamingHelper, + ConversationWindowManager conversationWindowManager, + AgentToolSet toolSet, + GoalService goalService, GoalProperties goalProperties) { this.chatModel = chatModel; this.planningService = planningService; this.streamingHelper = streamingHelper; this.conversationWindowManager = conversationWindowManager; this.toolSet = toolSet; + this.goalService = goalService; + this.goalProperties = goalProperties; } /** @@ -180,7 +204,65 @@ public class PlanGenerationNode implements NodeAction { */ @Deprecated public PlanGenerationNode(ChatModel chatModel, PlanningService planningService) { - this(chatModel, planningService, null, null, null); + this(chatModel, planningService, null, null, null, null, null); + } + + /** + * Auto-derive a goal from a freshly-generated multi-step plan so the + * Plan-Execute path engages the goal subsystem (the planner / step executor + * never call {@code setGoal} themselves). The plan steps become the goal's + * acceptance criteria — the plan IS the decomposition — so the first + * evaluation skips the bootstrap round and judges those criteria directly. + * + *

Returns the created goal (to inject into {@code ACTIVE_GOAL} so THIS + * run's GoalEvaluationNode picks it up) or {@code null} when not applicable: + * feature off, fewer than {@link #MIN_STEPS_FOR_AUTO_GOAL} steps, no channel + * context, or the conversation already has an active goal. Best-effort — + * any failure is swallowed so planning is never blocked by goal bookkeeping. + */ + GoalEntity maybeAutoCreateGoal(PlanStateAccessor accessor, List steps) { + if (goalService == null || goalProperties == null + || !goalProperties.isEnabled() || !goalProperties.isAutoGoalFromPlan()) { + return null; + } + if (steps == null || steps.size() < MIN_STEPS_FOR_AUTO_GOAL) { + return null; + } + ChatOrigin origin = accessor.chatOrigin(); + String convId = origin.conversationId(); + if (convId == null || convId.isBlank() || origin.agentId() == null) { + return null; + } + try { + if (goalService.findActiveByConversation(convId) != null) { + return null; // respect an existing goal (incl. re-plan passes) + } + String request = stripInjectedContext(accessor.goal()).strip(); + GoalCreateRequest req = new GoalCreateRequest(); + req.setConversationId(convId); + req.setAgentId(origin.agentId()); + req.setWorkspaceId(origin.workspaceId() != null ? origin.workspaceId() : 1L); + req.setTitle(request.isEmpty() ? "多步任务" + : request.length() > AUTO_GOAL_TITLE_MAX + ? request.substring(0, AUTO_GOAL_TITLE_MAX) : request); + req.setDescription(request); + List criteria = steps.stream() + .filter(s -> s != null && !s.isBlank()) + .map(s -> new GoalCriterion("", s.strip(), false, "")) + .collect(Collectors.toList()); + if (!criteria.isEmpty()) { + req.setCriteria(criteria); + } + String username = origin.requesterId() != null && !origin.requesterId().isBlank() + ? origin.requesterId() : "system"; + GoalEntity created = goalService.create(req, username); + log.info("[PlanGeneration] Auto-derived goal {} from plan ({} criteria) for conversation {}", + created.getId(), criteria.size(), convId); + return created; + } catch (Exception e) { + log.warn("[PlanGeneration] Auto-goal-from-plan skipped (non-fatal): {}", e.toString()); + return null; + } } @Override @@ -375,7 +457,22 @@ public class PlanGenerationNode implements NodeAction { events.add(GraphEventPublisher.planCreated(plan.getId(), steps)); - return PlanStateAccessor.output() + // Auto-derive a goal from a genuine multi-step plan so the + // Plan-Execute path engages the goal subsystem. Injected into + // ACTIVE_GOAL so this same run's GoalEvaluationNode evaluates it. + GoalEntity autoGoal = maybeAutoCreateGoal(accessor, steps); + if (autoGoal != null && goalService != null) { + // Surface it to the UI exactly like the setGoal tool does + // ({goalId, conversationId, goal}) so the goal panel hydrates + // even though the user never called setGoal. Same SSE event the + // frontend goal store already listens for. + events.add(new GraphEventPublisher.GraphEvent("goal_created", Map.of( + "goalId", String.valueOf(autoGoal.getId()), + "conversationId", conversationId, + "goal", goalService.toResponse(autoGoal)), System.currentTimeMillis())); + } + + PlanStateAccessor.OutputBuilder planOut = PlanStateAccessor.output() .needsPlanning(true) .planId(plan.getId()) .planSteps(steps) @@ -385,8 +482,11 @@ public class PlanGenerationNode implements NodeAction { .contentStreamed(true) .thinkingStreamed(!result.thinking().isEmpty()) .mergeUsage(state, result) - .events(events) - .build(); + .events(events); + if (autoGoal != null) { + planOut.put(MateClawStateKeys.ACTIVE_GOAL, autoGoal); + } + return planOut.build(); } catch (Exception e) { log.error("[PlanGeneration] Triage failed, falling back to single-step plan: {}", e.getMessage(), e); diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepExecutionNode.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepExecutionNode.java index 5097e880..3f306efb 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepExecutionNode.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepExecutionNode.java @@ -89,6 +89,16 @@ public class StepExecutionNode implements NodeAction { * pathological cases where the agent appears frozen to the user. */ private static final long STEP_WALL_CLOCK_TIMEOUT_MS = 10 * 60 * 1000L; + + /** + * Max re-plans per graph run. When a step throws, the executor re-plans the + * remaining work around the failure instead of aborting the whole plan — a + * single transient tool error or one badly-scoped step no longer kills the + * task. Bounded so a step that fails every attempt can't re-plan forever; + * once exhausted the plan aborts as before. Kept small (the recursion + * ceiling already accommodates it) — raise with care. + */ + private static final int MAX_REPLANS_PER_RUN = 1; private static final ObjectMapper MAPPER = new ObjectMapper(); public StepExecutionNode(ChatModel chatModel, AgentToolSet toolSet, @@ -222,6 +232,12 @@ public class StepExecutionNode implements NodeAction { long stepStartedAtMs = System.currentTimeMillis(); boolean wallClockExceeded = false; + // Signature-based progress detector: nudges the model to change strategy + // when a round stalls (repeated failures / identical results), and flags + // the step as stuck past a hard threshold so we re-plan instead of + // burning the whole tool-call budget and advancing with junk. + StepProgressTracker progressTracker = new StepProgressTracker(); + try { while (toolCallCount < MAX_TOOL_CALLS_PER_STEP) { long elapsedMs = System.currentTimeMillis() - stepStartedAtMs; @@ -353,6 +369,30 @@ public class StepExecutionNode implements NodeAction { break; } + // Progress tracking: feed this round's tool results to the + // detector. A WARN-level stall injects a one-shot "change + // strategy" SystemMessage the model sees on its next call; a + // HALT-level stall stops the inner loop so the post-loop logic + // re-plans instead of spinning to the tool-call ceiling. + java.util.Map idToArgs = new java.util.HashMap<>(); + for (AssistantMessage.ToolCall tc : allToolCalls) { + if (tc != null && tc.id() != null) { + idToArgs.put(tc.id(), tc.arguments()); + } + } + for (ToolResponseMessage.ToolResponse tr : toolResponses) { + var nudge = progressTracker.record( + tr.name(), idToArgs.getOrDefault(tr.id(), ""), tr.responseData()); + if (nudge.isPresent()) { + messages.add(new SystemMessage(nudge.get())); + } + } + if (progressTracker.isStuck()) { + log.warn("[StepExecution] Step {} stalled ({}); stopping inner loop to re-plan", + stepIndex, progressTracker.haltReason()); + break; + } + // RFC-052: returnDirect short-circuit. Any direct tool in this // step ends the plan immediately; the dispatcher routes via // currentPhase=plan_aborted so no further LLM call happens. @@ -418,6 +458,56 @@ public class StepExecutionNode implements NodeAction { .build(); } + // Step-failure recovery WITHOUT an exception: the inner loop ended + // with no usable result — stalled (repeated failures / identical + // results, flagged by the progress tracker), hit the wall-clock or + // tool-call ceiling, or returned an empty answer. Re-plan the + // remaining work around it instead of advancing dependent steps with + // junk. Shares PLAN_REPLAN_COUNT with the exception path; once the + // budget is spent we fall through to the legacy "complete with a + // failure note" path below so the plan still terminates. + boolean noUsableResult = progressTracker.isStuck() + || finalResult == null || finalResult.isBlank(); + int noProgressReplanCount = accessor.replanCount(); + if (noUsableResult && noProgressReplanCount < MAX_REPLANS_PER_RUN) { + String reason = progressTracker.isStuck() + ? "本步骤陷入停滞(" + progressTracker.haltReason() + "),未取得有效结果" + : wallClockExceeded + ? "本步骤超过最大耗时限制,未取得有效结果" + : finalResult == null + ? "本步骤超过最大工具调用次数,未取得有效结果" + : "本步骤未产出有效结果"; + planningService.updateSubPlanFailure(planId, stepIndex, reason); + planningService.markPlanFailed(planId, "步骤" + (stepIndex + 1) + ":" + reason); + events.add(GraphEventPublisher.stepCompleted(stepIndex, reason)); + if (iterationEventsOn) { + events.add(GraphEventPublisher.iterationEnd(stepIndex, "parent", null, reason.length(), 0)); + } + events.add(new GraphEventPublisher.GraphEvent("plan_replan", Map.of( + "failedStepIndex", stepIndex, + "attempt", noProgressReplanCount + 1, + "maxReplans", MAX_REPLANS_PER_RUN, + "reason", reason), System.currentTimeMillis())); + log.warn("[StepExecution] Step {} produced no usable result ({}); re-planning (attempt {}/{})", + stepIndex + 1, reason, noProgressReplanCount + 1, MAX_REPLANS_PER_RUN); + return PlanStateAccessor.output() + .workingContext(buildReplanContext(accessor, stepIndex, reason)) + .currentPhase("plan_replan") + .replanCount(noProgressReplanCount + 1) + .planId(null) + .planSteps(List.of()) + .planValid(false) + .needsPlanning(true) + .currentStepIndex(0) + .currentStepTitle("") + .currentStepResult("") + .contentStreamed(false) + .put(MateClawStateKeys.PROMPT_TOKENS, state.value(MateClawStateKeys.PROMPT_TOKENS, 0) + stepPromptTokens) + .put(MateClawStateKeys.COMPLETION_TOKENS, state.value(MateClawStateKeys.COMPLETION_TOKENS, 0) + stepCompletionTokens) + .events(events) + .build(); + } + if (finalResult == null) { if (wallClockExceeded) { finalResult = "步骤执行超过最大耗时限制(" @@ -438,6 +528,46 @@ public class StepExecutionNode implements NodeAction { events.add(GraphEventPublisher.iterationEnd(stepIndex, "parent", null, shortError != null ? shortError.length() : 0, 0)); } + + // Step-failure recovery: rather than aborting the whole plan on a + // single failed step, re-plan the remaining work around the failure + // (up to MAX_REPLANS_PER_RUN). Completed steps are preserved in + // WORKING_CONTEXT, so the next PlanGeneration pass can skip them and + // route around (or retry) what broke. The mid-pass plan state is + // cleared so a fresh plan is derived; PLAN_REPLAN_COUNT bounds the loop. + int replanCount = accessor.replanCount(); + if (replanCount < MAX_REPLANS_PER_RUN) { + String replanContext = buildReplanContext(accessor, stepIndex, shortError); + events.add(new GraphEventPublisher.GraphEvent("plan_replan", Map.of( + "failedStepIndex", stepIndex, + "attempt", replanCount + 1, + "maxReplans", MAX_REPLANS_PER_RUN, + "error", shortError == null ? "" : shortError), + System.currentTimeMillis())); + log.warn("[StepExecution] Step {} failed; re-planning remaining work (attempt {}/{})", + stepIndex + 1, replanCount + 1, MAX_REPLANS_PER_RUN); + return PlanStateAccessor.output() + .workingContext(replanContext) + .currentPhase("plan_replan") + .replanCount(replanCount + 1) + // Wipe mid-pass plan state so PlanGenerationNode re-derives + // a fresh plan from goal + (failure-augmented) context. + .planId(null) + .planSteps(List.of()) + .planValid(false) + .needsPlanning(true) + .currentStepIndex(0) + .currentStepTitle("") + .currentStepResult("") + .contentStreamed(false) + .put(MateClawStateKeys.PROMPT_TOKENS, state.value(MateClawStateKeys.PROMPT_TOKENS, 0) + stepPromptTokens) + .put(MateClawStateKeys.COMPLETION_TOKENS, state.value(MateClawStateKeys.COMPLETION_TOKENS, 0) + stepCompletionTokens) + .events(events) + .build(); + } + + log.warn("[StepExecution] Step {} failed and re-plan budget exhausted ({}); aborting plan", + stepIndex + 1, MAX_REPLANS_PER_RUN); return PlanStateAccessor.output() .currentStepResult(shortError) .currentPhase("plan_aborted") @@ -612,6 +742,30 @@ public class StepExecutionNode implements NodeAction { } } + /** + * Augment the working context with a note about the failed step so the next + * PlanGeneration pass re-plans around it. The completed-step results are + * already encoded in {@code WORKING_CONTEXT}; this appends only the failure + * so the planner can skip what's done, retry differently, or route around + * the broken step. The note is an internal LLM prompt (Chinese, matching the + * surrounding planning/execution prompts). + */ + static String buildReplanContext(PlanStateAccessor accessor, int failedStepIndex, String error) { + List steps = accessor.planSteps(); + String failedTitle = (failedStepIndex >= 0 && failedStepIndex < steps.size()) + ? steps.get(failedStepIndex) : ("步骤 " + (failedStepIndex + 1)); + StringBuilder sb = new StringBuilder(accessor.workingContext()); + if (sb.length() > 0) { + sb.append("\n\n"); + } + sb.append("【上一轮计划执行失败】步骤 ").append(failedStepIndex + 1) + .append("(").append(failedTitle).append(")执行失败:") + .append(error == null ? "未知错误" : error) + .append("\n请基于上面已完成的工作,重新规划达成总目标所需的剩余步骤:") + .append("绕开或换一种方式完成失败的部分,不要重复已经完成的步骤。"); + return sb.toString(); + } + /** * 将异常转换为简短的错误摘要,避免将完整异常体(尤其是 429 JSON)写入后续 prompt。 *

    diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepProgressTracker.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepProgressTracker.java new file mode 100644 index 00000000..24ac9b04 --- /dev/null +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/node/StepProgressTracker.java @@ -0,0 +1,152 @@ +package vip.mate.agent.graph.plan.node; + +import java.util.HashMap; +import java.util.HashSet; +import java.util.Map; +import java.util.Optional; +import java.util.Set; + +/** + * Per-step progress detector for the Plan-Execute executor. + * + *

    A plan step runs its own inner tool-calling loop. Without progress + * tracking, a step can spin — repeatedly calling the same tool, or hammering + * different variants that all fail / return nothing — until it hits the + * tool-call ceiling, then "complete" with an empty result and let the plan + * plow into dependent steps that have no real input. + * + *

    This tracker watches the tool results of each round and recognises two + * signature-based stall patterns: + *

      + *
    • repeated failure — the same call (tool name + canonical args) + * keeps failing, or the same tool keeps failing with different args;
    • + *
    • no progress — a call keeps returning the same result, + * so re-issuing it yields nothing new.
    • + *
    + * + *

    Detection is graduated: at the WARN threshold it emits a one-shot nudge + * (injected back into the step's messages so the model changes strategy); + * past the HALT threshold it flags the step as stuck so the executor can stop + * the inner loop and re-plan instead of advancing with junk. Thresholds are + * deliberately low — the goal is to break a stall early, before the whole + * tool-call budget is burned. + * + *

    Not thread-safe; create one per step. + */ +public final class StepProgressTracker { + + /** Same exact call (tool + args) failing: nudge / halt thresholds. */ + static final int SAME_CALL_FAIL_WARN = 2; + static final int SAME_CALL_FAIL_HALT = 4; + /** Same tool failing across different args: nudge / halt thresholds. */ + static final int SAME_TOOL_FAIL_WARN = 3; + static final int SAME_TOOL_FAIL_HALT = 6; + /** Same call returning identical output (no new information): nudge / halt. */ + static final int NO_PROGRESS_WARN = 2; + static final int NO_PROGRESS_HALT = 4; + + /** + * Lower-cased markers that identify a tool result as a failure / empty + * outcome. Kept intentionally small and language-mixed: tool errors in this + * codebase surface as English exception text, while a few common "not + * found" phrasings also appear in Chinese tool output. + */ + private static final String[] FAILURE_MARKERS = { + "execution failed", "error:", "exception", "timeout", "timed out", + "enoent", "no such file", "not found", "authentication failed", + "permission denied", "failed to", "未找到", "不存在", "没有找到", "执行失败", "无法" + }; + + private final Map sameCallFail = new HashMap<>(); + private final Map sameToolFail = new HashMap<>(); + private final Map resultRepeat = new HashMap<>(); + private final Set warnedKeys = new HashSet<>(); + + private boolean stuck = false; + private String haltReason = null; + + /** + * Record one tool result from the current round. + * + * @param toolName the invoked tool's name (never null) + * @param argsJson the raw arguments JSON (may be empty when unresolved) + * @param resultText the tool's result text (may be null/empty) + * @return a nudge to inject into the step's messages when a WARN threshold + * was freshly crossed, otherwise empty. Each distinct warning fires + * at most once. + */ + public Optional record(String toolName, String argsJson, String resultText) { + String name = toolName == null ? "tool" : toolName; + String args = argsJson == null ? "" : argsJson; + String result = resultText == null ? "" : resultText; + boolean failure = looksLikeFailure(result); + + String callSig = name + "::" + args.hashCode(); + String resultKey = callSig + "##" + result.trim().hashCode(); + + // No-progress: identical result for the same call, regardless of success. + int repeats = resultRepeat.merge(resultKey, 1, Integer::sum); + if (repeats >= NO_PROGRESS_HALT) { + markStuck("no_progress:" + name); + } + Optional nudge = maybeWarn(repeats >= NO_PROGRESS_WARN, "np:" + resultKey, + "工具 " + name + " 已连续 " + repeats + " 次返回相同结果。不要重复同样的调用——" + + "改用已有结果、换查询/换工具,或直接基于现有信息给出本步骤结论。"); + + if (failure) { + int callFails = sameCallFail.merge(callSig, 1, Integer::sum); + int toolFails = sameToolFail.merge(name, 1, Integer::sum); + if (callFails >= SAME_CALL_FAIL_HALT || toolFails >= SAME_TOOL_FAIL_HALT) { + markStuck("repeated_failure:" + name); + } + if (nudge.isEmpty()) { + nudge = maybeWarn(callFails >= SAME_CALL_FAIL_WARN, "cf:" + callSig, + "工具 " + name + " 用相同参数已失败 " + callFails + " 次,像是死循环。" + + "先看错误原因再换一种方式,不要原样重试。"); + } + if (nudge.isEmpty()) { + nudge = maybeWarn(toolFails >= SAME_TOOL_FAIL_WARN, "tf:" + name, + "工具 " + name + " 本步骤已失败 " + toolFails + " 次。停止在同一条失败路径上重试," + + "换工具或换思路完成本步骤。"); + } + } + return nudge; + } + + /** True once a HALT threshold was crossed — the step should stop and re-plan. */ + public boolean isStuck() { + return stuck; + } + + /** Machine-readable reason for the halt, or null when not stuck. */ + public String haltReason() { + return haltReason; + } + + private Optional maybeWarn(boolean crossed, String key, String message) { + if (crossed && warnedKeys.add(key)) { + return Optional.of(message); + } + return Optional.empty(); + } + + private void markStuck(String reason) { + if (!stuck) { + stuck = true; + haltReason = reason; + } + } + + static boolean looksLikeFailure(String result) { + if (result == null || result.isBlank()) { + return true; // an empty result is no progress either + } + String lower = result.toLowerCase(); + for (String marker : FAILURE_MARKERS) { + if (lower.contains(marker)) { + return true; + } + } + return false; + } +} diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateAccessor.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateAccessor.java index dccce082..98d3e385 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateAccessor.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateAccessor.java @@ -73,6 +73,11 @@ public final class PlanStateAccessor { return state.>value(COMPLETED_RESULTS).orElse(List.of()); } + /** Re-plans already performed this run (0 at run start). */ + public int replanCount() { + return state.value(PLAN_REPLAN_COUNT, 0); + } + // ===== 终止 ===== public String finalSummary() { @@ -187,6 +192,10 @@ public final class PlanStateAccessor { return put(CURRENT_STEP_INDEX, index); } + public OutputBuilder replanCount(int count) { + return put(PLAN_REPLAN_COUNT, count); + } + public OutputBuilder currentStepTitle(String title) { return put(CURRENT_STEP_TITLE, title); } diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateKeys.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateKeys.java index 37ff58a2..ac4e720c 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateKeys.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/plan/state/PlanStateKeys.java @@ -27,6 +27,15 @@ public final class PlanStateKeys { public static final String CURRENT_STEP_RESULT = "current_step_result"; public static final String COMPLETED_RESULTS = "completed_results"; // APPEND 策略 + /** + * Number of re-plans performed in THIS graph run (REPLACE strategy). When a + * step throws, the executor re-plans the remaining work around the failure + * (carried in {@link #WORKING_CONTEXT}) instead of aborting outright, up to + * a small bound — this counter enforces that bound so a pathological failure + * loop can't re-plan forever. Implicitly 0 at run start. + */ + public static final String PLAN_REPLAN_COUNT = "plan_replan_count"; + // ===== 终止 ===== public static final String FINAL_SUMMARY = "final_summary"; public static final String DIRECT_ANSWER = "direct_answer"; // 简单问答的直接回答 diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateAccessor.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateAccessor.java index 96c18cef..23a8b6f8 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateAccessor.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateAccessor.java @@ -80,6 +80,11 @@ public final class MateClawStateAccessor { return state.value(LLM_CALL_COUNT, 0); } + /** Iterations refunded this run for setup-only (progressive-disclosure) rounds (0 at run start). */ + public int iterationRefundCount() { + return state.value(ITERATION_REFUND_COUNT, 0); + } + // ===== 观察历史 ===== @SuppressWarnings("unchecked") @@ -303,6 +308,11 @@ public final class MateClawStateAccessor { return state.value(GOAL_ACCOUNTED_LLM_CALL_COUNT, 0); } + /** Hard continuations (fresh-budget ReAct segments) performed this run (0 at run start). */ + public int goalHardContinuationCount() { + return state.value(GOAL_HARD_CONTINUATION_COUNT, 0); + } + /** * Bridge across ReAct and Plan-Execute: ReAct writes the terminal text * to {@link MateClawStateKeys#FINAL_ANSWER} via FinalAnswerNode; @@ -368,6 +378,10 @@ public final class MateClawStateAccessor { return put(NEEDS_TOOL_CALL, needs); } + public OutputBuilder iterationRefundCount(int count) { + return put(ITERATION_REFUND_COUNT, count); + } + // ---- 消息 ---- public OutputBuilder messages(List msgs) { return put(MESSAGES, msgs); @@ -552,6 +566,10 @@ public final class MateClawStateAccessor { return put(GOAL_ACCOUNTED_LLM_CALL_COUNT, n); } + public OutputBuilder goalHardContinuationCount(int n) { + return put(GOAL_HARD_CONTINUATION_COUNT, n); + } + /** Wipe FINAL_ANSWER on follow-up so the next graph pass doesn't * immediately re-terminate via the existing final text. */ public OutputBuilder clearFinalAnswer() { @@ -563,6 +581,18 @@ public final class MateClawStateAccessor { return put(FINISH_REASON, ""); } + /** + * Wipe the limit-exceeded draft + flag. Required before a hard + * continuation re-enters the ReAct loop: FinalAnswerNode prefers + * FINAL_ANSWER_DRAFT over a freshly reasoned answer, so a stale draft + * left by LimitExceededNode would otherwise resurface as the next + * segment's answer. + */ + public OutputBuilder clearLimitExceededDraft() { + put(FINAL_ANSWER_DRAFT, ""); + return put(LIMIT_EXCEEDED, false); + } + /** Plan-Execute follow-up: clear the terminal-side plan summary so * the next PlanGeneration pass starts clean. Identifier is the * string literal "final_summary" to avoid a compile-time link to diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateKeys.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateKeys.java index 2e269f57..56b66517 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateKeys.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/state/MateClawStateKeys.java @@ -30,6 +30,16 @@ public final class MateClawStateKeys { public static final String CURRENT_ITERATION = "current_iteration"; public static final String MAX_ITERATIONS = "max_iterations"; + /** + * Iterations refunded this run because a reasoning round did no real work — + * its whole tool batch was progressive-disclosure setup ({@code load_skill} + * / {@code enable_tool}). ObservationNode skips the iteration increment for + * such rounds so a tight budget isn't eaten by the load-then-use two-step; + * this counter bounds the refunds so a model that only ever loads skills + * still terminates. Implicitly 0 at run start. REPLACE strategy. + */ + public static final String ITERATION_REFUND_COUNT = "iteration_refund_count"; + // ===== 工具调用(REPLACE 策略)===== public static final String TOOL_CALLS = "tool_calls"; public static final String TOOL_RESULTS = "tool_results"; @@ -228,6 +238,22 @@ public final class MateClawStateKeys { */ public static final String GOAL_ACCOUNTED_LLM_CALL_COUNT = "goal_accounted_llm_call_count"; + /** + * Number of "hard continuations" already performed in THIS graph run — a + * hard continuation is a goal follow-up that re-enters the ReAct loop with + * a FRESH iteration budget (CURRENT_ITERATION reset to 0) after a turn that + * ended in {@link FinishReason#MAX_ITERATIONS_REACHED}. Unlike a normal + * follow-up (which shares the run's single iteration budget), a hard + * continuation grants the goal a brand-new ReAct segment so a task too big + * for one budget can keep going autonomously instead of stalling until the + * user sends another message. Because each such segment costs up to a full + * {@code maxIterations} worth of node visits, it is bounded by a dedicated, + * tighter cap ({@code mateclaw.goal.max-hard-continuations-per-run}, clamped + * to {@link vip.mate.goal.config.GoalProperties#MAX_HARD_CONTINUATIONS_CEILING}) + * and sized into the graph recursion ceiling. Implicitly 0 at run start. + */ + public static final String GOAL_HARD_CONTINUATION_COUNT = "goal_hard_continuation_count"; + /** Graph-node identifier for the GoalEvaluationNode. */ public static final String GOAL_EVALUATION_NODE = "goal_evaluation"; diff --git a/mateclaw-server/src/main/java/vip/mate/goal/config/GoalProperties.java b/mateclaw-server/src/main/java/vip/mate/goal/config/GoalProperties.java index 9c42da46..6909e9ef 100644 --- a/mateclaw-server/src/main/java/vip/mate/goal/config/GoalProperties.java +++ b/mateclaw-server/src/main/java/vip/mate/goal/config/GoalProperties.java @@ -16,6 +16,16 @@ import org.springframework.stereotype.Component; @ConfigurationProperties(prefix = "mateclaw.goal") public class GoalProperties { + /** + * Compile-time ceiling for {@link #maxHardContinuationsPerRun}. The graph + * recursion limit is sized statically to accommodate this many extra + * fresh-budget ReAct segments per run, so the runtime value is clamped to + * it — an operator cannot push hard continuations past what the recursion + * backstop was sized for. Raising this requires re-sizing the recursion + * ceiling in {@code AgentGraphBuilder.frameworkRecursionLimit()}. + */ + public static final int MAX_HARD_CONTINUATIONS_CEILING = 3; + /** * Master switch — when off, the graph never invokes GoalEvaluationNode * (the conditional edge sees no active goal, so the node is unreachable). @@ -39,6 +49,19 @@ public class GoalProperties { */ private boolean allowAutoFollowup = true; + /** + * Auto-derive a goal from a multi-step Plan-Execute plan. The Plan-Execute + * planner decomposes the request into steps and the step executor is a + * narrow "task runner" — neither calls {@code setGoal}, so without this a + * Plan-Execute run never engages the goal subsystem. When enabled, a goal is + * created server-side at plan generation (title = request, acceptance + * criteria seeded from the plan steps), so the already-wired + * GoalEvaluationNode tracks completion. Gated by {@link #enabled}; only + * fires for genuine multi-step plans and when the conversation has no active + * goal yet. Set to {@code false} to keep Plan-Execute goal-free. + */ + private boolean autoGoalFromPlan = true; + /** Default turn budget when the user doesn't override. */ private int defaultTurnBudget = 20; @@ -57,6 +80,20 @@ public class GoalProperties { */ private int maxFollowupsPerRun = 8; + /** + * Max "hard continuations" per single graph run. A hard continuation is a + * goal follow-up that re-enters the ReAct loop with a FRESH iteration + * budget after a turn that hit {@code MAX_ITERATIONS_REACHED} — letting a + * task too large for one budget keep going autonomously instead of stalling + * until the user sends another message. Each one costs up to a full + * {@code maxIterations} worth of node visits, so this is a dedicated cap on + * top of {@link #maxFollowupsPerRun}, clamped to + * {@link #MAX_HARD_CONTINUATIONS_CEILING} and sized into the graph recursion + * ceiling. The goal's cross-turn turn / LLM-call budgets still apply. Set to + * 0 to keep the previous behaviour (max-iterations turns end the run). + */ + private int maxHardContinuationsPerRun = 1; + /** * Provider/model id for the evaluator. Empty string means "use the * same model as the chat agent" — convenient for dev, expensive in diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/LaneDPerformanceFixesTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/LaneDPerformanceFixesTest.java index e00454cc..f3b18f5f 100644 --- a/mateclaw-server/src/test/java/vip/mate/agent/graph/LaneDPerformanceFixesTest.java +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/LaneDPerformanceFixesTest.java @@ -159,6 +159,10 @@ class LaneDPerformanceFixesTest { }); var helper = helper(model); + // Shrink backoff to ~1ms so the test doesn't sleep through the real + // 3s/6s exponential backoff; a huge time budget keeps the retry-count + // logic (not the wall-clock cap) the thing under test. + helper.setRetryTimingForTest(1, 1, Long.MAX_VALUE); var result = helper.streamCall(model, smallPrompt(), "conv-d2a", "reasoning"); // With MAX_RETRIES_RATE_LIMIT=2, attempts are: 0, 1, 2 = 3 total calls @@ -179,6 +183,13 @@ class LaneDPerformanceFixesTest { }); var helper = helper(model); + // Shrink backoff to ~1ms and lift the wall-clock budget so the full + // MAX_RETRIES path runs to completion. Without this, the real timing + // (exponential backoff capped at 60s vs a 3-minute total budget) cuts + // the loop off at ~8 calls after running for ~4 minutes — this test + // is about the retry COUNT, not the time budget (covered separately by + // serverErrorTimeBudgetCapsRetries). + helper.setRetryTimingForTest(1, 1, Long.MAX_VALUE); var result = helper.streamCall(model, smallPrompt(), "conv-d2b", "reasoning"); // SERVER_ERROR should use the full MAX_RETRIES budget, NOT the reduced @@ -194,6 +205,32 @@ class LaneDPerformanceFixesTest { "(attempt 0 through " + NodeStreamingChatHelper.MAX_RETRIES + ")"); } + @Test + @DisplayName("SERVER_ERROR stops early when the total-time budget is exhausted") + void serverErrorTimeBudgetCapsRetries() { + AtomicInteger callCount = new AtomicInteger(0); + ChatModel model = mock(ChatModel.class); + when(model.stream(any(Prompt.class))).thenAnswer(inv -> { + callCount.incrementAndGet(); + return Flux.error(new RuntimeException("500 Internal Server Error")); + }); + + var helper = helper(model); + // 50ms backoff but only a 10ms total budget: the wall-clock cap — not + // MAX_RETRIES — bounds a sustained server-error loop. The loop should + // bail after the first backoff pushes elapsed time past the budget, + // well before the full MAX_RETRIES would be consumed. + helper.setRetryTimingForTest(50, 50, 10); + var result = helper.streamCall(model, smallPrompt(), "conv-d2d", "reasoning"); + + assertTrue(callCount.get() >= 1, "At least the initial attempt should run"); + assertTrue(callCount.get() < NodeStreamingChatHelper.MAX_RETRIES + 1, + "Time budget should cut SERVER_ERROR retries short of the full " + + "MAX_RETRIES budget, but got " + callCount.get()); + assertNotEquals(NodeStreamingChatHelper.ErrorType.NONE, result.errorType(), + "Result should be an error after the time budget is exhausted"); + } + @Test @DisplayName("AUTH_ERROR is not retried (unchanged behavior)") void authErrorNotRetried() { diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/node/GoalEvaluationNodeContinuationTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/node/GoalEvaluationNodeContinuationTest.java new file mode 100644 index 00000000..91d4bd87 --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/node/GoalEvaluationNodeContinuationTest.java @@ -0,0 +1,237 @@ +package vip.mate.agent.graph.node; + +import com.alibaba.cloud.ai.graph.OverAllState; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.messages.Message; +import org.springframework.ai.chat.messages.UserMessage; +import vip.mate.agent.context.ConversationWindowManager; +import vip.mate.agent.graph.state.FinishReason; +import vip.mate.agent.graph.state.MateClawStateKeys; +import vip.mate.goal.config.GoalProperties; +import vip.mate.goal.model.GoalEntity; +import vip.mate.goal.model.GoalEvaluationResult; +import vip.mate.goal.model.GoalResponse; +import vip.mate.goal.service.GoalEvaluationService; +import vip.mate.goal.service.GoalFollowupService; +import vip.mate.goal.service.GoalService; +import vip.mate.goal.service.GraphFlavor; +import vip.mate.workspace.conversation.ConversationService; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertInstanceOf; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.mockito.ArgumentMatchers.any; +import static org.mockito.ArgumentMatchers.anyList; +import static org.mockito.ArgumentMatchers.anyString; +import static org.mockito.ArgumentMatchers.eq; +import static org.mockito.Mockito.lenient; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.never; +import static org.mockito.Mockito.verify; +import static org.mockito.Mockito.when; + +/** + * Locks the goal self-continuation behaviour on terminal turns: + *

      + *
    • MAX_ITERATIONS_REACHED now continues with a FRESH iteration budget + * ("hard continuation") instead of skipping silently.
    • + *
    • EVIDENCE_INSUFFICIENT now continues (corrective follow-up) without a + * budget reset.
    • + *
    • STOPPED / RETURN_DIRECT / ERROR_FALLBACK still skip.
    • + *
    • The hard-continuation cap bounds the fresh-budget loop.
    • + *
    + */ +class GoalEvaluationNodeContinuationTest { + + // ===== Pure decision helpers ===== + + @Test + void hardSkip_coversUserAndNonProgressTerminals_only() { + assertTrue(GoalEvaluationNode.isHardSkipFinishReason(FinishReason.STOPPED.getValue())); + assertTrue(GoalEvaluationNode.isHardSkipFinishReason(FinishReason.RETURN_DIRECT.getValue())); + assertTrue(GoalEvaluationNode.isHardSkipFinishReason(FinishReason.ERROR_FALLBACK.getValue())); + // The two that must now fall through to evaluation: + assertFalse(GoalEvaluationNode.isHardSkipFinishReason(FinishReason.MAX_ITERATIONS_REACHED.getValue())); + assertFalse(GoalEvaluationNode.isHardSkipFinishReason(FinishReason.EVIDENCE_INSUFFICIENT.getValue())); + assertFalse(GoalEvaluationNode.isHardSkipFinishReason(FinishReason.NORMAL.getValue())); + } + + @Test + void iterationCapReached_onlyForMaxIterations() { + assertTrue(GoalEvaluationNode.isIterationCapReached(FinishReason.MAX_ITERATIONS_REACHED.getValue())); + assertFalse(GoalEvaluationNode.isIterationCapReached(FinishReason.EVIDENCE_INSUFFICIENT.getValue())); + assertFalse(GoalEvaluationNode.isIterationCapReached(FinishReason.NORMAL.getValue())); + } + + // ===== resolveActiveGoal: same-turn activation ===== + + @Test + void resolveActiveGoal_prefersStateSnapshot_noDbLookup() { + GoalService goalService = mock(GoalService.class); + GoalEntity snap = new GoalEntity(); + snap.setId(7L); + OverAllState s = new OverAllState(Map.of( + MateClawStateKeys.ACTIVE_GOAL, snap, + MateClawStateKeys.CONVERSATION_ID, "c1")); + + Optional out = GoalEvaluationNode.resolveActiveGoal(s, goalService); + + assertTrue(out.isPresent()); + assertEquals(7L, out.get().getId()); + // Snapshot hit must not touch the DB. + verify(goalService, never()).findActiveByConversation(anyString()); + } + + @Test + void resolveActiveGoal_fallsBackToDb_whenSnapshotEmpty() { + GoalService goalService = mock(GoalService.class); + GoalEntity fromDb = new GoalEntity(); + fromDb.setId(9L); + when(goalService.findActiveByConversation("c1")).thenReturn(fromDb); + OverAllState s = new OverAllState(Map.of(MateClawStateKeys.CONVERSATION_ID, "c1")); + + Optional out = GoalEvaluationNode.resolveActiveGoal(s, goalService); + + assertTrue(out.isPresent(), "a goal set mid-turn must be found via the DB fallback"); + assertEquals(9L, out.get().getId()); + } + + @Test + void resolveActiveGoal_emptyWhenNoSnapshotNoDbGoal() { + GoalService goalService = mock(GoalService.class); + when(goalService.findActiveByConversation(anyString())).thenReturn(null); + OverAllState s = new OverAllState(Map.of(MateClawStateKeys.CONVERSATION_ID, "c1")); + assertTrue(GoalEvaluationNode.resolveActiveGoal(s, goalService).isEmpty()); + } + + // ===== apply() behaviour ===== + + @Test + void maxIterations_injectsFollowup_andResetsIterationBudget() throws Exception { + Fixture f = new Fixture(); + Map out = f.node().apply( + f.state(FinishReason.MAX_ITERATIONS_REACHED.getValue(), 0, 0)); + + // Followup injected for the run-to-completion loop. + assertEquals(Boolean.TRUE, out.get(MateClawStateKeys.GOAL_FOLLOWUP_INJECTED)); + assertEquals(1, out.get(MateClawStateKeys.GOAL_FOLLOWUP_COUNT)); + // Hard continuation: fresh ReAct segment. + assertEquals(0, out.get(MateClawStateKeys.CURRENT_ITERATION)); + assertEquals(1, out.get(MateClawStateKeys.GOAL_HARD_CONTINUATION_COUNT)); + // Stale limit-exceeded draft/flag cleared so it can't resurface. + assertEquals("", out.get(MateClawStateKeys.FINAL_ANSWER_DRAFT)); + assertEquals(Boolean.FALSE, out.get(MateClawStateKeys.LIMIT_EXCEEDED)); + assertEquals("", out.get(MateClawStateKeys.FINAL_ANSWER)); + assertEquals("", out.get(MateClawStateKeys.FINISH_REASON)); + // Not a terminal pass — the next answer must be re-evaluated. + assertFalse(Boolean.TRUE.equals(out.get(MateClawStateKeys.GOAL_EVALUATED_THIS_RUN))); + // Followup appended as a user message. + @SuppressWarnings("unchecked") + List msgs = (List) out.get(MateClawStateKeys.MESSAGES); + assertEquals(1, msgs.size()); + assertInstanceOf(UserMessage.class, msgs.get(0)); + } + + @Test + void evidenceInsufficient_injectsFollowup_withoutBudgetReset() throws Exception { + Fixture f = new Fixture(); + Map out = f.node().apply( + f.state(FinishReason.EVIDENCE_INSUFFICIENT.getValue(), 0, 0)); + + assertEquals(Boolean.TRUE, out.get(MateClawStateKeys.GOAL_FOLLOWUP_INJECTED)); + // No iteration reset / hard-continuation accounting on this path. + assertFalse(out.containsKey(MateClawStateKeys.CURRENT_ITERATION)); + assertFalse(out.containsKey(MateClawStateKeys.GOAL_HARD_CONTINUATION_COUNT)); + assertFalse(out.containsKey(MateClawStateKeys.FINAL_ANSWER_DRAFT)); + } + + @Test + void stopped_skipsEvaluationEntirely() throws Exception { + Fixture f = new Fixture(); + Map out = f.node().apply( + f.state(FinishReason.STOPPED.getValue(), 0, 0)); + + assertEquals(Boolean.TRUE, out.get(MateClawStateKeys.GOAL_EVALUATED_THIS_RUN)); + assertFalse(out.containsKey(MateClawStateKeys.GOAL_FOLLOWUP_INJECTED)); + // Evaluator must not even be called on a user-stopped turn. + verify(f.evaluationService, never()).evaluate(any(), anyList(), anyString()); + } + + @Test + void maxIterations_hardCapReached_endsRunWithoutReset() throws Exception { + Fixture f = new Fixture(); + // hardContinuationCount already at the cap (default cap = 1). + Map out = f.node().apply( + f.state(FinishReason.MAX_ITERATIONS_REACHED.getValue(), 0, 1)); + + // Falls through to the terminal "continue, no followup" path. + assertEquals(Boolean.TRUE, out.get(MateClawStateKeys.GOAL_EVALUATED_THIS_RUN)); + assertFalse(out.containsKey(MateClawStateKeys.GOAL_FOLLOWUP_INJECTED)); + assertFalse(out.containsKey(MateClawStateKeys.CURRENT_ITERATION)); + } + + // ===== Test fixture ===== + + private static final class Fixture { + final GoalEvaluationService evaluationService = mock(GoalEvaluationService.class); + final GoalFollowupService followupService = mock(GoalFollowupService.class); + final GoalService goalService = mock(GoalService.class); + final GoalProperties properties = new GoalProperties(); + final ConversationWindowManager windowManager = mock(ConversationWindowManager.class); + final ConversationService conversationService = mock(ConversationService.class); + + Fixture() { + GoalEntity goal = new GoalEntity(); + goal.setId(1L); + goal.setTitle("ship the feature"); + + GoalEvaluationResult continueResult = new GoalEvaluationResult( + 0.5, "missing tests", GoalEvaluationResult.DECISION_CONTINUE, false, + "stub-model", 1, 5L, List.of(), null); + + lenient().when(evaluationService.evaluate(any(), anyList(), anyString())) + .thenReturn(continueResult); + lenient().when(goalService.getById(eq(1L))).thenReturn(goal); + lenient().when(goalService.isBudgetExhausted(any())).thenReturn(false); + lenient().when(goalService.toResponse(any())).thenReturn(mock(GoalResponse.class)); + lenient().when(followupService.maybeBuildFollowup(any(), any())) + .thenReturn(Optional.of("Continue toward the goal. Take the next concrete step.")); + } + + GoalEvaluationNode node() { + return new GoalEvaluationNode(evaluationService, followupService, goalService, + properties, windowManager, conversationService, GraphFlavor.REACT); + } + + /** + * Build a mocked graph state for an active-goal terminal turn. + * + * @param finishReason the REACT finishReason under test + * @param followupCount goal_followup_count already this run + * @param hardContinuationCount goal_hard_continuation_count already this run + */ + OverAllState state(String finishReason, int followupCount, int hardContinuationCount) { + GoalEntity goal = new GoalEntity(); + goal.setId(1L); + goal.setTitle("ship the feature"); + + Map vals = new HashMap<>(); + vals.put(MateClawStateKeys.ACTIVE_GOAL, goal); + vals.put(MateClawStateKeys.GOAL_EVALUATED_THIS_RUN, false); + vals.put(MateClawStateKeys.FINISH_REASON, finishReason); + vals.put(MateClawStateKeys.AWAITING_APPROVAL, false); + vals.put(MateClawStateKeys.FINAL_ANSWER, "partial answer so far"); + vals.put(MateClawStateKeys.LLM_CALL_COUNT, 10); + vals.put(MateClawStateKeys.GOAL_ACCOUNTED_LLM_CALL_COUNT, 0); + vals.put(MateClawStateKeys.GOAL_FOLLOWUP_COUNT, followupCount); + vals.put(MateClawStateKeys.GOAL_HARD_CONTINUATION_COUNT, hardContinuationCount); + return new OverAllState(vals); + } + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java new file mode 100644 index 00000000..29bb5325 --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/node/ObservationNodeRefundTest.java @@ -0,0 +1,86 @@ +package vip.mate.agent.graph.node; + +import com.alibaba.cloud.ai.graph.OverAllState; +import org.junit.jupiter.api.DisplayName; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.messages.ToolResponseMessage; +import vip.mate.agent.graph.observation.ObservationProcessor; +import vip.mate.config.GraphObservationProperties; + +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNull; +import static vip.mate.agent.graph.state.MateClawStateKeys.*; + +/** + * Iteration-refund behaviour: a reasoning round whose entire tool batch was + * progressive-disclosure setup (load_skill / enable_tool) must not consume an + * iteration, bounded by a per-run cap. + */ +class ObservationNodeRefundTest { + + private ObservationNode node() { + return new ObservationNode(new ObservationProcessor(new GraphObservationProperties())); + } + + private static ToolResponseMessage.ToolResponse result(String name) { + return new ToolResponseMessage.ToolResponse("id-" + name, name, "ok"); + } + + private OverAllState state(int iteration, int refundCount, List results) { + Map m = new HashMap<>(); + m.put(CURRENT_ITERATION, iteration); + m.put(MAX_ITERATIONS, 25); + m.put(ITERATION_REFUND_COUNT, refundCount); + m.put(OBSERVATION_HISTORY, new ArrayList()); + m.put(TOOL_RESULTS, results); + m.put(TOOL_CALL_COUNT, 0); + return new OverAllState(m); + } + + @Test + @DisplayName("纯渐进披露轮(load_skill)退还迭代,不递增") + void setupOnlyRound_refundsIteration() throws Exception { + Map out = node().apply(state(3, 0, List.of(result("load_skill")))); + assertEquals(3, out.get(CURRENT_ITERATION), "setup-only round must not advance the iteration"); + assertEquals(1, out.get(ITERATION_REFUND_COUNT)); + } + + @Test + @DisplayName("enable_tool 同样视为 setup-only") + void enableToolRound_refundsIteration() throws Exception { + Map out = node().apply(state(5, 1, List.of(result("enable_tool")))); + assertEquals(5, out.get(CURRENT_ITERATION)); + assertEquals(2, out.get(ITERATION_REFUND_COUNT)); + } + + @Test + @DisplayName("真实工具轮正常计费") + void realToolRound_consumesIteration() throws Exception { + Map out = node().apply(state(3, 0, List.of(result("web_search")))); + assertEquals(4, out.get(CURRENT_ITERATION)); + assertNull(out.get(ITERATION_REFUND_COUNT), "no refund on a real-work round"); + } + + @Test + @DisplayName("混合批次(披露+真实工具)正常计费") + void mixedRound_consumesIteration() throws Exception { + Map out = node().apply( + state(3, 0, List.of(result("load_skill"), result("web_search")))); + assertEquals(4, out.get(CURRENT_ITERATION)); + assertNull(out.get(ITERATION_REFUND_COUNT)); + } + + @Test + @DisplayName("退还次数达上限后不再退还") + void refundCapReached_consumesIteration() throws Exception { + // cap is 3; refundCount already 3 -> charged normally + Map out = node().apply(state(7, 3, List.of(result("load_skill")))); + assertEquals(8, out.get(CURRENT_ITERATION)); + assertNull(out.get(ITERATION_REFUND_COUNT)); + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcherTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcherTest.java new file mode 100644 index 00000000..352bb798 --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/edge/StepProgressDispatcherTest.java @@ -0,0 +1,58 @@ +package vip.mate.agent.graph.plan.edge; + +import com.alibaba.cloud.ai.graph.OverAllState; +import com.alibaba.cloud.ai.graph.StateGraph; +import org.junit.jupiter.api.Test; +import vip.mate.agent.graph.plan.state.PlanStateKeys; +import vip.mate.agent.graph.state.MateClawStateKeys; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; + +/** + * Routing coverage for the Plan-Execute step dispatcher, including the + * step-failure re-plan edge (phase=plan_replan → PLAN_GENERATION). + */ +class StepProgressDispatcherTest { + + private final StepProgressDispatcher dispatcher = new StepProgressDispatcher(); + + private OverAllState state(String phase, int stepIndex, List steps) { + Map vals = new HashMap<>(); + vals.put(MateClawStateKeys.CURRENT_PHASE, phase); + vals.put(PlanStateKeys.CURRENT_STEP_INDEX, stepIndex); + vals.put(PlanStateKeys.PLAN_STEPS, steps); + return new OverAllState(vals); + } + + @Test + void replanPhase_routesToPlanGeneration() { + String next = dispatcher.apply(state("plan_replan", 0, List.of("a", "b"))); + assertEquals(PlanStateKeys.PLAN_GENERATION_NODE, next); + } + + @Test + void abortedPhase_routesToEnd() { + assertEquals(StateGraph.END, dispatcher.apply(state("plan_aborted", 1, List.of("a", "b")))); + } + + @Test + void awaitingApproval_routesToEnd() { + assertEquals(StateGraph.END, dispatcher.apply(state("awaiting_approval", 0, List.of("a")))); + } + + @Test + void moreStepsRemaining_routesToStepExecution() { + assertEquals(PlanStateKeys.STEP_EXECUTION_NODE, + dispatcher.apply(state("step_completed", 1, List.of("a", "b", "c")))); + } + + @Test + void allStepsDone_routesToPlanSummary() { + assertEquals(PlanStateKeys.PLAN_SUMMARY_NODE, + dispatcher.apply(state("step_completed", 2, List.of("a", "b")))); + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/PlanGenerationAutoGoalTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/PlanGenerationAutoGoalTest.java new file mode 100644 index 00000000..a2c793db --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/PlanGenerationAutoGoalTest.java @@ -0,0 +1,120 @@ +package vip.mate.agent.graph.plan.node; + +import com.alibaba.cloud.ai.graph.OverAllState; +import org.junit.jupiter.api.Test; +import org.mockito.ArgumentCaptor; +import vip.mate.agent.context.ChatOrigin; +import vip.mate.agent.graph.plan.state.PlanStateAccessor; +import vip.mate.agent.graph.plan.state.PlanStateKeys; +import vip.mate.agent.graph.state.MateClawStateKeys; +import vip.mate.goal.config.GoalProperties; +import vip.mate.goal.model.GoalCreateRequest; +import vip.mate.goal.model.GoalEntity; +import vip.mate.goal.service.GoalService; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.mockito.ArgumentMatchers.any; +import static org.mockito.ArgumentMatchers.eq; +import static org.mockito.Mockito.lenient; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.never; +import static org.mockito.Mockito.verify; +import static org.mockito.Mockito.when; + +/** + * Covers auto-deriving a goal from a multi-step Plan-Execute plan: gating + * conditions and that the plan steps seed the goal's acceptance criteria. + */ +class PlanGenerationAutoGoalTest { + + private final GoalService goalService = mock(GoalService.class); + private final GoalProperties properties = new GoalProperties(); + + private PlanGenerationNode node() { + return new PlanGenerationNode(null, null, null, null, null, goalService, properties); + } + + private PlanStateAccessor accessor(boolean withAgent, String goalText) { + ChatOrigin origin = ChatOrigin.web("conv_1", "admin", 1L, null); + if (withAgent) { + origin = origin.withAgent(1000000001L); + } + Map vals = new HashMap<>(); + vals.put(MateClawStateKeys.CHAT_ORIGIN, origin); + vals.put(PlanStateKeys.GOAL, goalText); + return new PlanStateAccessor(new OverAllState(vals)); + } + + @Test + void multiStepPlan_createsGoal_seededWithStepCriteria() { + GoalEntity created = new GoalEntity(); + created.setId(99L); + when(goalService.create(any(), eq("admin"))).thenReturn(created); + + GoalEntity result = node().maybeAutoCreateGoal( + accessor(true, "分三步完成:读取、分析、汇总"), + List.of("读取文件", "列出建议", "汇总计划")); + + assertNotNull(result); + assertEquals(99L, result.getId()); + + ArgumentCaptor cap = ArgumentCaptor.forClass(GoalCreateRequest.class); + verify(goalService).create(cap.capture(), eq("admin")); + GoalCreateRequest req = cap.getValue(); + assertEquals("conv_1", req.getConversationId()); + assertEquals(1000000001L, req.getAgentId()); + // Plan steps become acceptance criteria. + assertNotNull(req.getCriteria()); + assertEquals(3, req.getCriteria().size()); + assertEquals("读取文件", req.getCriteria().get(0).text()); + } + + @Test + void featureDisabled_returnsNull_noCreate() { + properties.setAutoGoalFromPlan(false); + GoalEntity result = node().maybeAutoCreateGoal( + accessor(true, "g"), List.of("a", "b")); + assertNull(result); + verify(goalService, never()).create(any(), any()); + } + + @Test + void singleStepPlan_returnsNull() { + GoalEntity result = node().maybeAutoCreateGoal(accessor(true, "g"), List.of("only one")); + assertNull(result); + verify(goalService, never()).create(any(), any()); + } + + @Test + void existingActiveGoal_returnsNull_noCreate() { + when(goalService.findActiveByConversation("conv_1")).thenReturn(new GoalEntity()); + GoalEntity result = node().maybeAutoCreateGoal( + accessor(true, "g"), List.of("a", "b")); + assertNull(result); + verify(goalService, never()).create(any(), any()); + } + + @Test + void missingAgentContext_returnsNull() { + GoalEntity result = node().maybeAutoCreateGoal( + accessor(false, "g"), List.of("a", "b")); + assertNull(result); + verify(goalService, never()).create(any(), any()); + } + + @Test + void masterSwitchOff_returnsNull() { + properties.setEnabled(false); + lenient().when(goalService.create(any(), any())).thenReturn(new GoalEntity()); + GoalEntity result = node().maybeAutoCreateGoal( + accessor(true, "g"), List.of("a", "b")); + assertNull(result); + verify(goalService, never()).create(any(), any()); + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/StepExecutionReplanContextTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/StepExecutionReplanContextTest.java new file mode 100644 index 00000000..ec680962 --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/StepExecutionReplanContextTest.java @@ -0,0 +1,66 @@ +package vip.mate.agent.graph.plan.node; + +import com.alibaba.cloud.ai.graph.OverAllState; +import org.junit.jupiter.api.Test; +import vip.mate.agent.graph.plan.state.PlanStateAccessor; +import vip.mate.agent.graph.plan.state.PlanStateKeys; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Verifies the re-plan context that StepExecutionNode hands to the next + * PlanGeneration pass after a step fails: it must preserve the completed work + * (carried in WORKING_CONTEXT) and append the failed step + error so the + * planner can route around it. + */ +class StepExecutionReplanContextTest { + + private PlanStateAccessor accessor(String workingContext, List steps) { + Map vals = new HashMap<>(); + vals.put(PlanStateKeys.WORKING_CONTEXT, workingContext); + vals.put(PlanStateKeys.PLAN_STEPS, steps); + return new PlanStateAccessor(new OverAllState(vals)); + } + + @Test + void replanContext_preservesPriorWork_andDescribesFailure() { + PlanStateAccessor a = accessor( + "已完成:步骤1 读取配置完成", + List.of("读取配置", "迁移数据", "验证结果")); + + String ctx = StepExecutionNode.buildReplanContext(a, 1, "connection timeout"); + + // Prior completed work is carried forward. + assertTrue(ctx.contains("已完成:步骤1 读取配置完成")); + // The failed step (1-based) + its title + the error are described. + assertTrue(ctx.contains("步骤 2")); + assertTrue(ctx.contains("迁移数据")); + assertTrue(ctx.contains("connection timeout")); + // Instructs the planner not to redo completed work. + assertTrue(ctx.contains("不要重复")); + } + + @Test + void replanContext_handlesEmptyPriorContext() { + PlanStateAccessor a = accessor("", List.of("only step")); + String ctx = StepExecutionNode.buildReplanContext(a, 0, "boom"); + // No leading blank separator when there was no prior context. + assertFalse(ctx.startsWith("\n")); + assertTrue(ctx.contains("步骤 1")); + assertTrue(ctx.contains("only step")); + assertTrue(ctx.contains("boom")); + } + + @Test + void replanContext_toleratesOutOfRangeIndexAndNullError() { + PlanStateAccessor a = accessor("ctx", List.of("a")); + String ctx = StepExecutionNode.buildReplanContext(a, 5, null); + assertTrue(ctx.contains("步骤 6")); + assertTrue(ctx.contains("未知错误")); + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/StepProgressTrackerTest.java b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/StepProgressTrackerTest.java new file mode 100644 index 00000000..2b405f2f --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/agent/graph/plan/node/StepProgressTrackerTest.java @@ -0,0 +1,74 @@ +package vip.mate.agent.graph.plan.node; + +import org.junit.jupiter.api.Test; + +import java.util.Optional; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Covers the signature-based stall detection: graduated WARN nudge then HALT + * for repeated identical results, repeated identical failing calls, and a tool + * that keeps failing across different arguments. + */ +class StepProgressTrackerTest { + + @Test + void identicalResults_warnThenHalt_noProgress() { + StepProgressTracker t = new StepProgressTracker(); + // result is a success payload (not a failure marker) -> pure no-progress + assertTrue(t.record("search_files", "{\"q\":\"x\"}", "found 0 matches list A").isEmpty(), "1st: no warn"); + assertTrue(t.record("search_files", "{\"q\":\"x\"}", "found 0 matches list A").isPresent(), "2nd: WARN nudge"); + assertFalse(t.isStuck(), "not stuck at WARN"); + assertTrue(t.record("search_files", "{\"q\":\"x\"}", "found 0 matches list A").isEmpty(), "3rd: nudge de-duped"); + t.record("search_files", "{\"q\":\"x\"}", "found 0 matches list A"); // 4th -> HALT + assertTrue(t.isStuck(), "stuck at HALT threshold"); + assertTrue(t.haltReason().startsWith("no_progress")); + } + + @Test + void sameCallFailing_warnThenHalt() { + StepProgressTracker t = new StepProgressTracker(); + // same args, distinct error texts -> isolates the same-call-failure path + assertTrue(t.record("read_file", "{\"p\":\"a\"}", "Error: e1").isEmpty()); + assertTrue(t.record("read_file", "{\"p\":\"a\"}", "Error: e2").isPresent(), "2nd failure: WARN"); + assertFalse(t.isStuck()); + t.record("read_file", "{\"p\":\"a\"}", "Error: e3"); + t.record("read_file", "{\"p\":\"a\"}", "Error: e4"); // 4th failure -> HALT + assertTrue(t.isStuck()); + assertTrue(t.haltReason().startsWith("repeated_failure")); + } + + @Test + void sameToolFailingDifferentArgs_warnThenHalt() { + StepProgressTracker t = new StepProgressTracker(); + boolean anyWarn = false; + for (int i = 1; i <= 6; i++) { + Optional n = t.record("terminal", "{\"cmd\":\"c" + i + "\"}", "execution failed: boom" + i); + anyWarn |= n.isPresent(); + } + assertTrue(anyWarn, "should emit a same-tool-failure nudge by the 3rd distinct failure"); + assertTrue(t.isStuck(), "6 distinct failures of the same tool -> HALT"); + } + + @Test + void variedSuccessfulResults_noStall() { + StepProgressTracker t = new StepProgressTracker(); + for (int i = 0; i < 6; i++) { + assertTrue(t.record("web_search", "{\"q\":\"q" + i + "\"}", "result payload number " + i).isEmpty()); + } + assertFalse(t.isStuck(), "distinct successful results never stall"); + } + + @Test + void looksLikeFailure_classification() { + assertTrue(StepProgressTracker.looksLikeFailure(""), "empty is no-progress"); + assertTrue(StepProgressTracker.looksLikeFailure(" "), "blank is no-progress"); + assertTrue(StepProgressTracker.looksLikeFailure("Error: ENOENT: no such file or directory")); + assertTrue(StepProgressTracker.looksLikeFailure("java.util.concurrent.TimeoutException: ...")); + assertTrue(StepProgressTracker.looksLikeFailure("Authentication Failed: Requires authentication")); + assertTrue(StepProgressTracker.looksLikeFailure("未找到匹配的文件")); + assertFalse(StepProgressTracker.looksLikeFailure("Here is the summary of the file: ...")); + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/architecture/GoalStateKeyDoubleRegistrationTest.java b/mateclaw-server/src/test/java/vip/mate/architecture/GoalStateKeyDoubleRegistrationTest.java index 71fc80e1..bb4ba9c1 100644 --- a/mateclaw-server/src/test/java/vip/mate/architecture/GoalStateKeyDoubleRegistrationTest.java +++ b/mateclaw-server/src/test/java/vip/mate/architecture/GoalStateKeyDoubleRegistrationTest.java @@ -34,6 +34,7 @@ class GoalStateKeyDoubleRegistrationTest { "GOAL_FOLLOWUP_INJECTED", "GOAL_FOLLOWUP_PROMPT", "GOAL_EVALUATED_THIS_RUN", + "GOAL_HARD_CONTINUATION_COUNT", }; @Test diff --git a/mateclaw-ui/src/views/ChatConsole.vue b/mateclaw-ui/src/views/ChatConsole.vue index 454a08b6..823cc723 100644 --- a/mateclaw-ui/src/views/ChatConsole.vue +++ b/mateclaw-ui/src/views/ChatConsole.vue @@ -1154,6 +1154,19 @@ watch(currentConversationId, async (cid) => { } }, { immediate: true }) +// Re-fetch the active goal when a turn finishes. A goal can be created or +// mutated mid-conversation — e.g. auto-derived server-side from a Plan-Execute +// plan, or completed by the agent — without the conversation id changing and +// without a goal_* SSE event reaching this client (skipped creation, missed +// event, reconnect). The per-conversation watch above only fires on switch, so +// this transition-to-idle refresh is what keeps the goal ring honest after +// every turn against the persisted truth. +watch(isGenerating, async (generating, wasGenerating) => { + if (wasGenerating && !generating && currentConversationId.value) { + await goalStore.loadActiveForConversation(currentConversationId.value) + } +}) + // Derive props for the inline prompt + system-line slots that sit // between MessageList and ChatInput. The prompt shows only when: // 1) there's a current conversation, agent, and at least one assistant