* 纯构建器,不做执行。从 AgentService 中提取出所有 Agent 实例构建逻辑,
* 包括模型创建、图编译、prompt 增强等。
*
* @author MateClaw Team
*/
@Slf4j
@Component
@RequiredArgsConstructor
public class AgentGraphBuilder {
private final ToolRegistry toolRegistry;
private final AgentBindingService agentBindingService;
private final SkillService skillService;
private final vip.mate.skill.runtime.SkillRuntimeService skillRuntimeService;
private final ConversationService conversationService;
private final ModelConfigService modelConfigService;
private final ModelProviderService modelProviderService;
private final vip.mate.llm.service.ModelCapabilityService modelCapabilityService;
private final ProviderRouter providerRouter;
private final PlanningService planningService;
private final ToolGuardService toolGuardService;
private final vip.mate.tool.guard.service.ToolGuardConfigService toolGuardConfigService;
private final ApprovalWorkflowService approvalService;
private final ChatStreamTracker streamTracker;
private final SystemSettingService systemSettingService;
// PR-0b: dashScopeChatModel + dashScopeConnectionProperties live on AgentDashScopeChatModelBuilder now.
private final RetryTemplate retryTemplate;
private final ObjectProvider observationRegistryProvider;
private final ObjectProvider restClientBuilderProvider;
private final ObjectProvider webClientBuilderProvider;
private final ObjectMapper objectMapper;
private final GraphObservationProperties graphObservationProperties;
private final vip.mate.config.ToolTimeoutProperties toolTimeoutProperties;
private final MemoryManager memoryManager;
private final WorkspaceFileService workspaceFileService;
private final vip.mate.agent.context.ConversationWindowManager conversationWindowManager;
private final vip.mate.llm.chatgpt.ChatGPTResponsesClient chatGPTResponsesClient;
private final WikiContextService wikiContextService;
private final vip.mate.workspace.core.service.WorkspaceService workspaceService;
private final vip.mate.llm.cache.AnthropicCacheOptionsFactory anthropicCacheOptionsFactory;
private final vip.mate.llm.cache.LlmCacheMetricsAggregator llmCacheMetricsAggregator;
private final vip.mate.agent.graph.executor.ToolResultStorage toolResultStorage;
private final vip.mate.tool.ToolConcurrencyRegistry toolConcurrencyRegistry;
private final vip.mate.i18n.I18nService i18nService;
private final vip.mate.llm.failover.ProviderHealthTracker providerHealthTracker;
private final vip.mate.llm.chatmodel.ProviderChatModelFactory chatModelFactory;
private final vip.mate.llm.failover.AvailableProviderPool providerPool;
private final vip.mate.tool.document.GeneratedFileCache generatedFileCache;
/** PR-0b: DashScope-specific construction lives here now; we only call into it for the search-on log. */
private final vip.mate.agent.chatmodel.AgentDashScopeChatModelBuilder dashScopeBuilder;
private final vip.mate.llm.routing.MultimodalRouter multimodalRouter;
private final vip.mate.llm.routing.MediaCaptionService mediaCaptionService;
/**
* Optional audit pipeline. Setter injection (rather than a constructor
* parameter) keeps existing constructor-based wiring + tests intact.
* When present, the executor receives it so child-agent denied-tool
* attempts can be recorded.
*/
private vip.mate.audit.service.AuditEventService auditEventService;
@org.springframework.beans.factory.annotation.Autowired(required = false)
public void setAuditEventService(vip.mate.audit.service.AuditEventService s) {
this.auditEventService = s;
}
/**
* 根据 AgentEntity 构建完整的 Agent 实例
*/
public BaseAgent build(AgentEntity entity) {
AgentToolSet toolSet = toolRegistry.getEnabledToolSet();
// 过滤掉 denied 工具,使模型完全看不到它们(防止 prompt injection 利用 schema)
toolSet = toolSet.withDeniedToolsFiltered(toolGuardConfigService.getDeniedTools());
// RFC-090 §14.2 — single entry point that merges:
// (a) tools expanded from bound skills' active features, and
// (b) directly bound atomic tools (the Advanced bypass, §9.2 调整 B).
// Three-state semantics: null = no agent-level restriction (use
// global default); non-null (possibly empty) = explicit allowlist.
Set boundTools = agentBindingService.getEffectiveToolNames(entity.getId());
toolSet = toolSet.withAllowedToolsOnly(boundTools); // null = 全局默认
// RFC-090 §9.2 调整 C — pick a primary model that satisfies
// the agent's bound-skill requires-model. Falls back to the
// global default when no preferred provider satisfies, so the
// existing "no default model" error path stays intact.
// Honor per-Agent model override when set.
// resolveModel() looks up entity.modelName in enabled-only models;
// null / blank / unmatched silently fall back to getDefaultModel(),
// preserving the legacy behavior for Agents without an override.
ModelConfigEntity globalDefault;
try {
globalDefault = modelConfigService.resolveModel(entity.getModelName());
} catch (Exception e) {
throw new MateClawException("err.agent.no_default_model", "无法构建 Agent:请先在「设置 → 模型」中配置并启用默认模型");
}
ModelConfigEntity runtimeModel;
try {
runtimeModel = providerRouter.selectPrimary(entity.getId(), globalDefault);
if (runtimeModel == null) runtimeModel = globalDefault;
} catch (Exception e) {
log.debug("[ProviderRouter] primary selection failed, falling back to global default: {}",
e.getMessage());
runtimeModel = globalDefault;
}
// Even after the upgrade, log a WARN when the chosen primary
// still doesn't satisfy needs (e.g. no preferred provider was
// capable). The diagnostic is observability-only.
try {
providerRouter.diagnosePrimary(entity.getId(), runtimeModel);
} catch (Exception e) {
log.debug("[ProviderRouter] diagnostic failed: {}", e.getMessage());
}
ModelProviderEntity provider;
try {
provider = modelProviderService.getProviderConfig(runtimeModel.getProvider());
} catch (Exception e) {
throw new MateClawException("err.agent.model_not_configured", "模型 " + runtimeModel.getModelName()
+ " 的 Provider(" + runtimeModel.getProvider() + ")未配置,请检查模型设置");
}
// Safety net: getDefaultModel() already skips unconfigured providers, but guard here
// too so a stale cached model doesn't silently proceed to a broken API call.
if (!modelProviderService.isProviderConfigured(provider.getProviderId())) {
String reason = modelProviderService.getProviderUnavailableReason(provider.getProviderId());
log.warn("Runtime model {}/{} provider not configured ({}); trying fallback",
runtimeModel.getProvider(), runtimeModel.getModelName(), reason);
ModelConfigEntity fallback = findFirstAvailableChatModel();
if (fallback == null) {
throw new MateClawException("err.agent.no_configured_model",
"默认模型 Provider「" + runtimeModel.getProvider() + "」未配置(" + reason
+ "),且找不到其他已配置的 Provider,请先在「设置 → 模型」中完成配置");
}
runtimeModel = fallback;
try {
provider = modelProviderService.getProviderConfig(runtimeModel.getProvider());
} catch (Exception e) {
throw new MateClawException("err.agent.model_not_configured", "备用模型 " + runtimeModel.getModelName()
+ " 的 Provider(" + runtimeModel.getProvider() + ")获取失败");
}
}
ModelProtocol protocol = ModelProtocol.fromChatModel(provider.getChatModel());
// 内置搜索检测(DashScope / Kimi),但不再移除 WebSearchTool — 两者协同而非互斥
boolean builtinSearchEnabled = false;
Map providerKwargs = modelProviderService.readProviderGenerateKwargs(provider);
if (protocol == ModelProtocol.DASHSCOPE_NATIVE) {
builtinSearchEnabled = dashScopeBuilder.isBuiltinSearchEnabled(runtimeModel, provider);
} else if (isKimiProvider(provider) && Boolean.TRUE.equals(providerKwargs.get("enableSearch"))) {
builtinSearchEnabled = true;
}
if (builtinSearchEnabled) {
// Phase 2: 不再移除 search 工具,改为在 prompt 中设定优先级引导
// 内置搜索作为首选,search 工具作为补充/兜底
log.info("内置搜索已开启 (provider={}),search 工具保留作为补充通道", provider.getProviderId());
}
// Default 100 if DB row leaves max_iterations null. Negative or zero is an
// explicit opt-in to "no soft cap" — ObservationDispatcher already treats
// maxIterations<=0 as "do not enforce", so the agent runs until the LLM
// emits a final answer (or returnDirect short-circuits). Positive values
// are clamped to the hard ceiling so a misconfigured row can't skip the
// safety net unintentionally.
int rawMaxIter = entity.getMaxIterations() != null ? entity.getMaxIterations() : 100;
int maxIter;
if (rawMaxIter <= 0) {
maxIter = 0;
log.info("Agent {} max_iterations={} → unlimited soft cap (LLM controls termination)",
entity.getId(), rawMaxIter);
} else {
maxIter = Math.min(rawMaxIter, BaseAgent.MAX_ITERATIONS_HARD_CEILING);
if (maxIter != rawMaxIter) {
log.warn("Agent {} max_iterations={} clamped to {} (1..{})",
entity.getId(), rawMaxIter, maxIter, BaseAgent.MAX_ITERATIONS_HARD_CEILING);
}
}
String enhancedPrompt = buildEnhancedPrompt(entity, builtinSearchEnabled,
boundTools, runtimeModel.getMaxInputTokens());
// 当前仅支持 DashScope 和 OpenAI-compatible,其他协议直接拒绝
if (!supportsStateGraph(protocol)) {
throw new MateClawException("err.agent.protocol_not_supported", "当前不支持协议 " + protocol.getId()
+ ",请切换到 DashScope 或 OpenAI-compatible 模型");
}
BaseAgent agent;
boolean toolCallingEnabled;
if ("plan_execute".equals(entity.getAgentType())) {
agent = buildPlanExecuteAgent(toolSet, runtimeModel, maxIter, entity.getId());
toolCallingEnabled = true;
log.info("Built StateGraph Plan-Execute agent: {} (maxIterations={}, tools={}, protocol={})",
entity.getName(), maxIter, toolSet.size(), protocol.getId());
} else {
agent = buildReActAgent(toolSet, runtimeModel, maxIter, entity.getId());
// StateGraph 路径下工具调用由 ActionNode 控制,始终启用
toolCallingEnabled = true;
log.info("Built StateGraph ReAct agent: {} (maxIterations={}, tools={}, protocol={})",
entity.getName(), maxIter, toolSet.size(), protocol.getId());
}
// 设置通用属性
agent.agentId = String.valueOf(entity.getId());
agent.agentName = entity.getName();
agent.systemPrompt = enhancedPrompt;
agent.maxIterations = maxIter;
agent.modelName = runtimeModel.getModelName();
agent.modelCapabilities = modelCapabilityService.resolve(
runtimeModel.getModelName(), runtimeModel.getModalities());
agent.runtimeProviderId = provider != null ? provider.getProviderId() : "";
agent.runtimeModelConfig = runtimeModel;
agent.toolSet = toolSet;
agent.multimodalRouter = multimodalRouter;
agent.mediaCaptionService = mediaCaptionService;
agent.userLocale = resolveLocale();
agent.temperature = runtimeModel.getTemperature();
agent.maxTokens = runtimeModel.getMaxTokens();
agent.maxInputTokens = runtimeModel.getMaxInputTokens();
agent.topP = runtimeModel.getTopP();
agent.toolCallingEnabled = toolCallingEnabled;
// 查找工作区活动目录
if (entity.getWorkspaceId() != null) {
try {
var workspace = workspaceService.getById(entity.getWorkspaceId());
if (workspace != null && workspace.getBasePath() != null && !workspace.getBasePath().isBlank()) {
agent.workspaceBasePath = workspace.getBasePath();
log.info("Agent {} bound to workspace basePath: {}", entity.getName(), agent.workspaceBasePath);
}
} catch (Exception e) {
log.warn("Failed to lookup workspace basePath for agent {}: {}", entity.getName(), e.getMessage());
}
}
log.info("Built agent instance: {} (type={}, protocol={}, tools={}, toolCallingEnabled={})",
entity.getName(), entity.getAgentType(), protocol.getId(),
toolSet.size(), agent.toolCallingEnabled);
return agent;
}
// ==================== Agent 构建方法 ====================
StateGraphReActAgent buildReActAgent(AgentToolSet toolSet, ModelConfigEntity runtimeModel, int maxIter) {
return buildReActAgent(toolSet, runtimeModel, maxIter, null);
}
StateGraphReActAgent buildReActAgent(AgentToolSet toolSet, ModelConfigEntity runtimeModel,
int maxIter, Long agentId) {
ChatModel chatModel = buildRuntimeChatModel(runtimeModel);
ChatClient chatClient = ChatClient.create(chatModel);
String reasoningEffort = resolveReasoningEffortForModel(runtimeModel);
CompiledGraph compiledGraph = buildReActGraph(toolSet, chatModel, maxIter, reasoningEffort, runtimeModel, agentId);
return new StateGraphReActAgent(chatClient, conversationService, compiledGraph,
chatModel, conversationWindowManager, toolSet);
}
StateGraphPlanExecuteAgent buildPlanExecuteAgent(AgentToolSet toolSet, ModelConfigEntity runtimeModel, int maxIter) {
return buildPlanExecuteAgent(toolSet, runtimeModel, maxIter, null);
}
StateGraphPlanExecuteAgent buildPlanExecuteAgent(AgentToolSet toolSet, ModelConfigEntity runtimeModel,
int maxIter, Long agentId) {
ChatModel chatModel = buildRuntimeChatModel(runtimeModel);
ChatClient chatClient = ChatClient.create(chatModel);
String reasoningEffort = resolveReasoningEffortForModel(runtimeModel);
CompiledGraph graph = buildPlanExecuteGraph(toolSet, chatModel, maxIter, reasoningEffort, runtimeModel, agentId);
return new StateGraphPlanExecuteAgent(chatClient, conversationService, graph, planningService,
chatModel, conversationWindowManager, toolSet);
}
CompiledGraph buildPlanExecuteGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations, String reasoningEffort) {
return buildPlanExecuteGraph(toolSet, chatModel, maxIterations, reasoningEffort, null, null);
}
CompiledGraph buildPlanExecuteGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations,
String reasoningEffort, ModelConfigEntity primaryModelConfig) {
return buildPlanExecuteGraph(toolSet, chatModel, maxIterations, reasoningEffort, primaryModelConfig, null);
}
CompiledGraph buildPlanExecuteGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations,
String reasoningEffort, ModelConfigEntity primaryModelConfig,
Long agentId) {
try {
List fallbackChain = buildFallbackChain(primaryModelConfig, agentId);
NodeStreamingChatHelper streamingHelper = new NodeStreamingChatHelper(
streamTracker, fallbackChain, llmCacheMetricsAggregator, providerHealthTracker,
primaryModelConfig != null ? primaryModelConfig.getProvider() : null,
providerPool);
ToolExecutionExecutor executor = new ToolExecutionExecutor(toolSet, toolGuardService, approvalService, streamTracker, toolTimeoutProperties, toolResultStorage, toolConcurrencyRegistry);
// Issue #46: enable skill-aware "Tool not found" hint so when the
// LLM mis-calls a skill name as a tool, the response tells it
// the right invocation pattern instead of a dead-end error.
executor.setSkillRuntimeService(skillRuntimeService);
// Optional: route child-agent denied-tool audit events through
// the audit pipeline. Null when audit is not wired (legacy / test).
if (auditEventService != null) {
executor.setAuditEventService(auditEventService);
}
PlanGenerationNode planGenerationNode = new PlanGenerationNode(chatModel, planningService, streamingHelper, conversationWindowManager, toolSet);
StepExecutionNode stepExecutionNode = new StepExecutionNode(chatModel, toolSet, executor, planningService, streamTracker, reasoningEffort, streamingHelper, conversationWindowManager);
PlanSummaryNode planSummaryNode = new PlanSummaryNode(chatModel, planningService, streamingHelper);
DirectAnswerNode directAnswerNode = new DirectAnswerNode();
KeyStrategyFactory keyStrategyFactory = KeyStrategy.builder()
// 共享键
.addStrategy(MateClawStateKeys.PENDING_EVENTS, KeyStrategy.APPEND)
.addStrategy(MateClawStateKeys.CURRENT_PHASE, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.SYSTEM_PROMPT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.CONVERSATION_ID, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.TRACE_ID, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.AGENT_ID, KeyStrategy.REPLACE)
// 会话消息(复用 ReAct 的 MESSAGES key,APPEND 策略)
.addStrategy(MateClawStateKeys.MESSAGES, KeyStrategy.APPEND)
// Plan 特有键
.addStrategy(PlanStateKeys.GOAL, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.PLAN_ID, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.PLAN_STEPS, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.PLAN_VALID, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.NEEDS_PLANNING, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.CURRENT_STEP_INDEX, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.CURRENT_STEP_TITLE, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.CURRENT_STEP_RESULT, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.COMPLETED_RESULTS, KeyStrategy.APPEND)
.addStrategy(PlanStateKeys.FINAL_SUMMARY, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.DIRECT_ANSWER, KeyStrategy.REPLACE)
// 工作上下文(REPLACE 策略,每次重新生成)
.addStrategy(PlanStateKeys.WORKING_CONTEXT, KeyStrategy.REPLACE)
// Thinking 键
.addStrategy(PlanStateKeys.FINAL_SUMMARY_THINKING, KeyStrategy.REPLACE)
.addStrategy(PlanStateKeys.CURRENT_STEP_THINKING, KeyStrategy.REPLACE)
// 流式防重键
.addStrategy(MateClawStateKeys.CONTENT_STREAMED, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.THINKING_STREAMED, KeyStrategy.REPLACE)
// 流式内容暂存(AWAITING_APPROVAL 路径持久化使用)
.addStrategy(MateClawStateKeys.STREAMED_CONTENT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.STREAMED_THINKING, KeyStrategy.REPLACE)
// 请求者身份(审批身份校验使用)
.addStrategy(MateClawStateKeys.REQUESTER_ID, KeyStrategy.REPLACE)
// 审批重放键
.addStrategy(MateClawStateKeys.FORCED_TOOL_CALL, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.PRE_APPROVED_TOOL_CALL, KeyStrategy.REPLACE)
// RFC-063r §2.5: ChatOrigin must survive every node merge so
// sub-graph nodes (StepExecutionNode + DelegateAgentTool's
// child agents) can read the originating channel binding.
// Without explicit REPLACE the framework's merge drops it
// on multi-iteration paths — root cause of the channel-binding
// flakiness reported on first deployment.
.addStrategy(MateClawStateKeys.CHAT_ORIGIN, KeyStrategy.REPLACE)
// Caught by StateKeyRegistrationCoverageTest — these state keys
// were silently unregistered before the post-deploy audit.
// WORKSPACE_BASE_PATH: written by buildInitialState; sub-graph
// tools read it via WorkspacePathGuard.
// STOP_REQUESTED: external cancel flag checked by every node.
// RETURN_DIRECT_TRIGGERED / DIRECT_TOOL_OUTPUTS (RFC-052):
// Plan-Execute itself doesn't trigger returnDirect, but
// DelegateAgentTool sub-agents could; register defensively.
.addStrategy(MateClawStateKeys.WORKSPACE_BASE_PATH, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.STOP_REQUESTED, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.RETURN_DIRECT_TRIGGERED, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.DIRECT_TOOL_OUTPUTS, KeyStrategy.REPLACE)
// Token Usage
.addStrategy(MateClawStateKeys.PROMPT_TOKENS, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.COMPLETION_TOKENS, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.RUNTIME_MODEL_NAME, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.RUNTIME_PROVIDER_ID, KeyStrategy.REPLACE)
// SourceEvidenceLedger: ActionNode 把每轮 ToolResponse 抽取出的
// (sourcePaths, sourceSymbols, failedPaths) merge 进这个 ledger,
// 后续 ReasoningNode / FinalAnswerNode 调 validateAnswer 校验
// 模型引用是否有真实证据。漏注册时框架在多 node merge 时会偶发
// 丢这个键,evidence_insufficient 检查会"静默地不生效" ——
// StateKeyRegistrationCoverageTest 专门兜这条。
.addStrategy(MateClawStateKeys.SOURCE_EVIDENCE_LEDGER, KeyStrategy.REPLACE)
// Multimodal sidecar routing decision for the current turn.
.addStrategy(MateClawStateKeys.ROUTING_DECISION, KeyStrategy.REPLACE)
.build();
// Graph 拓扑:
// START → PLAN_GENERATION → (PlanGenerationDispatcher)
// ├→ DIRECT_ANSWER_NODE → END
// └→ STEP_EXECUTION → (StepProgressDispatcher)
// ├→ STEP_EXECUTION (loop)
// └→ PLAN_SUMMARY → END
StateGraph graph = new StateGraph("plan-execute-agent", keyStrategyFactory)
.addNode(PlanStateKeys.PLAN_GENERATION_NODE,
AsyncNodeAction.node_async(planGenerationNode))
.addNode(PlanStateKeys.STEP_EXECUTION_NODE,
AsyncNodeAction.node_async(stepExecutionNode))
.addNode(PlanStateKeys.PLAN_SUMMARY_NODE,
AsyncNodeAction.node_async(planSummaryNode))
.addNode(PlanStateKeys.DIRECT_ANSWER_NODE,
AsyncNodeAction.node_async(directAnswerNode))
.addEdge(StateGraph.START, PlanStateKeys.PLAN_GENERATION_NODE)
.addConditionalEdges(PlanStateKeys.PLAN_GENERATION_NODE,
AsyncEdgeAction.edge_async(new PlanGenerationDispatcher()),
Map.of(
PlanStateKeys.STEP_EXECUTION_NODE, PlanStateKeys.STEP_EXECUTION_NODE,
PlanStateKeys.DIRECT_ANSWER_NODE, PlanStateKeys.DIRECT_ANSWER_NODE))
.addConditionalEdges(PlanStateKeys.STEP_EXECUTION_NODE,
AsyncEdgeAction.edge_async(new StepProgressDispatcher()),
Map.of(
PlanStateKeys.STEP_EXECUTION_NODE, PlanStateKeys.STEP_EXECUTION_NODE,
PlanStateKeys.PLAN_SUMMARY_NODE, PlanStateKeys.PLAN_SUMMARY_NODE,
StateGraph.END, StateGraph.END))
.addEdge(PlanStateKeys.PLAN_SUMMARY_NODE, StateGraph.END)
.addEdge(PlanStateKeys.DIRECT_ANSWER_NODE, StateGraph.END);
return graph.compile(CompileConfig.builder()
.recursionLimit(frameworkRecursionLimit())
.build());
} catch (Exception e) {
throw new MateClawException("err.agent.plan_compile_failed", "Plan-Execute StateGraph 编译失败: " + e.getMessage());
}
}
/**
* Hard ceiling for the underlying graph framework's recursion guard.
*
* The framework treats "recursion limit reached" as a normal completion —
* it emits a {@code done} signal with no exception and no log. That makes
* it indistinguishable from a real final answer downstream, and is the
* mechanism by which a turn can silently stop mid-execution and persist
* only whatever partial content the accumulator happened to hold.
*
* To avoid that class of bug, the recursion limit must be sized so it can
* never trip before the soft cap (ObservationDispatcher →
* LimitExceededNode), which is the only path that produces a proper
* {@code finish_reason} and human-facing message. Sized for the maximum
* effective soft cap (DB hard ceiling + thinking-mode bonus) multiplied
* by 4 (each iteration is worst-case reasoning + summarizing + action +
* observation) plus a 100-step buffer for phase nodes, approval replays
* and tool-result chunking. Decoupled from the per-agent value so a small
* {@code max_iterations} can never accidentally re-introduce the silent
* killer.
*/
private static int frameworkRecursionLimit() {
return (BaseAgent.MAX_ITERATIONS_HARD_CEILING + 5) * 4 + 100;
}
CompiledGraph buildReActGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations, String reasoningEffort) {
return buildReActGraph(toolSet, chatModel, maxIterations, reasoningEffort, null, null);
}
CompiledGraph buildReActGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations,
String reasoningEffort, ModelConfigEntity primaryModelConfig) {
return buildReActGraph(toolSet, chatModel, maxIterations, reasoningEffort, primaryModelConfig, null);
}
CompiledGraph buildReActGraph(AgentToolSet toolSet, ChatModel chatModel, int maxIterations,
String reasoningEffort, ModelConfigEntity primaryModelConfig,
Long agentId) {
try {
List fallbackChain = buildFallbackChain(primaryModelConfig, agentId);
NodeStreamingChatHelper streamingHelper = new NodeStreamingChatHelper(
streamTracker, fallbackChain, llmCacheMetricsAggregator, providerHealthTracker,
primaryModelConfig != null ? primaryModelConfig.getProvider() : null,
providerPool);
ToolExecutionExecutor executor = new ToolExecutionExecutor(toolSet, toolGuardService, approvalService, streamTracker, toolTimeoutProperties, toolResultStorage, toolConcurrencyRegistry);
// Issue #46: enable skill-aware "Tool not found" hint so when the
// LLM mis-calls a skill name as a tool, the response tells it
// the right invocation pattern instead of a dead-end error.
executor.setSkillRuntimeService(skillRuntimeService);
// Optional: route child-agent denied-tool audit events through
// the audit pipeline. Null when audit is not wired (legacy / test).
if (auditEventService != null) {
executor.setAuditEventService(auditEventService);
}
// PR-1.2 (RFC-049 L1-B): propagate the bound model's capability so ReasoningNode
// can gate the ThinkingLevelHolder override explicitly, rather than inferring
// capability from reasoningEffort == null.
boolean supportsReasoningEffort = primaryModelConfig != null
&& ModelFamily.detect(primaryModelConfig.getModelName()).supportsReasoningEffort();
ReasoningNode reasoningNode = new ReasoningNode(chatModel, toolSet, reasoningEffort,
supportsReasoningEffort,
streamingHelper, conversationWindowManager, streamTracker, 0, wikiContextService);
ActionNode actionNode = new ActionNode(executor, streamTracker);
ObservationProcessor observationProcessor = new ObservationProcessor(graphObservationProperties);
ObservationNode observationNode = new ObservationNode(observationProcessor, streamTracker);
SummarizingNode summarizingNode = new SummarizingNode(chatModel, streamingHelper, streamTracker);
LimitExceededNode limitExceededNode = new LimitExceededNode(chatModel, observationProcessor, streamingHelper, i18nService);
FinalAnswerNode finalAnswerNode = new FinalAnswerNode(generatedFileCache);
KeyStrategyFactory keyStrategyFactory = KeyStrategy.builder()
// 输入字段
.addStrategy(MateClawStateKeys.USER_MESSAGE, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.CONVERSATION_ID, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.SYSTEM_PROMPT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.AGENT_ID, KeyStrategy.REPLACE)
// 消息列表(追加策略)
.addStrategy(MateClawStateKeys.MESSAGES, KeyStrategy.APPEND)
// 迭代控制
.addStrategy(MateClawStateKeys.CURRENT_ITERATION, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.MAX_ITERATIONS, KeyStrategy.REPLACE)
// 工具调用
.addStrategy(MateClawStateKeys.TOOL_CALLS, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.TOOL_RESULTS, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.TOOL_CALL_COUNT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.LLM_CALL_COUNT, KeyStrategy.REPLACE)
// 控制流
.addStrategy(MateClawStateKeys.FINAL_ANSWER, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.NEEDS_TOOL_CALL, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.ERROR, KeyStrategy.REPLACE)
// 观察历史(REPLACE 策略,由 ObservationNode 手动累加,SummarizingNode 可清空)
.addStrategy(MateClawStateKeys.OBSERVATION_HISTORY, KeyStrategy.REPLACE)
// Summarizing
.addStrategy(MateClawStateKeys.SUMMARIZED_CONTEXT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.FINAL_ANSWER_DRAFT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.SHOULD_SUMMARIZE, KeyStrategy.REPLACE)
// 终止控制
.addStrategy(MateClawStateKeys.FINISH_REASON, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.LIMIT_EXCEEDED, KeyStrategy.REPLACE)
// 统计与追踪
.addStrategy(MateClawStateKeys.ERROR_COUNT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.TRACE_ID, KeyStrategy.REPLACE)
// 事件流
.addStrategy(MateClawStateKeys.PENDING_EVENTS, KeyStrategy.APPEND)
.addStrategy(MateClawStateKeys.CURRENT_PHASE, KeyStrategy.REPLACE)
// Thinking
.addStrategy(MateClawStateKeys.FINAL_THINKING, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.CURRENT_THINKING, KeyStrategy.REPLACE)
// 流式防重
.addStrategy(MateClawStateKeys.CONTENT_STREAMED, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.THINKING_STREAMED, KeyStrategy.REPLACE)
// 审批控制
.addStrategy(MateClawStateKeys.AWAITING_APPROVAL, KeyStrategy.REPLACE)
// 流式内容暂存(AWAITING_APPROVAL 路径持久化使用)
.addStrategy(MateClawStateKeys.STREAMED_CONTENT, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.STREAMED_THINKING, KeyStrategy.REPLACE)
// 请求者身份(审批身份校验使用)
.addStrategy(MateClawStateKeys.REQUESTER_ID, KeyStrategy.REPLACE)
// 审批重放
.addStrategy(MateClawStateKeys.FORCED_TOOL_CALL, KeyStrategy.REPLACE)
// RFC-063r §2.5: ChatOrigin must survive every node merge so
// ActionNode (and DelegateAgentTool's child agents) can read
// the originating channel binding across multi-iteration ReAct
// loops. Without explicit REPLACE the framework's merge drops
// it after the first node transition — root cause of the
// channel-binding flakiness reported on first deployment.
.addStrategy(MateClawStateKeys.CHAT_ORIGIN, KeyStrategy.REPLACE)
// Caught by StateKeyRegistrationCoverageTest — silently
// unregistered before the audit. WORKSPACE_BASE_PATH from
// initial state; STOP_REQUESTED is the external cancel flag;
// RETURN_DIRECT_TRIGGERED / DIRECT_TOOL_OUTPUTS are RFC-052
// returnDirect short-circuit signals consumed by ObservationDispatcher.
.addStrategy(MateClawStateKeys.WORKSPACE_BASE_PATH, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.STOP_REQUESTED, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.RETURN_DIRECT_TRIGGERED, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.DIRECT_TOOL_OUTPUTS, KeyStrategy.REPLACE)
// Token Usage
.addStrategy(MateClawStateKeys.PROMPT_TOKENS, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.COMPLETION_TOKENS, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.RUNTIME_MODEL_NAME, KeyStrategy.REPLACE)
.addStrategy(MateClawStateKeys.RUNTIME_PROVIDER_ID, KeyStrategy.REPLACE)
// SourceEvidenceLedger: ActionNode 把每轮 ToolResponse 抽取出的
// (sourcePaths, sourceSymbols, failedPaths) merge 进这个 ledger,
// 后续 ReasoningNode / FinalAnswerNode 调 validateAnswer 校验
// 模型引用是否有真实证据。漏注册时框架在多 node merge 时会偶发
// 丢这个键,evidence_insufficient 检查会"静默地不生效" ——
// StateKeyRegistrationCoverageTest 专门兜这条。
.addStrategy(MateClawStateKeys.SOURCE_EVIDENCE_LEDGER, KeyStrategy.REPLACE)
// Multimodal sidecar routing decision for the current turn.
.addStrategy(MateClawStateKeys.ROUTING_DECISION, KeyStrategy.REPLACE)
.build();
StateGraph graph = new StateGraph("react-agent-v2", keyStrategyFactory)
.addNode(MateClawStateKeys.REASONING_NODE,
AsyncNodeAction.node_async(reasoningNode))
.addNode(MateClawStateKeys.ACTION_NODE,
AsyncNodeAction.node_async(actionNode))
.addNode(MateClawStateKeys.OBSERVATION_NODE,
AsyncNodeAction.node_async(observationNode))
.addNode(MateClawStateKeys.SUMMARIZING_NODE,
AsyncNodeAction.node_async(summarizingNode))
.addNode(MateClawStateKeys.LIMIT_EXCEEDED_NODE,
AsyncNodeAction.node_async(limitExceededNode))
.addNode(MateClawStateKeys.FINAL_ANSWER_NODE,
AsyncNodeAction.node_async(finalAnswerNode))
.addEdge(StateGraph.START, MateClawStateKeys.REASONING_NODE)
.addConditionalEdges(MateClawStateKeys.REASONING_NODE,
AsyncEdgeAction.edge_async(new ReasoningDispatcher()),
Map.of(MateClawStateKeys.ACTION_NODE, MateClawStateKeys.ACTION_NODE,
MateClawStateKeys.SUMMARIZING_NODE, MateClawStateKeys.SUMMARIZING_NODE,
MateClawStateKeys.FINAL_ANSWER_NODE, MateClawStateKeys.FINAL_ANSWER_NODE,
MateClawStateKeys.LIMIT_EXCEEDED_NODE, MateClawStateKeys.LIMIT_EXCEEDED_NODE))
.addEdge(MateClawStateKeys.ACTION_NODE, MateClawStateKeys.OBSERVATION_NODE)
.addConditionalEdges(MateClawStateKeys.OBSERVATION_NODE,
AsyncEdgeAction.edge_async(new ObservationDispatcher()),
Map.of(MateClawStateKeys.REASONING_NODE, MateClawStateKeys.REASONING_NODE,
MateClawStateKeys.SUMMARIZING_NODE, MateClawStateKeys.SUMMARIZING_NODE,
MateClawStateKeys.LIMIT_EXCEEDED_NODE, MateClawStateKeys.LIMIT_EXCEEDED_NODE,
MateClawStateKeys.FINAL_ANSWER_NODE, MateClawStateKeys.FINAL_ANSWER_NODE))
.addEdge(MateClawStateKeys.SUMMARIZING_NODE, MateClawStateKeys.REASONING_NODE)
.addEdge(MateClawStateKeys.LIMIT_EXCEEDED_NODE, MateClawStateKeys.FINAL_ANSWER_NODE)
.addEdge(MateClawStateKeys.FINAL_ANSWER_NODE, StateGraph.END);
return graph.compile(CompileConfig.builder()
.recursionLimit(frameworkRecursionLimit())
.withLifecycleListener(new ReActLifecycleListener())
.build());
} catch (Exception e) {
throw new MateClawException("err.agent.graph_compile_failed", "StateGraph v2 编译失败: " + e.getMessage());
}
}
// ==================== 协议能力判断 ====================
private boolean supportsStateGraph(ModelProtocol protocol) {
return protocol == ModelProtocol.DASHSCOPE_NATIVE
|| protocol == ModelProtocol.OPENAI_COMPATIBLE
|| protocol == ModelProtocol.ANTHROPIC_MESSAGES
// RFC-062: Claude Code OAuth tunnels through the same Messages API
// wrapped in AnthropicChatModel — same StateGraph capability surface.
|| protocol == ModelProtocol.ANTHROPIC_CLAUDE_CODE
|| protocol == ModelProtocol.OPENAI_CHATGPT;
}
// ==================== 模型构建 ====================
/**
* 构建运行时 ChatModel(不包装为 ChatClient)
* 用于 StateGraph 节点直接调用。使用注入的共享 {@link #retryTemplate} 作为 Spring AI
* 内层重试策略。
*/
public ChatModel buildRuntimeChatModel(ModelConfigEntity runtimeModel) {
return buildRuntimeChatModel(runtimeModel, this.retryTemplate);
}
/**
* Resolve the user-facing locale used for sidecar caption prompts.
* Reads {@code language} from system settings; falls back to
* {@code zh-CN} so CN deployments stay consistent with the chat UI.
*/
private java.util.Locale resolveLocale() {
try {
String lang = systemSettingService.getLanguage();
if (lang == null || lang.isBlank()) return java.util.Locale.SIMPLIFIED_CHINESE;
return java.util.Locale.forLanguageTag(lang);
} catch (Exception e) {
return java.util.Locale.SIMPLIFIED_CHINESE;
}
}
/**
* 构建运行时 ChatModel,并指定自定义的 Spring AI {@link RetryTemplate}。
*
* 用于调用方(如 Wiki 消化管线)已经有自己的外层重试策略,
* 希望绕过 Spring AI 内层重试、独占重试控制权的场景:传入
* {@code RetryTemplate.builder().maxAttempts(1).build()} 即可把内层降级为"只跑一次"。
*
* DashScope 和 OpenAI-ChatGPT 分支不走 Spring AI 的 RetryTemplate 接口,
* 本参数对它们无效(它们各自有内部重试或直通)。
*/
public ChatModel buildRuntimeChatModel(ModelConfigEntity runtimeModel, RetryTemplate retryOverride) {
// PR-0 (RFC-009 Phase 4 prelude): protocol switch extracted to
// ProviderChatModelFactory + per-protocol ChatModelBuilder strategies.
// Per-protocol builders (DashScope / OpenAI-compatible / Anthropic /
// ChatGPT-Responses) live in vip.mate.agent.chatmodel + vip.mate.llm.chatmodel.
// See RFC-009 Phase 4 plan for the rationale (circular-dep break for
// ProviderInitProbe + AgentGraphBuilder slimming).
return chatModelFactory.buildFor(runtimeModel, retryOverride);
}
/**
* RFC-009: build the full multi-provider failover chain for a primary
* model. Providers are read from {@code mate_model_provider} ordered by
* {@code fallback_priority ASC} (positive values only), each resolved to
* its default {@link ModelConfigEntity} and turned into a {@link ChatModel}
* via {@link #buildRuntimeChatModel(ModelConfigEntity, RetryTemplate)}.
*
*
Providers whose API key / base URL is missing (build throws) are
* silently skipped with a warning — fallback should never break
* the primary call path. The returned list preserves chain order; the
* streaming helper tries entries in order until one succeeds.
*
*
The primary model is excluded from the chain when its provider +
* model name matches a chain entry. Previously only reference equality
* was checked, which meant a DashScope-primary deployment ended up with
* {@code null} fallback — exactly the case RFC-009 targets.
*
* @param primaryModelConfig the {@code ModelConfigEntity} used to build
* the primary model; used to identity-filter the chain
* @return ordered, possibly-empty list of fallback {@link ChatModel}s
*/
List buildFallbackChain(ModelConfigEntity primaryModelConfig) {
return buildFallbackChain(primaryModelConfig, null);
}
/**
* RFC-009 PR-3 overload: when {@code agentId} is non-null, the agent's
* {@code mate_agent_provider_preference} rows bias the chain order — listed
* providers come first in their declared {@code sort_order}, then the
* remaining providers fall in by global {@code fallback_priority} ascending,
* tie-broken by provider id alphabetically. {@code null} agentId keeps the
* pre-PR-3 ordering (pure global priority) — that's the path for legacy
* callers and tests.
*
*
Source = the available pool (RFC-009 follow-up). Earlier this
* method only considered providers with {@code fallback_priority > 0}, which
* meant any provider the user hadn't explicitly opted into the chain was
* silently excluded — even if it was healthy and in the pool. The pool is
* the source of truth for "what's usable right now"; {@code fallback_priority}
* is just an ordering hint within the pool.
*
*
Per-provider model selection falls back gracefully: the
* provider's {@code is_default=true} chat model wins, otherwise we pick
* the first enabled chat model on that provider. Forcing users to mark a
* default per provider was administrative friction with no real benefit.
*/
List buildFallbackChain(ModelConfigEntity primaryModelConfig,
Long agentId) {
List providers;
try {
// Pull every configured provider, not just the ones with
// fallback_priority > 0 — pool membership is what gates usability,
// not this admin-set hint.
providers = modelProviderService.listProviders().stream()
.filter(dto -> Boolean.TRUE.equals(dto.getConfigured()))
.map(dto -> {
try {
return modelProviderService.getProviderConfig(dto.getId());
} catch (Exception e) {
return null;
}
})
.filter(java.util.Objects::nonNull)
.collect(java.util.stream.Collectors.toCollection(ArrayList::new));
} catch (Exception e) {
log.warn("[LlmFailover] failed to load configured providers: {}; running without fallback",
e.getMessage());
return List.of();
}
if (providers.isEmpty()) {
return List.of();
}
// Order: explicit fallback_priority > 0 wins (asc), priority == 0 trails alphabetically.
providers.sort((a, b) -> {
int pa = a.getFallbackPriority() == null ? 0 : a.getFallbackPriority();
int pb = b.getFallbackPriority() == null ? 0 : b.getFallbackPriority();
if (pa > 0 && pb > 0) return Integer.compare(pa, pb);
if (pa > 0) return -1; // a has explicit priority, comes first
if (pb > 0) return 1; // b has explicit priority, comes first
return a.getProviderId().compareTo(b.getProviderId()); // both 0: alphabetical
});
String primaryProviderId = primaryModelConfig != null ? primaryModelConfig.getProvider() : null;
String primaryModelName = primaryModelConfig != null ? primaryModelConfig.getModelName() : null;
// RFC-009 PR-3: bias by agent preferences (if any). Listed providers win
// their declared order; everything else keeps the global priority order.
List preferred = agentId == null
? java.util.Collections.emptyList()
: agentBindingService.getPreferredProviderIds(agentId);
if (!preferred.isEmpty()) {
providers = reorderByPreferences(providers, preferred);
log.debug("[LlmFailover] agent={} preferences={} -> chain head reordered", agentId, preferred);
}
// RFC-090 §9.2 调整 C — second-pass reorder: lift providers
// that satisfy the bound-skill capability set (vision / video /
// audio) ahead of those that don't. Stable otherwise so the
// user-preferred order still wins among capable providers.
try {
providers = new ArrayList<>(providerRouter.reorderForCapabilities(agentId, providers));
} catch (Exception e) {
log.debug("[ProviderRouter] chain reorder failed: {}", e.getMessage());
}
List chain = new ArrayList<>();
for (ModelProviderEntity p : providers) {
// Don't put the primary provider's row into the fallback chain — same-instance
// skipping is also done in the runtime walker, but excluding here saves building
// a duplicate ChatModel at agent-build time.
if (primaryProviderId != null && primaryProviderId.equals(p.getProviderId())) {
log.debug("[LlmFailover] skipping primary provider {} in fallback chain", primaryProviderId);
continue;
}
// RFC-009 Phase 4: skip providers known-bad at build time. The runtime walker in
// NodeStreamingChatHelper re-checks pool membership per request, so a provider
// that re-enters the pool later still gets used (the graph is rebuilt on
// ModelConfigChangedEvent).
if (providerPool != null && !providerPool.contains(p.getProviderId())) {
log.debug("[LlmFailover] skipping provider {} — not in available pool",
p.getProviderId());
continue;
}
ModelConfigEntity fallbackConfig = pickFallbackModel(p.getProviderId());
if (fallbackConfig == null) {
log.debug("[LlmFailover] skipping provider {} — no enabled chat model",
p.getProviderId());
continue;
}
if (primaryModelName != null && primaryModelName.equals(fallbackConfig.getModelName())) {
// Same model name picked for a different provider — exact same call, skip.
continue;
}
try {
ChatModel m = buildRuntimeChatModel(fallbackConfig, RetryTemplate.builder().maxAttempts(1).build());
chain.add(new vip.mate.llm.failover.FallbackEntry(p.getProviderId(), m));
log.info("[LlmFailover] chain[{}] = {}/{} (priority={})",
chain.size(), p.getProviderId(), fallbackConfig.getModelName(),
p.getFallbackPriority());
} catch (Exception e) {
log.warn("[LlmFailover] skipping provider {} — chat model build failed: {}",
p.getProviderId(), e.getMessage());
}
}
return chain;
}
/**
* Pick a chat model to use as a fallback for the given provider:
*
*
Provider's explicit default ({@code is_default=true}) — most user-aligned.
*
First enabled chat model on the provider — pragmatic fallback so the user
* isn't required to mark a default per provider just to participate in failover.
*
* Returns {@code null} when the provider has no usable chat model.
*/
private ModelConfigEntity pickFallbackModel(String providerId) {
try {
ModelConfigEntity defaultModel = modelConfigService.getDefaultModelByProvider(providerId);
if (defaultModel != null) return defaultModel;
} catch (Exception ignored) {
// No default — fall through to first-enabled lookup.
}
try {
return modelConfigService.listModelsByProvider(providerId).stream()
.filter(m -> Boolean.TRUE.equals(m.getEnabled()))
.filter(m -> m.getModelType() == null || "chat".equals(m.getModelType()))
.findFirst()
.orElse(null);
} catch (Exception e) {
log.warn("[LlmFailover] cannot list models for provider {}: {}", providerId, e.getMessage());
return null;
}
}
/**
* Reorder a provider list by an agent's preference list. Listed provider
* ids come first in their preference order; any provider not in the
* preference list keeps its original position relative to other unlisted
* providers (stable partition). Preference entries that don't match any
* actual provider are silently dropped.
*/
/** Package-private for unit testing — see {@code AgentGraphBuilderPreferenceTest}. */
static List reorderByPreferences(List providers,
List preferredOrder) {
Map byId = new java.util.LinkedHashMap<>();
for (ModelProviderEntity p : providers) {
byId.put(p.getProviderId(), p);
}
List reordered = new ArrayList<>(providers.size());
Set placed = new java.util.HashSet<>();
for (String prefId : preferredOrder) {
ModelProviderEntity p = byId.get(prefId);
if (p != null && placed.add(prefId)) {
reordered.add(p);
}
}
for (ModelProviderEntity p : providers) {
if (placed.add(p.getProviderId())) {
reordered.add(p);
}
}
return reordered;
}
/**
* Finds the first enabled chat model whose provider is fully configured.
* Used as a fallback when the default model's provider is not available.
*/
private ModelConfigEntity findFirstAvailableChatModel() {
return modelConfigService.listByType("chat").stream()
.filter(m -> Boolean.TRUE.equals(m.getEnabled()))
.filter(m -> {
try {
return modelProviderService.isProviderConfigured(m.getProvider());
} catch (Exception e) {
return false;
}
})
.findFirst()
.orElse(null);
}
// PR-0b: legacy single-fallback buildFallbackModel deleted (already @Deprecated, no callers).
// PR-0b: isDashScopeSearchEnabled moved to AgentDashScopeChatModelBuilder.
// ==================== Prompt 构建 ====================
private String buildEnhancedPrompt(AgentEntity entity, boolean builtinSearchEnabled,
Set boundTools, Integer maxInputTokens) {
// The agent's own systemPrompt encodes its identity (role / goal /
// backstory). The memory block from workspace files (AGENTS.md, SOUL.md,
// PROFILE.md, MEMORY.md, ...) augments that identity with durable
// context. Both are independently optional, but when both exist they
// must be joined — earlier this branch picked memory and silently
// dropped the identity prompt, so editor-side identity changes never
// reached runtime if the agent had any workspace files.
String identityPrompt = entity.getSystemPrompt() != null ? entity.getSystemPrompt().trim() : "";
String memoryPrompt = memoryManager.buildSystemPromptBlock(entity.getId());
StringBuilder basePromptBuilder = new StringBuilder();
if (!identityPrompt.isEmpty()) {
basePromptBuilder.append(identityPrompt);
}
if (memoryPrompt != null && !memoryPrompt.isBlank()) {
if (basePromptBuilder.length() > 0) {
basePromptBuilder.append("\n\n");
}
basePromptBuilder.append(memoryPrompt);
}
String basePrompt = basePromptBuilder.toString();
// 使用 skill runtime 构建技能增强(per-agent 绑定过滤)
Set boundSkillIds = agentBindingService.getBoundSkillIds(entity.getId());
String skillEnhancement = skillRuntimeService.buildSkillPromptEnhancement(
boundSkillIds, boundTools, maxInputTokens, entity.getId());
// 工具调用指导
String toolGuidance = """
## Runtime Context
- Current Agent ID: %s
## Workspace Memory Guidelines
Your durable memory is stored in database-backed workspace markdown files for this agent:
- `PROFILE.md`: stable user profile, preferences, collaboration style
- `MEMORY.md`: distilled long-term memory, durable facts, lessons, recurring patterns
- `memory/YYYY-MM-DD.md`: daily notes, raw events, temporary observations, open loops
Use workspace memory tools instead of local filesystem tools for those files:
- `list_workspace_memory_files(agentId=..., filenamePrefix=...)`
- `read_workspace_memory_file(agentId=..., filename=...)`
- `write_workspace_memory_file(agentId=..., filename=..., content=...)`
- `edit_workspace_memory_file(agentId=..., filename=..., oldText=..., newText=...)`
Memory writing policy:
- Stable user preference, identity, collaboration habit -> `PROFILE.md`
- Stable project fact, workflow, tool setup, lesson learned, recurring decision -> `MEMORY.md`
- One-off event, meeting note, temporary context, today's decision trace -> `memory/YYYY-MM-DD.md`
- Read before write unless you are creating a brand new daily note
- Do not store secrets or highly sensitive data unless the user explicitly asks
- Updating workspace memory files is internal state maintenance for this agent and can be done proactively when useful
Memory emergence policy:
- If the same preference, constraint, workflow, or lesson appears repeatedly, consolidate it from daily notes into `MEMORY.md`
- Prefer updating an existing section over appending duplicate bullets
- Treat `MEMORY.md` as a compact mental model, not a raw transcript dump
- When answering tasks involving prior decisions, preferences, habits, or ongoing work, proactively consult relevant workspace memory first
## Structured Memory Tools
For discrete, typed facts use structured memory tools (separate from workspace files):
- `remember_structured(agentId, type, key, content)` — store a typed entry
- `recall_structured(agentId, type, keyword)` — search entries by type and/or keyword
- `forget_structured(agentId, type, key)` — remove an entry
Types:
- `user`: preferences, expertise, communication style, role
- `feedback`: behavioral corrections or confirmed approaches (include WHY)
- `project`: decisions, deadlines, constraints not derivable from code/git
- `reference`: pointers to external systems (Linear boards, Grafana dashboards, Slack channels)
Use workspace memory tools (MEMORY.md, daily notes) for long-form narrative notes.
Use structured memory tools for key-value facts the system can query efficiently.
## Session Search
- `session_search(agentId, currentConversationId, mode, query, limit)` — search conversation history
- mode="recent": list recent conversations (titles, times, message counts)
- mode="search": keyword full-text search across past messages
- Use this to recall previous discussions, look up past decisions, or find context from earlier conversations
## Tool Usage Guidelines
When you have available tools, use them to access local system information, files, or execute commands.
Do not assume you cannot access local resources - try calling the appropriate tool first.
If a tool requires approval due to security policies, the system will prompt the user for confirmation.
Only state you cannot access something if no relevant tool is available.
## Multi-Part Question Guidelines
When the user asks multiple questions or requests multiple tasks in a single message:
1. Structure your final answer with numbered sections, one per sub-task
2. Each section must contain the complete, detailed result for that sub-task
3. Never compress earlier sub-tasks into summary sentences while expanding the last one
4. If observations were summarized during processing, reconstruct each section from the summary
5. Treat each sub-task's result as equally important regardless of processing order
## File Reading Guidelines
**Text Files** (use read_file):
For .txt, .md, .json, .yaml, .csv, .log, .py, .java, .js, .html, .xml, .sql, .conf, .ini, .toml files.
**Office/PDF Documents** (DO NOT use read_file):
For .pdf, .docx, .doc, .xlsx, .xls, .pptx, .ppt files, NEVER use read_file.
Instead use:
- detect_file_type(filePath="...") - to check file type first
- extract_document_text(filePath="...") - general document extraction
- extract_pdf_text(filePath="...") - for PDF files
- extract_docx_text(filePath="...") - for Word documents
Example workflow for document:
1. detect_file_type(filePath="/path/to/document.pdf")
2. Based on result, use extract_pdf_text() or extract_document_text()
3. Process the extracted text content
If you try to read a PDF/Office file with read_file, you will get binary garbage or an error.
""".formatted(entity.getId());
// Web-search vs browser_use priority guidance — emitted unconditionally so the rule
// also reaches OpenAI-compatible / Anthropic / Gemini / DeepSeek / Ollama agents that
// do not have builtin search. Issue #40: without this rule the model treats
// browser_use as a search tool and gets stuck in a Playwright launch loop on Windows.
String searchGuidance = """
## Web Search Capability
### Tool Priority
- For plain web search or fetching public page content, call the `search` tool. It supports advanced parameters: `freshness` (day/week/month/year), `language` (zh-CN/en), `count` (1-10).
- Call `browser_use` ONLY when you need to interact with a page (click, fill forms, screenshot, run JS, follow a logged-in flow). Do NOT use `browser_use` as a search alternative.
- **NEVER** call both `browser_use` and `search` for the same query.
- When searching for news, use the standard format: `📰 [Category] Title — Source | Time + Summary`, up to 5 results per category.
""";
if (builtinSearchEnabled) {
searchGuidance += """
### Built-in Search (preferred when available)
Your responses automatically incorporate live web search results from the model provider. For most queries, answer directly — your reply already includes real-time search data. Do NOT say you cannot search.
Use the `search` tool ONLY when you need precise time filtering (e.g., "yesterday's news" → freshness=day), a specific language, or when built-in results feel insufficient.
""";
}
// Wiki 知识库上下文注入
String wikiContext = wikiContextService.buildWikiContext(entity.getId());
return basePrompt + skillEnhancement + toolGuidance + searchGuidance + wikiContext;
}
// ==================== 模型选项构建 ====================
// PR-0b: buildDashScopeOptions moved to AgentDashScopeChatModelBuilder
/** Transitional public visibility for {@code chatmodel} sub-package builders; will move into the builder in PR-0c (OpenAI). */
public OpenAiChatOptions buildOpenAiOptions(ModelConfigEntity runtimeModel, ModelProviderEntity provider) {
OpenAiChatOptions.Builder builder = OpenAiChatOptions.builder();
Map kwargs = modelProviderService.readProviderGenerateKwargs(provider);
String modelName = runtimeModel.getModelName();
ModelFamily family = ModelFamily.detect(modelName);
if (StringUtils.hasText(modelName)) {
builder.model(modelName);
}
// temperature:部分模型族强制 1.0
Double temperature = resolveOpenAiTemperature(modelName, runtimeModel.getTemperature(), kwargs, family);
if (temperature != null) {
builder.temperature(temperature);
}
// max_tokens / max_completion_tokens:按模型族路由
if (family.suppressMaxTokens()) {
// OPENAI_REASONING 族:禁止 max_tokens,改用 max_completion_tokens
// fallback 优先级:kwargs.maxCompletionTokens > kwargs.maxTokens > config.maxTokens
Integer kwargsMaxTokens = resolveIntegerOption("maxTokens", runtimeModel.getMaxTokens(), kwargs);
Integer maxCompletionTokens = resolveIntegerOption("maxCompletionTokens", kwargsMaxTokens, kwargs);
if (maxCompletionTokens != null) {
builder.maxCompletionTokens(maxCompletionTokens);
}
log.debug("ModelFamily {} suppressed max_tokens, using max_completion_tokens={} for model {}",
family, maxCompletionTokens, modelName);
} else {
// 其他模型族:正常使用 max_tokens
Integer maxTokens = resolveIntegerOption("maxTokens", runtimeModel.getMaxTokens(), kwargs);
if (maxTokens != null) {
builder.maxTokens(maxTokens);
}
// 仍允许通过 generateKwargs 手动指定 maxCompletionTokens
Integer maxCompletionTokens = resolveIntegerOption("maxCompletionTokens", null, kwargs);
if (maxCompletionTokens != null) {
builder.maxCompletionTokens(maxCompletionTokens);
}
}
// top_p:部分模型族禁止发送
Double topP = resolveOpenAiTopP(modelName, runtimeModel.getTopP(), kwargs, family);
if (topP != null) {
builder.topP(topP);
}
// reasoning_effort:仅支持的模型族才注入
String reasoningEffort = resolveReasoningEffort(modelName, kwargs, family);
if (StringUtils.hasText(reasoningEffort)) {
builder.reasoningEffort(reasoningEffort);
}
// 内置搜索:模型级字段优先,provider generateKwargs 作为 fallback
boolean searchEnabled = Boolean.TRUE.equals(runtimeModel.getEnableSearch())
|| Boolean.TRUE.equals(kwargs.get("enableSearch"));
if (searchEnabled) {
String strategy = runtimeModel.getSearchStrategy();
if (!StringUtils.hasText(strategy)) {
strategy = (String) kwargs.get("searchStrategy");
}
OpenAiApi.ChatCompletionRequest.WebSearchOptions.SearchContextSize contextSize;
try {
contextSize = StringUtils.hasText(strategy)
? OpenAiApi.ChatCompletionRequest.WebSearchOptions.SearchContextSize.valueOf(strategy.toUpperCase())
: OpenAiApi.ChatCompletionRequest.WebSearchOptions.SearchContextSize.MEDIUM;
} catch (IllegalArgumentException e) {
contextSize = OpenAiApi.ChatCompletionRequest.WebSearchOptions.SearchContextSize.MEDIUM;
}
builder.webSearchOptions(new OpenAiApi.ChatCompletionRequest.WebSearchOptions(contextSize, null));
}
OpenAiChatOptions options = builder.build();
options.setInternalToolExecutionEnabled(false);
// 注意:不设置 parallelToolCalls — 设为 false 会导致无 tools 时 OpenAI 返回 400:
// "parallel_tool_calls is only allowed when 'tools' are specified"
// 保持 null 让 Spring AI 不序列化该字段,由各 Node 在有 tools 时自行控制。
options.setStreamUsage(true);
return options;
}
// ==================== OpenAI API 构建 ====================
/** Transitional public visibility for {@code chatmodel} sub-package builders; will move into the builder in PR-0b. */
public OpenAiApi buildOpenAiApi(ModelProviderEntity provider) {
return buildOpenAiApi(provider, null);
}
/**
* Overload that accepts a per-model read-timeout override (seconds).
* Threaded into both the sync RestClient and streaming WebClient so
* timeout behavior is consistent across blocking and streaming chat
* completions. Null falls back to the default 180s.
*/
public OpenAiApi buildOpenAiApi(ModelProviderEntity provider, Integer readTimeoutOverride) {
if (provider == null || !modelProviderService.isProviderConfigured(provider.getProviderId())) {
throw new MateClawException("err.agent.provider_not_configured", "Provider 未完成配置,请在模型设置中填写有效的 API Key 和 Base URL");
}
String apiKey = provider.getApiKey();
// Honor the provider's requireApiKey flag instead of hard-failing on every empty key.
// Local + key-free providers (Ollama, LM Studio, MLX, llama.cpp, OpenCode) declare
// requireApiKey=false; for them an empty / placeholder key means "no Authorization
// header" — Spring AI's NoopApiKey expresses that. Without this the chat path
// rejected providers that probe / discovery / connection-test all considered usable.
boolean keyRequired = !Boolean.FALSE.equals(provider.getRequireApiKey());
if (keyRequired && !modelProviderService.hasUsableApiKey(apiKey)) {
throw new MateClawException("err.agent.provider_apikey_invalid", "Provider API Key 未配置或无效: " + provider.getProviderId());
}
String baseUrl = normalizeOpenAiBaseUrl(provider.getBaseUrl());
if (!StringUtils.hasText(baseUrl)) {
throw new MateClawException("err.agent.provider_baseurl_missing", "Provider Base URL 未配置: " + provider.getProviderId());
}
Map kwargs = modelProviderService.readProviderGenerateKwargs(provider);
MultiValueMap headers = buildOpenAiHeaders(kwargs);
String completionsPath = resolveOpenAiCompletionsPath(baseUrl, kwargs);
RestClient.Builder restClientBuilder = applyHttpTimeouts(
restClientBuilderProvider.getIfAvailable(RestClient::builder), readTimeoutOverride);
WebClient.Builder webClientBuilder = applyHttpTimeoutsToWebClient(
webClientBuilderProvider.getIfAvailable(WebClient::builder), readTimeoutOverride);
// Spring AI OpenAiApi 构造函数会先 set User-Agent 为 "spring-ai",再 addAll 我们的 headers,
// 导致自定义 User-Agent 被追加而非覆盖。因此对需要伪装客户端身份的 provider(如 kimi-code),
// 通过 RestClient/WebClient 拦截器在请求发出前强制覆盖 headers。
Map overrideHeaders = extractOverrideHeaders(kwargs);
if (!overrideHeaders.isEmpty()) {
restClientBuilder = restClientBuilder.requestInterceptor((request, body, execution) -> {
HttpHeaders reqHeaders = request.getHeaders();
overrideHeaders.forEach(reqHeaders::set);
return execution.execute(request, body);
});
webClientBuilder = webClientBuilder.filter((request, next) -> {
org.springframework.web.reactive.function.client.ClientRequest modified =
org.springframework.web.reactive.function.client.ClientRequest.from(request)
.headers(h -> overrideHeaders.forEach(h::set))
.build();
return next.exchange(modified);
});
}
boolean kimiSearchEnabled = isKimiProvider(provider)
&& Boolean.TRUE.equals(kwargs.get("enableSearch"));
ApiKey apiKeyImpl = (keyRequired && StringUtils.hasText(apiKey))
? new SimpleApiKey(apiKey.trim())
: new NoopApiKey();
return new OpenAiApi(
baseUrl,
apiKeyImpl,
headers,
completionsPath,
"/v1/embeddings",
restClientBuilder,
webClientBuilder,
RetryUtils.DEFAULT_RESPONSE_ERROR_HANDLER) {
@Override
public org.springframework.http.ResponseEntity chatCompletionEntity(
OpenAiApi.ChatCompletionRequest chatRequest,
MultiValueMap additionalHttpHeader) {
chatRequest = sanitizeReasoningEffortForProvider(chatRequest, provider);
chatRequest = patchReasoningContent(chatRequest, provider);
chatRequest = stripReasoningEffortIfIncompatible(chatRequest);
chatRequest = stripAutoToolChoice(chatRequest);
chatRequest = patchVideoMediaContent(chatRequest);
if (kimiSearchEnabled) {
chatRequest = injectKimiWebSearch(chatRequest);
}
logOpenAiRequest(provider, chatRequest);
try {
return super.chatCompletionEntity(chatRequest, additionalHttpHeader);
} catch (WebClientResponseException e) {
logOpenAiError(provider, e);
throw e;
}
}
@Override
public Flux chatCompletionStream(
OpenAiApi.ChatCompletionRequest chatRequest,
MultiValueMap additionalHttpHeader) {
chatRequest = sanitizeReasoningEffortForProvider(chatRequest, provider);
chatRequest = patchReasoningContent(chatRequest, provider);
chatRequest = stripReasoningEffortIfIncompatible(chatRequest);
chatRequest = stripAutoToolChoice(chatRequest);
chatRequest = patchVideoMediaContent(chatRequest);
if (kimiSearchEnabled) {
chatRequest = injectKimiWebSearch(chatRequest);
}
logOpenAiRequest(provider, chatRequest);
return super.chatCompletionStream(chatRequest, additionalHttpHeader)
.doOnError(error -> {
if (error instanceof WebClientResponseException e) {
logOpenAiError(provider, e);
}
});
}
};
}
// ==================== DashScope API 构建 ====================
// PR-0b: buildDashScopeApi moved to AgentDashScopeChatModelBuilder
// ==================== Anthropic API 构建 ====================
// PR-0b: buildAnthropicApi + buildAnthropicOptions moved to AgentAnthropicChatModelBuilder
// ==================== 参数解析辅助方法 ====================
private Double resolveOpenAiTemperature(String modelName, Double configuredTemperature,
Map kwargs, ModelFamily family) {
Double overriddenTemperature = resolveDoubleOption("temperature", configuredTemperature, kwargs);
if (family.fixedTemperatureOne()) {
if (overriddenTemperature == null || Double.compare(overriddenTemperature, 1.0d) != 0) {
log.info("ModelFamily {} forced temperature=1.0 for model {}", family, modelName);
}
return 1.0d;
}
return overriddenTemperature;
}
private Double resolveOpenAiTopP(String modelName, Double configuredTopP,
Map kwargs, ModelFamily family) {
if (family.suppressTopP()) {
return null;
}
return resolveDoubleOption("topP", configuredTopP, kwargs);
}
private boolean requiresFixedTemperatureOne(String modelName) {
return ModelFamily.detect(modelName).fixedTemperatureOne();
}
private String resolveReasoningEffort(String modelName, Map kwargs, ModelFamily family) {
// PR-1.1 (RFC-049 L1-A): Only families that actually accept reasoning_effort may receive
// it. Previously only the default-inject branch checked capability; the generateKwargs
// override branch did not, so a provider-level `reasoningEffort: "high"` would leak to
// deepseek-chat / kimi-k2 / deepseek-reasoner etc., triggering the incident documented
// in RFC-049 (DeepSeek "reasoning_content missing" 400).
if (!family.supportsReasoningEffort()) {
Object overridden = findOptionValue(kwargs, "reasoningEffort");
if (overridden != null) {
log.warn("Dropping reasoningEffort='{}' from generateKwargs — model '{}' (family={}) "
+ "does not accept reasoning_effort. For DeepSeek thinking use "
+ "extra_body.thinking; for Kimi thinking the model activates it natively.",
overridden, modelName, family);
}
return null;
}
// generateKwargs 显式覆盖始终优先(仅在白名单族内)
Object value = findOptionValue(kwargs, "reasoningEffort");
if (value instanceof String text && StringUtils.hasText(text)) {
return text.trim();
}
// 仅支持 reasoning_effort 的模型族才自动注入默认值
if (family.isThinking()) {
return "medium";
}
return null;
}
private boolean isThinkingModel(String modelName) {
return ModelFamily.detect(modelName).isThinking();
}
/**
* 从 ModelConfigEntity 中解析 reasoningEffort,用于传递给 StepExecutionNode / ReasoningNode。
* 复用已有的 resolveReasoningEffort + isThinkingModel 逻辑。
*/
private String resolveReasoningEffortForModel(ModelConfigEntity runtimeModel) {
ModelProviderEntity provider = modelProviderService.getProviderConfig(runtimeModel.getProvider());
Map kwargs = modelProviderService.readProviderGenerateKwargs(provider);
ModelFamily family = ModelFamily.detect(runtimeModel.getModelName());
return resolveReasoningEffort(runtimeModel.getModelName(), kwargs, family);
}
private Double resolveDoubleOption(String key, Double fallback, Map kwargs) {
Object value = findOptionValue(kwargs, key);
if (value instanceof Number number) {
return number.doubleValue();
}
if (value instanceof String text && StringUtils.hasText(text)) {
try {
return Double.parseDouble(text.trim());
} catch (NumberFormatException ignored) {
log.warn("Invalid double generateKwargs value for {}: {}", key, text);
}
}
return fallback;
}
private Integer resolveIntegerOption(String key, Integer fallback, Map kwargs) {
Object value = findOptionValue(kwargs, key);
if (value instanceof Number number) {
return number.intValue();
}
if (value instanceof String text && StringUtils.hasText(text)) {
try {
return Integer.parseInt(text.trim());
} catch (NumberFormatException ignored) {
log.warn("Invalid integer generateKwargs value for {}: {}", key, text);
}
}
return fallback;
}
@SuppressWarnings("unchecked")
private Object findOptionValue(Map kwargs, String key) {
Object direct = findKwarg(kwargs, key);
if (direct != null) {
return direct;
}
String snakeCase = key.replaceAll("([a-z])([A-Z])", "$1_$2").toLowerCase();
if (!snakeCase.equals(key)) {
return findKwarg(kwargs, snakeCase);
}
return null;
}
@SuppressWarnings("unchecked")
private Object findKwarg(Map kwargs, String key) {
if (kwargs == null || kwargs.isEmpty()) {
return null;
}
if (kwargs.containsKey(key)) {
return kwargs.get(key);
}
Object chatOptions = kwargs.get("chatOptions");
if (chatOptions instanceof Map, ?> optionsMap) {
return ((Map) optionsMap).get(key);
}
return null;
}
// ==================== URL 规范化 ====================
// PR-0b: normalizeDashScopeBaseUrl moved to AgentDashScopeChatModelBuilder
private String normalizeOpenAiBaseUrl(String baseUrl) {
if (!StringUtils.hasText(baseUrl)) {
return null;
}
String normalized = baseUrl.trim();
if (normalized.endsWith("/")) {
normalized = normalized.substring(0, normalized.length() - 1);
}
if (normalized.endsWith("/v1")) {
normalized = normalized.substring(0, normalized.length() - 3);
}
return normalized;
}
// ==================== Kimi 内置搜索 ====================
private static boolean isKimiProvider(ModelProviderEntity provider) {
if (provider == null) return false;
String id = provider.getProviderId();
return "kimi-cn".equals(id) || "kimi-intl".equals(id);
}
/**
* 为 Kimi 请求注入 $web_search builtin tool。
* Kimi 的内置搜索通过 tools 数组中声明 {"type":"builtin_function","function":{"name":"$web_search"}} 实现。
* 由于 Spring AI 的 FunctionTool.Type 只有 FUNCTION,无法直接构造 builtin_function 类型,
* 因此通过 extraBody 注入原始 JSON 结构覆盖 tools 字段(包含原有 tools + $web_search)。
*/
private static OpenAiApi.ChatCompletionRequest injectKimiWebSearch(OpenAiApi.ChatCompletionRequest request) {
// 构造 $web_search entry 作为 Map
Map webSearchTool = Map.of(
"type", "builtin_function",
"function", Map.of("name", "$web_search")
);
// 将原有 tools 转为 List