From b65887f93d7240f7ba10f34dc960e3b9e47ef6fa Mon Sep 17 00:00:00 2001 From: matevip Date: Wed, 3 Jun 2026 21:19:26 +0800 Subject: [PATCH] feat(goal): dual-mode checklist evaluator with structured output + Evaluator SPI --- .../mate/goal/model/GoalCriteriaCodec.java | 112 ++++++ .../goal/service/GoalEvaluationService.java | 334 +++++++++++------- .../mate/tool/builtin/GoalManagementTool.java | 13 + 3 files changed, 339 insertions(+), 120 deletions(-) create mode 100644 mateclaw-server/src/main/java/vip/mate/goal/model/GoalCriteriaCodec.java diff --git a/mateclaw-server/src/main/java/vip/mate/goal/model/GoalCriteriaCodec.java b/mateclaw-server/src/main/java/vip/mate/goal/model/GoalCriteriaCodec.java new file mode 100644 index 00000000..f059c6f1 --- /dev/null +++ b/mateclaw-server/src/main/java/vip/mate/goal/model/GoalCriteriaCodec.java @@ -0,0 +1,112 @@ +package vip.mate.goal.model; + +import com.fasterxml.jackson.core.JsonProcessingException; +import com.fasterxml.jackson.core.type.TypeReference; +import com.fasterxml.jackson.databind.ObjectMapper; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import java.util.ArrayList; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +/** + * Shared (de)serialization and merge helpers for a goal's checklist stored + * as JSON text in {@code mate_agent_goal.criteria}. + * + *

Centralizes the String JSON ↔ {@code List} boundary so the + * evaluator, service and node never reimplement parsing. Parse failures fail + * soft to an empty list (logged) rather than throwing — a corrupt column must + * never break a chat turn or an API response. + */ +public final class GoalCriteriaCodec { + + private static final Logger log = LoggerFactory.getLogger(GoalCriteriaCodec.class); + private static final TypeReference> LIST_TYPE = new TypeReference<>() { + }; + + private GoalCriteriaCodec() { + } + + /** Parse the JSON column into a mutable list; empty list on null/blank/corrupt. */ + public static List parse(String json, ObjectMapper mapper) { + if (json == null || json.isBlank()) { + return new ArrayList<>(); + } + try { + List parsed = mapper.readValue(json, LIST_TYPE); + return parsed != null ? parsed : new ArrayList<>(); + } catch (Exception e) { + log.warn("[GoalCriteria] failed to parse criteria JSON, treating as empty: {}", e.getMessage()); + return new ArrayList<>(); + } + } + + /** Serialize a checklist to JSON text; {@code null} for a null list. */ + public static String serialize(List criteria, ObjectMapper mapper) { + if (criteria == null) { + return null; + } + try { + return mapper.writeValueAsString(criteria); + } catch (JsonProcessingException e) { + log.warn("[GoalCriteria] failed to serialize criteria, storing null: {}", e.getMessage()); + return null; + } + } + + /** + * Merge a per-round verdict delta into the full checklist by id. Criteria + * absent from the delta are preserved unchanged; the criterion text is + * always kept from the existing item (the verdict never carries text). + */ + public static List merge(List existing, + List verdicts) { + if (existing == null || existing.isEmpty()) { + return existing == null ? new ArrayList<>() : existing; + } + Map byId = new LinkedHashMap<>(); + if (verdicts != null) { + for (GoalChecklistVerdict.CriterionVerdict v : verdicts) { + if (v != null && v.id() != null) { + byId.put(v.id(), v); + } + } + } + List merged = new ArrayList<>(existing.size()); + for (GoalCriterion c : existing) { + GoalChecklistVerdict.CriterionVerdict v = byId.get(c.id()); + merged.add(v == null + ? c + : new GoalCriterion(c.id(), c.text(), v.passed(), + v.evidence() != null ? v.evidence() : "")); + } + return merged; + } + + /** True only when the list is non-empty and every criterion is passed. */ + public static boolean allPassed(List criteria) { + return criteria != null && !criteria.isEmpty() + && criteria.stream().allMatch(GoalCriterion::passed); + } + + /** Criteria not yet passed (used for the continuation prompt + gap text). */ + public static List remaining(List criteria) { + if (criteria == null) { + return List.of(); + } + return criteria.stream().filter(c -> !c.passed()).toList(); + } + + /** Reassign stable ids {@code C1..Cn} in list order. */ + public static List reindex(List criteria) { + List out = new ArrayList<>(criteria.size()); + int n = 1; + for (GoalCriterion c : criteria) { + out.add(new GoalCriterion("C" + n, c.text(), c.passed(), c.evidence() == null ? "" : c.evidence())); + n++; + } + return out; + } +} diff --git a/mateclaw-server/src/main/java/vip/mate/goal/service/GoalEvaluationService.java b/mateclaw-server/src/main/java/vip/mate/goal/service/GoalEvaluationService.java index 3f36cbc1..cd8a5e02 100644 --- a/mateclaw-server/src/main/java/vip/mate/goal/service/GoalEvaluationService.java +++ b/mateclaw-server/src/main/java/vip/mate/goal/service/GoalEvaluationService.java @@ -1,6 +1,5 @@ package vip.mate.goal.service; -import com.fasterxml.jackson.databind.JsonNode; import com.fasterxml.jackson.databind.ObjectMapper; import lombok.extern.slf4j.Slf4j; import org.springframework.ai.chat.messages.Message; @@ -10,9 +9,17 @@ import org.springframework.ai.chat.model.ChatModel; import org.springframework.ai.chat.model.ChatResponse; import org.springframework.ai.chat.prompt.ChatOptions; import org.springframework.ai.chat.prompt.Prompt; +import org.springframework.ai.converter.BeanOutputConverter; +import org.springframework.ai.evaluation.EvaluationRequest; +import org.springframework.ai.evaluation.EvaluationResponse; +import org.springframework.ai.evaluation.Evaluator; import org.springframework.retry.support.RetryTemplate; import org.springframework.stereotype.Service; import vip.mate.goal.config.GoalProperties; +import vip.mate.goal.model.GoalChecklistVerdict; +import vip.mate.goal.model.GoalCriteriaCodec; +import vip.mate.goal.model.GoalCriteriaDraft; +import vip.mate.goal.model.GoalCriterion; import vip.mate.goal.model.GoalEntity; import vip.mate.goal.model.GoalEvaluationResult; import vip.mate.llm.chatmodel.ProviderChatModelFactory; @@ -20,50 +27,52 @@ import vip.mate.llm.model.ModelConfigEntity; import vip.mate.llm.service.ModelConfigService; import java.util.ArrayList; +import java.util.LinkedHashMap; import java.util.List; +import java.util.Map; /** * Evaluates whether the assistant's latest reply satisfies a goal's exit - * criteria. Drives the persistent-goal completion path (the - * "auto-followup until score hits 1.0" loop), so it sits on the hot path - * of every chat turn that has an active goal. + * checklist, and bootstraps that checklist on first run. Sits on the hot + * path of every chat turn that has an active goal. * - *

Returns a deterministic {@link GoalEvaluationResult#fallback fallback} - * when the LLM call is unavailable, errors out, or returns un-parseable - * JSON — the {@code GoalEvaluationNode} treats fallback as "skip - * bookkeeping deltas, no event log, stay safe". This degrades cleanly - * when the evaluator provider is misconfigured or transiently down. + *

Two evaluation modes, chosen by whether the goal already has criteria: + *

* - *

Model selection: - *

    - *
  1. If {@code mateclaw.goal.evaluator-model} names an enabled model, - * use it.
  2. - *
  3. Otherwise fall back to {@link ModelConfigService#getDefaultModel()} - * — convenient for dev, but operators are encouraged to pin a cheap - * evaluator-only model in production since this fires on every turn. - *
  4. - *
+ *

Output is shaped by {@link BeanOutputConverter}, which injects a JSON + * format instruction and parses the reply. On any failure (no model, empty + * reply, unparseable output, provider error) a deterministic + * {@link GoalEvaluationResult#fallback fallback} is returned so the node can + * degrade cleanly. * - *

Prompt is short and JSON-only: the evaluator returns one object with - * {@code score} (0.0–1.0 fraction of criteria satisfied), {@code gap} - * (plain-text description of what's missing), and {@code completed} (bool). + *

Implements Spring AI's {@link Evaluator} for interface uniformity and + * testability; the goal-aware overloads carry the context the generic SPI + * request cannot. */ @Slf4j @Service -public class GoalEvaluationService { +public class GoalEvaluationService implements Evaluator { /** - * Token budget for the evaluator response. Reasoning-mode models - * (DeepSeek V4 Pro, Kimi for Coding, GLM-Z1, …) consume a chunk of - * this budget on internal {@code } content before emitting - * the JSON answer; 400 was empirically too tight and produced - * empty responses on every reasoning provider. 2000 leaves comfort - * for ~1500 tokens of reasoning + the small JSON object we need. + * Token budget for the evaluator response. Reasoning-mode models consume + * a chunk of this on internal thinking before emitting JSON; 2000 leaves + * comfort for the reasoning trace plus the small object we need. */ private static final int MAX_OUTPUT_TOKENS = 2000; private static final int MAX_CONVERSATION_CHARS = 6_000; private static final int MAX_TERMINAL_ANSWER_CHARS = 4_000; - /** Skip-retry template — the goal node has its own try/catch, no need to double-retry. */ + private static final int MIN_BOOTSTRAP_CRITERIA = 1; + private static final int MAX_BOOTSTRAP_CRITERIA = 8; + /** Skip-retry template — the goal node has its own try/catch. */ private static final RetryTemplate ONESHOT = RetryTemplate.builder().maxAttempts(1).build(); private final GoalProperties properties; @@ -71,6 +80,11 @@ public class GoalEvaluationService { private final ProviderChatModelFactory chatModelFactory; private final ObjectMapper objectMapper; + private final BeanOutputConverter draftConverter = + new BeanOutputConverter<>(GoalCriteriaDraft.class); + private final BeanOutputConverter verdictConverter = + new BeanOutputConverter<>(GoalChecklistVerdict.class); + public GoalEvaluationService(GoalProperties properties, ModelConfigService modelConfigService, ProviderChatModelFactory chatModelFactory, @@ -82,19 +96,12 @@ public class GoalEvaluationService { } /** - * Evaluate one terminal answer against the goal's exit criteria. + * Evaluate one terminal answer against the goal's checklist (or bootstrap + * the checklist when none exists yet). * - *

Returns a {@link GoalEvaluationResult} carrying the score, gap - * description, decision, model id, and elapsed latency. The - * {@code llmCallsConsumed} field is 1 on success (one evaluator - * call) and 0 on fallback paths so the per-goal LLM-call budget - * stays accurate. - * - * @param goal the active goal under evaluation; never {@code null} - * @param recentMessages most-recent N messages from the parent conversation - * for context; the node already trims by - * {@link GoalProperties#getEvaluatorContextMessages()} - * @param terminalAnswer the assistant's just-emitted final answer text + * @param goal the active goal under evaluation; never {@code null} + * @param recentMessages most-recent N messages for context (already trimmed) + * @param terminalAnswer the assistant's just-emitted final answer text */ public GoalEvaluationResult evaluate(GoalEntity goal, List recentMessages, @@ -108,19 +115,24 @@ public class GoalEvaluationService { ModelConfigEntity model = resolveEvaluatorModel(); if (model == null) { - log.warn("[GoalEvaluation] no evaluator model available (configured={}, default lookup empty)", + log.warn("[GoalEvaluation] no evaluator model available (configured={})", properties.getEvaluatorModel()); return GoalEvaluationResult.fallback("no_model"); } + List existing = GoalCriteriaCodec.parse(goal.getCriteria(), objectMapper); + boolean bootstrap = existing.isEmpty(); + long start = System.currentTimeMillis(); try { ChatModel chatModel = chatModelFactory.buildFor(model, ONESHOT); - String prompt = buildUserPrompt(goal, recentMessages, terminalAnswer); + String format = bootstrap ? draftConverter.getFormat() : verdictConverter.getFormat(); + String userPrompt = buildUserPrompt(goal, existing, recentMessages, terminalAnswer, bootstrap) + + "\n\n" + format; List messages = new ArrayList<>(2); - messages.add(new SystemMessage(SYSTEM_PROMPT)); - messages.add(new UserMessage(prompt)); + messages.add(new SystemMessage(bootstrap ? BOOTSTRAP_SYSTEM_PROMPT : VERDICT_SYSTEM_PROMPT)); + messages.add(new UserMessage(userPrompt)); ChatOptions options = ChatOptions.builder() .temperature(0.1) @@ -136,7 +148,9 @@ public class GoalEvaluationService { return GoalEvaluationResult.fallback("empty_response"); } - return parseJson(body, model.getModelName(), elapsed); + return bootstrap + ? parseBootstrap(body, model.getModelName(), elapsed) + : parseVerdict(body, existing, model.getModelName(), elapsed); } catch (Throwable t) { long elapsed = System.currentTimeMillis() - start; log.warn("[GoalEvaluation] evaluator call failed after {}ms: {}", elapsed, t.toString()); @@ -144,38 +158,82 @@ public class GoalEvaluationService { } } + // ==================== Evaluator SPI ==================== + + /** + * Generic SPI surface: judge whether {@code request.getResponseContent()} + * satisfies the objective in {@code request.getUserText()}. Used for + * standardization/testing; goal-aware callers use the + * {@link #evaluate(GoalEntity, List, String)} overload which carries the + * checklist context the request cannot. Detail rides in metadata. + */ + @Override + public EvaluationResponse evaluate(EvaluationRequest request) { + GoalEntity probe = new GoalEntity(); + probe.setTitle(request.getUserText()); + probe.setDescription(""); + GoalEvaluationResult r = evaluate(probe, List.of(), request.getResponseContent()); + Map metadata = new LinkedHashMap<>(); + metadata.put("decision", r.decision()); + if (r.bootstrapCriteria() != null) { + metadata.put("bootstrapCriteria", r.bootstrapCriteria()); + } else { + metadata.put("criterionVerdicts", r.criterionVerdicts()); + } + return new EvaluationResponse(r.completed(), (float) r.score(), + r.gap() == null ? "" : r.gap(), metadata); + } + // ==================== Internals ==================== private ModelConfigEntity resolveEvaluatorModel() { String name = properties.getEvaluatorModel(); if (name != null && !name.isBlank()) { - // resolveModel returns the default model when the named one - // can't be found, which is exactly the desired "graceful - // degradation" semantics for a misconfigured evaluator id. + // resolveModel returns the default model when the named one can't + // be found — the desired graceful-degradation semantics. return modelConfigService.resolveModel(name); } return modelConfigService.getDefaultModel(); } - private static final String SYSTEM_PROMPT = - "You are a goal-completion evaluator. You judge whether an AI " - + "assistant's latest reply satisfies a user's stated goal. " - + "Output exactly ONE JSON object with the keys score, gap, " - + "completed. No markdown, no commentary, no extra prose."; + private static final String BOOTSTRAP_SYSTEM_PROMPT = + "You decompose a user's goal into a short checklist of concrete, " + + "independently verifiable acceptance criteria. Each criterion " + + "must be checkable from observable evidence (an output, a file, " + + "a command result), not a vague aspiration. Output only the " + + "requested JSON."; + + private static final String VERDICT_SYSTEM_PROMPT = + "You judge, criterion by criterion, whether an AI assistant's latest " + + "reply satisfies a goal's checklist. For each criterion you MUST " + + "cite concrete evidence from the reply (an output line, a file " + + "excerpt, a command result). Do NOT accept generic phrases like " + + "'all requirements met'. If a criterion lacks specific evidence, " + + "mark it not passed. Output only the requested JSON."; private String buildUserPrompt(GoalEntity goal, + List existing, List recentMessages, - String terminalAnswer) { + String terminalAnswer, + boolean bootstrap) { StringBuilder sb = new StringBuilder(2048); sb.append("Goal title: ").append(safe(goal.getTitle())).append('\n'); if (goal.getDescription() != null && !goal.getDescription().isBlank()) { sb.append("Goal description: ").append(safe(goal.getDescription())).append('\n'); } if (goal.getExitCriteria() != null && !goal.getExitCriteria().isBlank()) { - sb.append("Exit criteria:\n").append(safe(goal.getExitCriteria())).append('\n'); + sb.append("Exit criteria (free text):\n").append(safe(goal.getExitCriteria())).append('\n'); } sb.append('\n'); + if (!bootstrap) { + sb.append("Current checklist (judge each by id):\n"); + for (GoalCriterion c : existing) { + sb.append("- ").append(c.id()).append(": ").append(c.text()).append('\n'); + } + sb.append('\n'); + } + if (recentMessages != null && !recentMessages.isEmpty()) { sb.append("Recent conversation (oldest first):\n"); String convo = serializeMessages(recentMessages); @@ -191,22 +249,112 @@ public class GoalEvaluationService { } sb.append("\nAssistant's latest final answer to evaluate:\n").append(answer).append('\n'); - sb.append('\n') - .append("Return exactly:\n") - .append("{\n") - .append(" \"score\": ,\n") - .append(" \"gap\": \"\",\n") - .append(" \"completed\": \n") - .append("}"); + sb.append('\n'); + if (bootstrap) { + sb.append("Produce between ").append(MIN_BOOTSTRAP_CRITERIA).append(" and ") + .append(MAX_BOOTSTRAP_CRITERIA) + .append(" criteria. Leave every 'passed' false and 'evidence' empty — " + + "this round only defines the checklist."); + } else { + sb.append("For every criterion above, return its id with passed=true ONLY when " + + "the reply shows concrete evidence; otherwise passed=false with a short " + + "note of what is missing."); + } return sb.toString(); } + private GoalEvaluationResult parseBootstrap(String body, String modelName, long latencyMs) { + try { + GoalCriteriaDraft dto = draftConverter.convert(stripFences(body)); + if (dto == null || dto.criteria() == null || dto.criteria().isEmpty()) { + return GoalEvaluationResult.fallback("bootstrap_empty"); + } + List normalized = new ArrayList<>(); + for (GoalCriterion c : dto.criteria()) { + if (c != null && c.text() != null && !c.text().isBlank()) { + normalized.add(new GoalCriterion("", c.text().trim(), false, "")); + } + } + if (normalized.isEmpty()) { + return GoalEvaluationResult.fallback("bootstrap_empty"); + } + normalized = GoalCriteriaCodec.reindex(normalized); + // Bootstrap never judges completion: the checklist is freshly created. + return new GoalEvaluationResult( + 0.0, "checklist created", GoalEvaluationResult.DECISION_CONTINUE, false, + modelName != null ? modelName : "", 1, latencyMs, + List.of(), normalized); + } catch (Exception e) { + log.warn("[GoalEvaluation] bootstrap parse failed: {}", e.getMessage()); + return GoalEvaluationResult.fallback("parse_failed"); + } + } + + private GoalEvaluationResult parseVerdict(String body, + List existing, + String modelName, + long latencyMs) { + try { + GoalChecklistVerdict verdict = verdictConverter.convert(stripFences(body)); + List deltas = + verdict != null && verdict.criterionVerdicts() != null + ? verdict.criterionVerdicts() : List.of(); + List merged = GoalCriteriaCodec.merge(existing, deltas); + boolean completed = GoalCriteriaCodec.allPassed(merged); + int total = merged.size(); + int passed = (int) merged.stream().filter(GoalCriterion::passed).count(); + double score = total == 0 ? 0.0 : (double) passed / total; + String gap = completed ? "" : buildGap(GoalCriteriaCodec.remaining(merged)); + String decision = completed + ? GoalEvaluationResult.DECISION_COMPLETED + : GoalEvaluationResult.DECISION_CONTINUE; + return new GoalEvaluationResult( + score, gap, decision, completed, + modelName != null ? modelName : "", 1, latencyMs, + deltas, null); + } catch (Exception e) { + log.warn("[GoalEvaluation] verdict parse failed: {}", e.getMessage()); + return GoalEvaluationResult.fallback("parse_failed"); + } + } + + private static String buildGap(List remaining) { + if (remaining.isEmpty()) { + return ""; + } + StringBuilder sb = new StringBuilder("Still missing: "); + for (int i = 0; i < remaining.size(); i++) { + if (i > 0) { + sb.append("; "); + } + sb.append(remaining.get(i).text()); + } + return sb.toString(); + } + + /** Strip ```json fences the model may add despite instructions. */ + private static String stripFences(String body) { + String t = body.strip(); + if (t.startsWith("```")) { + int nl = t.indexOf('\n'); + if (nl > 0) { + t = t.substring(nl + 1); + } + if (t.endsWith("```")) { + t = t.substring(0, t.length() - 3); + } + } + return t.strip(); + } + private String serializeMessages(List messages) { StringBuilder sb = new StringBuilder(); for (Message m : messages) { String role = m.getMessageType() != null ? m.getMessageType().getValue() : "msg"; String text = m.getText(); - if (text == null) text = ""; + if (text == null) { + text = ""; + } sb.append(role).append(": ").append(text.strip()).append('\n'); } return sb.toString(); @@ -222,11 +370,8 @@ public class GoalEvaluationService { if (text != null && !text.isBlank()) { return text; } - // Fallback for reasoning models: some providers (DeepSeek-style - // OpenAI-compatible streaming, MiMo) emit the entire output as - // `reasoning_content` and leave the regular content field empty - // when the token budget gets eaten by thinking. The JSON object - // we want often appears at the tail of the reasoning trace. + // Fallback for reasoning models that emit everything as reasoningContent + // and leave the regular content empty; the JSON often tails the trace. var metadata = output.getMetadata(); if (metadata != null) { Object rc = metadata.get("reasoningContent"); @@ -237,57 +382,6 @@ public class GoalEvaluationService { return text; } - /** - * Parse the evaluator's JSON output. The model may wrap the object in - * ```json fences despite the system prompt telling it not to, so we - * locate the first {@code {...}} substring and parse that. Anything - * else (non-numeric score, missing fields, malformed JSON) downgrades - * to a fallback result rather than throwing. - */ - private GoalEvaluationResult parseJson(String body, String modelName, long latencyMs) { - String trimmed = body.strip(); - int braceStart = trimmed.indexOf('{'); - int braceEnd = trimmed.lastIndexOf('}'); - if (braceStart < 0 || braceEnd <= braceStart) { - log.warn("[GoalEvaluation] no JSON object in evaluator output: {}", - trimmed.length() > 200 ? trimmed.substring(0, 200) + "..." : trimmed); - return GoalEvaluationResult.fallback("parse_no_object"); - } - String json = trimmed.substring(braceStart, braceEnd + 1); - try { - JsonNode node = objectMapper.readTree(json); - JsonNode scoreNode = node.get("score"); - if (scoreNode == null || !scoreNode.isNumber()) { - return GoalEvaluationResult.fallback("parse_missing_score"); - } - double score = clamp01(scoreNode.asDouble()); - String gap = node.hasNonNull("gap") ? node.get("gap").asText("") : ""; - boolean completed = node.hasNonNull("completed") && node.get("completed").asBoolean(false); - // Belt-and-braces: a perfect score implies completion; let the - // node's >= 0.95 threshold handle the gray zone. - if (score >= 1.0 - 1e-9) completed = true; - String decision = completed - ? GoalEvaluationResult.DECISION_COMPLETED - : GoalEvaluationResult.DECISION_CONTINUE; - return new GoalEvaluationResult( - score, gap, decision, completed, - modelName != null ? modelName : "", 1, latencyMs, - List.of(), null); - } catch (Exception e) { - log.warn("[GoalEvaluation] JSON parse failed: {} — body={}", - e.getMessage(), - json.length() > 200 ? json.substring(0, 200) + "..." : json); - return GoalEvaluationResult.fallback("parse_failed"); - } - } - - private static double clamp01(double v) { - if (Double.isNaN(v)) return 0.0; - if (v < 0.0) return 0.0; - if (v > 1.0) return 1.0; - return v; - } - private static String safe(String s) { return s == null ? "" : s; } diff --git a/mateclaw-server/src/main/java/vip/mate/tool/builtin/GoalManagementTool.java b/mateclaw-server/src/main/java/vip/mate/tool/builtin/GoalManagementTool.java index cebe11a3..b8ec1377 100644 --- a/mateclaw-server/src/main/java/vip/mate/tool/builtin/GoalManagementTool.java +++ b/mateclaw-server/src/main/java/vip/mate/tool/builtin/GoalManagementTool.java @@ -60,6 +60,9 @@ public class GoalManagementTool { required = false) Integer turnBudget, @ToolParam(description = "If true, the agent may auto-followup when progress is incomplete. Default false.", required = false) Boolean autoFollowup, + @ToolParam(description = "Optional initial checklist: a list of short, individually verifiable " + + "acceptance criteria. Omit to let the system derive the checklist on first evaluation.", + required = false) java.util.List criteria, @Nullable ToolContext ctx) { if (!properties.isEnabled()) { @@ -86,6 +89,16 @@ public class GoalManagementTool { req.setExitCriteria(exitCriteria); if (turnBudget != null) req.setTurnBudget(turnBudget); if (autoFollowup != null) req.setAutoFollowupEnabled(autoFollowup); + if (criteria != null && !criteria.isEmpty()) { + java.util.List items = new java.util.ArrayList<>(); + for (String text : criteria) { + if (text != null && !text.isBlank()) { + // Only text matters; create() assigns ids, forces passed=false, clears evidence. + items.add(new vip.mate.goal.model.GoalCriterion("", text.trim(), false, "")); + } + } + if (!items.isEmpty()) req.setCriteria(items); + } String username = origin.requesterId() != null && !origin.requesterId().isBlank() ? origin.requesterId() : "system";