From 35a241a62dae222d115e0cddb50e4fcef1bd9e14 Mon Sep 17 00:00:00 2001 From: mateaix <7333791@qq.com> Date: Sun, 13 Sep 2026 20:48:02 +0800 Subject: [PATCH] test(evaluation): add offline goal task replay and baseline --- .../evaluation/OfflineGoalTaskReplay.java | 147 ++++++++++ .../evaluation/OfflineGoalTaskReplayTest.java | 62 ++++ .../test/resources/agent-evaluation/README.md | 62 ++++ .../agent-evaluation/baseline-v1.json | 223 ++++++++++++++ .../agent-evaluation/goal-boundaries-v1.json | 272 ++++++++++++++++++ 5 files changed, 766 insertions(+) create mode 100644 mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplay.java create mode 100644 mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplayTest.java create mode 100644 mateclaw-server/src/test/resources/agent-evaluation/README.md create mode 100644 mateclaw-server/src/test/resources/agent-evaluation/baseline-v1.json create mode 100644 mateclaw-server/src/test/resources/agent-evaluation/goal-boundaries-v1.json diff --git a/mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplay.java b/mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplay.java new file mode 100644 index 00000000..188671aa --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplay.java @@ -0,0 +1,147 @@ +package vip.mate.evaluation; + +import com.fasterxml.jackson.databind.DeserializationFeature; +import com.fasterxml.jackson.databind.ObjectMapper; +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.model.Generation; +import org.springframework.ai.chat.prompt.Prompt; +import org.springframework.retry.support.RetryTemplate; +import vip.mate.goal.config.GoalProperties; +import vip.mate.goal.model.GoalCriteriaCodec; +import vip.mate.goal.model.GoalCriterion; +import vip.mate.goal.model.GoalEntity; +import vip.mate.goal.service.GoalEvaluationService; +import vip.mate.llm.chatmodel.ProviderChatModelFactory; +import vip.mate.llm.model.ModelConfigEntity; +import vip.mate.llm.service.ModelConfigService; + +import java.io.IOException; +import java.security.MessageDigest; +import java.security.NoSuchAlgorithmException; +import java.util.ArrayList; +import java.util.HashSet; +import java.util.HexFormat; +import java.util.List; +import java.util.Set; + +import static org.mockito.ArgumentMatchers.any; +import static org.mockito.Mockito.mock; +import static org.mockito.Mockito.when; + +/** Offline policy replay. The model is a fixture: no agent task is actually executed. */ +final class OfflineGoalTaskReplay { + static final ObjectMapper JSON = new ObjectMapper() + .enable(DeserializationFeature.FAIL_ON_TRAILING_TOKENS); + static final String MODE = "offline_synthetic_evaluator_replay"; + + record Suite(Integer schemaVersion, String suiteId, List tasks) { } + record Task(String id, String task, String source, String boundary, Boolean persistent, + List criteria, String terminalAnswer, String evaluatorResponse, + Expected expected) { } + record Expected(Boolean completed, Double score, String decision, List remainingIds, + Integer fixtureCalls) { } + record Actual(boolean completed, double score, String decision, List remainingIds, + int fixtureCalls) { } + record CaseResult(String id, String task, String source, String boundary, Expected expected, + Actual actual, boolean matched, String error) { } + record Report(int schemaVersion, String suiteId, String suiteSha256, String codeRevision, + String executionMode, int onlineModelCalls, String agentTaskSuccessRate, + String onlineCost, int matchedCases, int mismatchedCases, List cases) { } + + static Suite parse(byte[] bytes) throws IOException { + Suite suite = JSON.readValue(bytes, Suite.class); + require(suite != null && Integer.valueOf(1).equals(suite.schemaVersion()), "schemaVersion must be 1"); + require(nonblank(suite.suiteId()), "suiteId is required"); + require(suite.tasks() != null && !suite.tasks().isEmpty() && suite.tasks().size() <= 100, + "suite must contain 1..100 tasks"); + Set ids = new HashSet<>(); + for (Task task : suite.tasks()) { + require(task != null && nonblank(task.id()) && ids.add(task.id()), "task IDs must be unique and nonblank"); + require(nonblank(task.task()) && nonblank(task.source()) && nonblank(task.boundary()), + task.id() + ": task, source and boundary are required"); + require(task.persistent() != null && task.criteria() != null && task.terminalAnswer() != null + && task.evaluatorResponse() != null, task.id() + ": missing replay inputs"); + Set criterionIds = new HashSet<>(); + for (GoalCriterion criterion : task.criteria()) { + require(criterion != null && nonblank(criterion.id()) && nonblank(criterion.text()) + && criterionIds.add(criterion.id()), task.id() + ": invalid or duplicate criterion"); + } + Expected expected = task.expected(); + require(expected != null && expected.completed() != null && expected.score() != null + && Double.isFinite(expected.score()) && expected.score() >= 0 && expected.score() <= 1 + && Set.of("completed", "continue", "fallback").contains(expected.decision() == null ? "" : expected.decision()) + && expected.remainingIds() != null && expected.fixtureCalls() != null + && expected.fixtureCalls() >= 0 && expected.fixtureCalls() <= 1, + task.id() + ": incomplete or invalid expected outcome"); + require(expected.remainingIds().stream().allMatch(OfflineGoalTaskReplay::nonblank) + && new HashSet<>(expected.remainingIds()).size() == expected.remainingIds().size(), + task.id() + ": remainingIds must be unique and nonblank"); + } + return suite; + } + + static Report run(byte[] bytes, String revision) throws IOException { + Suite suite = parse(bytes); // Validate the whole suite before any fixture is replayed. + require(nonblank(revision), "codeRevision is required"); + List results = new ArrayList<>(); + for (Task task : suite.tasks()) { + try { + Actual actual = replay(task); + Expected expected = task.expected(); + boolean matched = expected.completed() == actual.completed() + && Math.abs(expected.score() - actual.score()) < 1e-9 + && expected.decision().equals(actual.decision()) + && expected.remainingIds().equals(actual.remainingIds()) + && expected.fixtureCalls() == actual.fixtureCalls(); + results.add(new CaseResult(task.id(), task.task(), task.source(), task.boundary(), + expected, actual, matched, null)); + } catch (RuntimeException error) { + results.add(new CaseResult(task.id(), task.task(), task.source(), task.boundary(), + task.expected(), null, false, error.getClass().getSimpleName())); + } + } + int matched = (int) results.stream().filter(CaseResult::matched).count(); + return new Report(1, suite.suiteId(), digest(bytes), revision, MODE, 0, + "not_measured", "not_measured", matched, results.size() - matched, List.copyOf(results)); + } + + private static Actual replay(Task task) { + ChatModel chat = mock(ChatModel.class); + ModelConfigService configs = mock(ModelConfigService.class); + ProviderChatModelFactory factory = mock(ProviderChatModelFactory.class); + ModelConfigEntity model = new ModelConfigEntity(); + model.setModelName("offline-fixture"); + when(configs.getDefaultModel()).thenReturn(model); + when(factory.buildFor(any(ModelConfigEntity.class), any(RetryTemplate.class))).thenReturn(chat); + int[] fixtureCalls = {0}; + when(chat.call(any(Prompt.class))).thenAnswer(call -> { + fixtureCalls[0]++; + return new ChatResponse(List.of(new Generation(new AssistantMessage(task.evaluatorResponse())))); + }); + GoalEntity goal = new GoalEntity(); + goal.setTitle(task.task()); + goal.setPersistentExecution(task.persistent()); + goal.setCriteria(GoalCriteriaCodec.serialize(task.criteria(), JSON)); + var evaluator = new GoalEvaluationService(new GoalProperties(), configs, factory, JSON); + var result = evaluator.evaluate(goal, List.of(), task.terminalAnswer()); + var merged = result.bootstrapCriteria() != null ? result.bootstrapCriteria() + : GoalCriteriaCodec.merge(task.criteria(), result.criterionVerdicts()); + return new Actual(result.completed(), result.score(), result.decision(), + GoalCriteriaCodec.remaining(merged).stream().map(GoalCriterion::id).toList(), fixtureCalls[0]); + } + + private static String digest(byte[] bytes) { + try { + return HexFormat.of().formatHex(MessageDigest.getInstance("SHA-256").digest(bytes)); + } catch (NoSuchAlgorithmException impossible) { + throw new IllegalStateException(impossible); + } + } + + private static boolean nonblank(String value) { return value != null && !value.isBlank(); } + private static void require(boolean valid, String message) { + if (!valid) throw new IllegalArgumentException(message); + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplayTest.java b/mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplayTest.java new file mode 100644 index 00000000..1487b01f --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/evaluation/OfflineGoalTaskReplayTest.java @@ -0,0 +1,62 @@ +package vip.mate.evaluation; + +import com.fasterxml.jackson.databind.node.ObjectNode; +import org.junit.jupiter.api.Test; + +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; + +import static org.junit.jupiter.api.Assertions.*; + +class OfflineGoalTaskReplayTest { + private byte[] fixture() throws Exception { + try (var stream = getClass().getResourceAsStream("/agent-evaluation/goal-boundaries-v1.json")) { + assertNotNull(stream); + return stream.readAllBytes(); + } + } + + @Test + void replayTasksAndWriteReportEvenWhenExpectationsMismatch() throws Exception { + String input = System.getProperty("agent.eval.suite"); + byte[] bytes = input == null ? fixture() : Files.readAllBytes(Path.of(input)); + var report = OfflineGoalTaskReplay.run(bytes, System.getProperty("agent.eval.revision", "unrecorded")); + Path output = Path.of(System.getProperty("agent.eval.report", "target/agent-evaluation/goal-baseline.json")); + Files.createDirectories(output.toAbsolutePath().getParent()); + OfflineGoalTaskReplay.JSON.writerWithDefaultPrettyPrinter().writeValue(output.toFile(), report); + assertEquals(0, report.mismatchedCases(), () -> "Replay mismatches: " + output.toAbsolutePath()); + assertEquals(0, report.onlineModelCalls()); + assertEquals("not_measured", report.agentTaskSuccessRate()); + } + + @Test + void invalidSuiteIsRejectedBeforeReplay() throws Exception { + ObjectNode root = (ObjectNode) OfflineGoalTaskReplay.JSON.readTree(fixture()); + var empty = root.deepCopy(); + empty.putArray("tasks"); + assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(empty.toString().getBytes(StandardCharsets.UTF_8))); + var duplicate = root.deepCopy(); + ((ObjectNode) duplicate.withArray("tasks").get(1)).put("id", root.withArray("tasks").get(0).get("id").asText()); + assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(duplicate.toString().getBytes(StandardCharsets.UTF_8))); + var invalid = root.deepCopy(); + ((ObjectNode) invalid.withArray("tasks").get(0).get("expected")).remove("completed"); + assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(invalid.toString().getBytes(StandardCharsets.UTF_8))); + var version = root.deepCopy(); + version.put("schemaVersion", 2); + assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(version.toString().getBytes(StandardCharsets.UTF_8))); + } + + @Test + void wrongExpectationIsReportedAndDoesNotStopRemainingCases() throws Exception { + ObjectNode root = (ObjectNode) OfflineGoalTaskReplay.JSON.readTree(fixture()); + var tasks = root.withArray("tasks"); + ObjectNode expected = (ObjectNode) tasks.get(0).get("expected"); + expected.put("completed", !expected.get("completed").asBoolean()); + var report = OfflineGoalTaskReplay.run(root.toString().getBytes(StandardCharsets.UTF_8), "test-revision"); + assertEquals(1, report.mismatchedCases()); + assertEquals(tasks.size(), report.cases().size()); + assertFalse(report.cases().getFirst().matched()); + assertNotNull(report.cases().getFirst().actual()); + } +} diff --git a/mateclaw-server/src/test/resources/agent-evaluation/README.md b/mateclaw-server/src/test/resources/agent-evaluation/README.md new file mode 100644 index 00000000..38ed31fd --- /dev/null +++ b/mateclaw-server/src/test/resources/agent-evaluation/README.md @@ -0,0 +1,62 @@ +# Offline Agent task policy replay (v1) + +This harness replays **synthetic evaluator replies** against the production +`GoalEvaluationService` and `GoalCriteriaCodec`. It executes neither a task-solving +Agent nor workspace commands, and it cannot establish model success rates, +artifact truth, prompt quality, online latency or cost. The mock ChatModel is the +only provider; no credentials or network are required by this harness. Maven may +need network access to resolve missing build dependencies. + +From the repository root (Java 21 and Maven): + +```sh +mvn -pl mateclaw-server -am -Dtest=OfflineGoalTaskReplayTest \ + -Dsurefire.failIfNoSpecifiedTests=false \ + -Dagent.eval.revision="$(git rev-parse HEAD)" test +``` + +The report is written to `mateclaw-server/target/agent-evaluation/goal-baseline.json`. +`-Dagent.eval.suite=/absolute/path/suite.json` selects an alternate suite; +`-Dagent.eval.report=/absolute/path/report.json` changes the output. Use separate +output paths when comparing versions. Without a revision argument the report +says `unrecorded`; do not use that report for revision comparisons. Record a dirty +working tree separately; a supplied Git revision identifies committed source, +not uncommitted modifications. The initial committed baseline's revision names +the production code from cycle-001; this new test harness was added afterward. + +Schema v1 has `schemaVersion`, `suiteId`, and 1–100 `tasks`. Each task declares: + +| Field | Meaning | +| --- | --- | +| `id` | Unique nonblank case ID, stable across revisions | +| `task` | Human-readable user task | +| `source` | Regression/test/design provenance; synthetic inputs are identified | +| `boundary` | Distinct behavior or known limitation under inspection | +| `persistent` | Whether cumulative Goal semantics apply | +| `criteria` | Initial Goal checklist, or `[]` for bootstrap | +| `terminalAnswer` | Fixed assistant final answer (empty is a valid boundary) | +| `evaluatorResponse` | Exact synthetic model response string, including malformed JSON cases | +| `expected` | Required completed flag, score, decision, ordered remaining criterion IDs and fixture call count | + +Expected scores must be finite in [0,1], decisions are `completed`, `continue`, or +`fallback`, and fixture call counts are 0 or 1. Empty suites, duplicate IDs, +missing expected values, unknown fields and trailing JSON are rejected. The +whole suite is validated before replay. Expectations are independent fixed +inputs, not generated by copying the current runtime result. For a deliberate +behavior change, review and explain each changed expectation. + +Reports bind the exact suite bytes with SHA-256 and record the supplied source +revision, execution mode, every expected/actual result, and mismatches. A mismatch +writes the report then fails the test/Maven command; all cases are still run. +Invalid suites fail before a new report is written, so an existing report may be +stale: always check the command exit status. `matchedCases` counts policy replay +agreement, **not successful Agent tasks**. `onlineModelCalls` is zero; +`agentTaskSuccessRate` and `onlineCost` are `not_measured`. + +Ten initial cases cover distinct completion boundaries. The forged-text case +intentionally expects semantic completion: no report file is created or checked. +That known limitation demonstrates why nonblank evidence and Observe are not +strong acceptance. A future trusted file recipe must use a separate execution +mode and record its actual filesystem/version checks. A real model comparison +must additionally pin runtime/skill/model versions and measure real calls; it +cannot reuse these matched counts as its success rate. diff --git a/mateclaw-server/src/test/resources/agent-evaluation/baseline-v1.json b/mateclaw-server/src/test/resources/agent-evaluation/baseline-v1.json new file mode 100644 index 00000000..ee8519f2 --- /dev/null +++ b/mateclaw-server/src/test/resources/agent-evaluation/baseline-v1.json @@ -0,0 +1,223 @@ +{ + "schemaVersion" : 1, + "suiteId" : "goal-boundaries-v1", + "suiteSha256" : "3399b3a40ea1f8ad2ac4307b85f4e045c3dba0b87d7474c9367486b3d6b6f78d", + "codeRevision" : "f606668565f5e340f79c3468d29551a5975a9e63", + "executionMode" : "offline_synthetic_evaluator_replay", + "onlineModelCalls" : 0, + "agentTaskSuccessRate" : "not_measured", + "onlineCost" : "not_measured", + "matchedCases" : 10, + "mismatchedCases" : 0, + "cases" : [ { + "id" : "empty-evidence", + "task" : "Deliver a report and verify its contents", + "source" : "cycle-001 reproduced failure; GoalCriteriaCodecTest.blankEvidenceCannotPassNewOrPersistedCriteria", + "boundary" : "Claimed pass without evidence is rejected", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1", "C2" ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1", "C2" ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "inherited-empty-pass", + "task" : "Finish a report after resuming a persisted checklist", + "source" : "cycle-001 persisted blank-evidence boundary", + "boundary" : "An omitted historical pass with blank evidence cannot complete", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "cumulative-progress", + "task" : "Verify a report created in a previous turn", + "source" : "GoalEvaluationServiceTest", + "boundary" : "Previously evidenced criterion survives omitted delta", + "expected" : { + "completed" : true, + "score" : 1.0, + "decision" : "completed", + "remainingIds" : [ ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : true, + "score" : 1.0, + "decision" : "completed", + "remainingIds" : [ ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "unknown-criterion", + "task" : "Create report.md with a fixed checklist", + "source" : "GoalCriteriaCodecTest.merge_unknownVerdictId_isIgnored", + "boundary" : "A model inventing C99 cannot pass C1", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "contradicted-prior", + "task" : "Recheck a previously created report", + "source" : "GoalEvaluationServiceTest", + "boundary" : "Explicit contradiction revokes the prior pass", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "bootstrap-only", + "task" : "Create a report with no checklist yet", + "source" : "GoalEvaluationServiceTest", + "boundary" : "Bootstrap defines criteria and cannot complete", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "continue", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "malformed-response", + "task" : "Verify the report after evaluator output corruption", + "source" : "GoalEvaluationServiceTest", + "boundary" : "Malformed model output degrades to fallback", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "fallback", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "fallback", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "valid-semantic-pass", + "task" : "Create report and verify a supplied excerpt", + "source" : "GoalEvaluationServiceTest", + "boundary" : "Nonblank semantic evidence retains legacy completion", + "expected" : { + "completed" : true, + "score" : 1.0, + "decision" : "completed", + "remainingIds" : [ ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : true, + "score" : 1.0, + "decision" : "completed", + "remainingIds" : [ ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + }, { + "id" : "empty-terminal-answer", + "task" : "Continue report task without a final answer", + "source" : "GoalEvaluationServiceTest", + "boundary" : "No answer means no evaluator fixture call", + "expected" : { + "completed" : false, + "score" : 0.0, + "decision" : "fallback", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 0 + }, + "actual" : { + "completed" : false, + "score" : 0.0, + "decision" : "fallback", + "remainingIds" : [ "C1" ], + "fixtureCalls" : 0 + }, + "matched" : true, + "error" : null + }, { + "id" : "forged-text-is-not-strong-acceptance", + "task" : "Deliver a file that exists in the workspace", + "source" : "RFC-096 text/Observe trust boundary; synthetic limitation probe, not observed model behavior", + "boundary" : "Known limitation: fabricated nonblank evidence still passes semantic policy; no file execution is performed", + "expected" : { + "completed" : true, + "score" : 1.0, + "decision" : "completed", + "remainingIds" : [ ], + "fixtureCalls" : 1 + }, + "actual" : { + "completed" : true, + "score" : 1.0, + "decision" : "completed", + "remainingIds" : [ ], + "fixtureCalls" : 1 + }, + "matched" : true, + "error" : null + } ] +} \ No newline at end of file diff --git a/mateclaw-server/src/test/resources/agent-evaluation/goal-boundaries-v1.json b/mateclaw-server/src/test/resources/agent-evaluation/goal-boundaries-v1.json new file mode 100644 index 00000000..ac12669b --- /dev/null +++ b/mateclaw-server/src/test/resources/agent-evaluation/goal-boundaries-v1.json @@ -0,0 +1,272 @@ +{ + "schemaVersion": 1, + "suiteId": "goal-boundaries-v1", + "tasks": [ + { + "id": "empty-evidence", + "task": "Deliver a report and verify its contents", + "source": "cycle-001 reproduced failure; GoalCriteriaCodecTest.blankEvidenceCannotPassNewOrPersistedCriteria", + "boundary": "Claimed pass without evidence is rejected", + "persistent": false, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": false, + "evidence": "" + }, + { + "id": "C2", + "text": "Verify report contents", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"\"}, {\"id\": \"C2\", \"passed\": true, \"evidence\": null}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": false, + "score": 0, + "decision": "continue", + "remainingIds": [ + "C1", + "C2" + ], + "fixtureCalls": 1 + } + }, + { + "id": "inherited-empty-pass", + "task": "Finish a report after resuming a persisted checklist", + "source": "cycle-001 persisted blank-evidence boundary", + "boundary": "An omitted historical pass with blank evidence cannot complete", + "persistent": true, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": true, + "evidence": "" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criterionVerdicts\": [], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": false, + "score": 0, + "decision": "continue", + "remainingIds": [ + "C1" + ], + "fixtureCalls": 1 + } + }, + { + "id": "cumulative-progress", + "task": "Verify a report created in a previous turn", + "source": "GoalEvaluationServiceTest", + "boundary": "Previously evidenced criterion survives omitted delta", + "persistent": true, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": true, + "evidence": "report snapshot from earlier turn" + }, + { + "id": "C2", + "text": "Verify report contents", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C2\", \"passed\": true, \"evidence\": \"Read report: required heading present\"}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": true, + "score": 1, + "decision": "completed", + "remainingIds": [], + "fixtureCalls": 1 + } + }, + { + "id": "unknown-criterion", + "task": "Create report.md with a fixed checklist", + "source": "GoalCriteriaCodecTest.merge_unknownVerdictId_isIgnored", + "boundary": "A model inventing C99 cannot pass C1", + "persistent": false, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C99\", \"passed\": true, \"evidence\": \"done\"}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": false, + "score": 0, + "decision": "continue", + "remainingIds": [ + "C1" + ], + "fixtureCalls": 1 + } + }, + { + "id": "contradicted-prior", + "task": "Recheck a previously created report", + "source": "GoalEvaluationServiceTest", + "boundary": "Explicit contradiction revokes the prior pass", + "persistent": true, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": true, + "evidence": "file observed" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": false, \"evidence\": \"report.md was removed\"}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": false, + "score": 0, + "decision": "continue", + "remainingIds": [ + "C1" + ], + "fixtureCalls": 1 + } + }, + { + "id": "bootstrap-only", + "task": "Create a report with no checklist yet", + "source": "GoalEvaluationServiceTest", + "boundary": "Bootstrap defines criteria and cannot complete", + "persistent": false, + "criteria": [], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criteria\": [{\"id\": \"ignored\", \"text\": \"Create report.md\", \"passed\": true, \"evidence\": \"model says done\"}]}", + "expected": { + "completed": false, + "score": 0, + "decision": "continue", + "remainingIds": [ + "C1" + ], + "fixtureCalls": 1 + } + }, + { + "id": "malformed-response", + "task": "Verify the report after evaluator output corruption", + "source": "GoalEvaluationServiceTest", + "boundary": "Malformed model output degrades to fallback", + "persistent": false, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{not JSON", + "expected": { + "completed": false, + "score": 0, + "decision": "fallback", + "remainingIds": [ + "C1" + ], + "fixtureCalls": 1 + } + }, + { + "id": "valid-semantic-pass", + "task": "Create report and verify a supplied excerpt", + "source": "GoalEvaluationServiceTest", + "boundary": "Nonblank semantic evidence retains legacy completion", + "persistent": false, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": false, + "evidence": "" + }, + { + "id": "C2", + "text": "Verify report contents", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "Report work is finished.", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"report.md excerpt: Results\"}, {\"id\": \"C2\", \"passed\": true, \"evidence\": \"Required Results section present\"}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": true, + "score": 1, + "decision": "completed", + "remainingIds": [], + "fixtureCalls": 1 + } + }, + { + "id": "empty-terminal-answer", + "task": "Continue report task without a final answer", + "source": "GoalEvaluationServiceTest", + "boundary": "No answer means no evaluator fixture call", + "persistent": false, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"done\"}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": false, + "score": 0, + "decision": "fallback", + "remainingIds": [ + "C1" + ], + "fixtureCalls": 0 + } + }, + { + "id": "forged-text-is-not-strong-acceptance", + "task": "Deliver a file that exists in the workspace", + "source": "RFC-096 text/Observe trust boundary; synthetic limitation probe, not observed model behavior", + "boundary": "Known limitation: fabricated nonblank evidence still passes semantic policy; no file execution is performed", + "persistent": false, + "criteria": [ + { + "id": "C1", + "text": "Create report.md", + "passed": false, + "evidence": "" + } + ], + "terminalAnswer": "I saved report.md.", + "evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"report.md saved and verified\"}], \"summary\": \"fixed synthetic reply\"}", + "expected": { + "completed": true, + "score": 1, + "decision": "completed", + "remainingIds": [], + "fixtureCalls": 1 + } + } + ] +}