test(evaluation): add offline goal task replay and baseline

This commit is contained in:
mateaix 2026-09-13 20:48:02 +08:00
parent 88cd78af5e
commit 35a241a62d
5 changed files with 766 additions and 0 deletions

View File

@ -0,0 +1,147 @@
package vip.mate.evaluation;
import com.fasterxml.jackson.databind.DeserializationFeature;
import com.fasterxml.jackson.databind.ObjectMapper;
import org.springframework.ai.chat.messages.AssistantMessage;
import org.springframework.ai.chat.model.ChatModel;
import org.springframework.ai.chat.model.ChatResponse;
import org.springframework.ai.chat.model.Generation;
import org.springframework.ai.chat.prompt.Prompt;
import org.springframework.retry.support.RetryTemplate;
import vip.mate.goal.config.GoalProperties;
import vip.mate.goal.model.GoalCriteriaCodec;
import vip.mate.goal.model.GoalCriterion;
import vip.mate.goal.model.GoalEntity;
import vip.mate.goal.service.GoalEvaluationService;
import vip.mate.llm.chatmodel.ProviderChatModelFactory;
import vip.mate.llm.model.ModelConfigEntity;
import vip.mate.llm.service.ModelConfigService;
import java.io.IOException;
import java.security.MessageDigest;
import java.security.NoSuchAlgorithmException;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.HexFormat;
import java.util.List;
import java.util.Set;
import static org.mockito.ArgumentMatchers.any;
import static org.mockito.Mockito.mock;
import static org.mockito.Mockito.when;
/** Offline policy replay. The model is a fixture: no agent task is actually executed. */
final class OfflineGoalTaskReplay {
static final ObjectMapper JSON = new ObjectMapper()
.enable(DeserializationFeature.FAIL_ON_TRAILING_TOKENS);
static final String MODE = "offline_synthetic_evaluator_replay";
record Suite(Integer schemaVersion, String suiteId, List<Task> tasks) { }
record Task(String id, String task, String source, String boundary, Boolean persistent,
List<GoalCriterion> criteria, String terminalAnswer, String evaluatorResponse,
Expected expected) { }
record Expected(Boolean completed, Double score, String decision, List<String> remainingIds,
Integer fixtureCalls) { }
record Actual(boolean completed, double score, String decision, List<String> remainingIds,
int fixtureCalls) { }
record CaseResult(String id, String task, String source, String boundary, Expected expected,
Actual actual, boolean matched, String error) { }
record Report(int schemaVersion, String suiteId, String suiteSha256, String codeRevision,
String executionMode, int onlineModelCalls, String agentTaskSuccessRate,
String onlineCost, int matchedCases, int mismatchedCases, List<CaseResult> cases) { }
static Suite parse(byte[] bytes) throws IOException {
Suite suite = JSON.readValue(bytes, Suite.class);
require(suite != null && Integer.valueOf(1).equals(suite.schemaVersion()), "schemaVersion must be 1");
require(nonblank(suite.suiteId()), "suiteId is required");
require(suite.tasks() != null && !suite.tasks().isEmpty() && suite.tasks().size() <= 100,
"suite must contain 1..100 tasks");
Set<String> ids = new HashSet<>();
for (Task task : suite.tasks()) {
require(task != null && nonblank(task.id()) && ids.add(task.id()), "task IDs must be unique and nonblank");
require(nonblank(task.task()) && nonblank(task.source()) && nonblank(task.boundary()),
task.id() + ": task, source and boundary are required");
require(task.persistent() != null && task.criteria() != null && task.terminalAnswer() != null
&& task.evaluatorResponse() != null, task.id() + ": missing replay inputs");
Set<String> criterionIds = new HashSet<>();
for (GoalCriterion criterion : task.criteria()) {
require(criterion != null && nonblank(criterion.id()) && nonblank(criterion.text())
&& criterionIds.add(criterion.id()), task.id() + ": invalid or duplicate criterion");
}
Expected expected = task.expected();
require(expected != null && expected.completed() != null && expected.score() != null
&& Double.isFinite(expected.score()) && expected.score() >= 0 && expected.score() <= 1
&& Set.of("completed", "continue", "fallback").contains(expected.decision() == null ? "" : expected.decision())
&& expected.remainingIds() != null && expected.fixtureCalls() != null
&& expected.fixtureCalls() >= 0 && expected.fixtureCalls() <= 1,
task.id() + ": incomplete or invalid expected outcome");
require(expected.remainingIds().stream().allMatch(OfflineGoalTaskReplay::nonblank)
&& new HashSet<>(expected.remainingIds()).size() == expected.remainingIds().size(),
task.id() + ": remainingIds must be unique and nonblank");
}
return suite;
}
static Report run(byte[] bytes, String revision) throws IOException {
Suite suite = parse(bytes); // Validate the whole suite before any fixture is replayed.
require(nonblank(revision), "codeRevision is required");
List<CaseResult> results = new ArrayList<>();
for (Task task : suite.tasks()) {
try {
Actual actual = replay(task);
Expected expected = task.expected();
boolean matched = expected.completed() == actual.completed()
&& Math.abs(expected.score() - actual.score()) < 1e-9
&& expected.decision().equals(actual.decision())
&& expected.remainingIds().equals(actual.remainingIds())
&& expected.fixtureCalls() == actual.fixtureCalls();
results.add(new CaseResult(task.id(), task.task(), task.source(), task.boundary(),
expected, actual, matched, null));
} catch (RuntimeException error) {
results.add(new CaseResult(task.id(), task.task(), task.source(), task.boundary(),
task.expected(), null, false, error.getClass().getSimpleName()));
}
}
int matched = (int) results.stream().filter(CaseResult::matched).count();
return new Report(1, suite.suiteId(), digest(bytes), revision, MODE, 0,
"not_measured", "not_measured", matched, results.size() - matched, List.copyOf(results));
}
private static Actual replay(Task task) {
ChatModel chat = mock(ChatModel.class);
ModelConfigService configs = mock(ModelConfigService.class);
ProviderChatModelFactory factory = mock(ProviderChatModelFactory.class);
ModelConfigEntity model = new ModelConfigEntity();
model.setModelName("offline-fixture");
when(configs.getDefaultModel()).thenReturn(model);
when(factory.buildFor(any(ModelConfigEntity.class), any(RetryTemplate.class))).thenReturn(chat);
int[] fixtureCalls = {0};
when(chat.call(any(Prompt.class))).thenAnswer(call -> {
fixtureCalls[0]++;
return new ChatResponse(List.of(new Generation(new AssistantMessage(task.evaluatorResponse()))));
});
GoalEntity goal = new GoalEntity();
goal.setTitle(task.task());
goal.setPersistentExecution(task.persistent());
goal.setCriteria(GoalCriteriaCodec.serialize(task.criteria(), JSON));
var evaluator = new GoalEvaluationService(new GoalProperties(), configs, factory, JSON);
var result = evaluator.evaluate(goal, List.of(), task.terminalAnswer());
var merged = result.bootstrapCriteria() != null ? result.bootstrapCriteria()
: GoalCriteriaCodec.merge(task.criteria(), result.criterionVerdicts());
return new Actual(result.completed(), result.score(), result.decision(),
GoalCriteriaCodec.remaining(merged).stream().map(GoalCriterion::id).toList(), fixtureCalls[0]);
}
private static String digest(byte[] bytes) {
try {
return HexFormat.of().formatHex(MessageDigest.getInstance("SHA-256").digest(bytes));
} catch (NoSuchAlgorithmException impossible) {
throw new IllegalStateException(impossible);
}
}
private static boolean nonblank(String value) { return value != null && !value.isBlank(); }
private static void require(boolean valid, String message) {
if (!valid) throw new IllegalArgumentException(message);
}
}

View File

@ -0,0 +1,62 @@
package vip.mate.evaluation;
import com.fasterxml.jackson.databind.node.ObjectNode;
import org.junit.jupiter.api.Test;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import static org.junit.jupiter.api.Assertions.*;
class OfflineGoalTaskReplayTest {
private byte[] fixture() throws Exception {
try (var stream = getClass().getResourceAsStream("/agent-evaluation/goal-boundaries-v1.json")) {
assertNotNull(stream);
return stream.readAllBytes();
}
}
@Test
void replayTasksAndWriteReportEvenWhenExpectationsMismatch() throws Exception {
String input = System.getProperty("agent.eval.suite");
byte[] bytes = input == null ? fixture() : Files.readAllBytes(Path.of(input));
var report = OfflineGoalTaskReplay.run(bytes, System.getProperty("agent.eval.revision", "unrecorded"));
Path output = Path.of(System.getProperty("agent.eval.report", "target/agent-evaluation/goal-baseline.json"));
Files.createDirectories(output.toAbsolutePath().getParent());
OfflineGoalTaskReplay.JSON.writerWithDefaultPrettyPrinter().writeValue(output.toFile(), report);
assertEquals(0, report.mismatchedCases(), () -> "Replay mismatches: " + output.toAbsolutePath());
assertEquals(0, report.onlineModelCalls());
assertEquals("not_measured", report.agentTaskSuccessRate());
}
@Test
void invalidSuiteIsRejectedBeforeReplay() throws Exception {
ObjectNode root = (ObjectNode) OfflineGoalTaskReplay.JSON.readTree(fixture());
var empty = root.deepCopy();
empty.putArray("tasks");
assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(empty.toString().getBytes(StandardCharsets.UTF_8)));
var duplicate = root.deepCopy();
((ObjectNode) duplicate.withArray("tasks").get(1)).put("id", root.withArray("tasks").get(0).get("id").asText());
assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(duplicate.toString().getBytes(StandardCharsets.UTF_8)));
var invalid = root.deepCopy();
((ObjectNode) invalid.withArray("tasks").get(0).get("expected")).remove("completed");
assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(invalid.toString().getBytes(StandardCharsets.UTF_8)));
var version = root.deepCopy();
version.put("schemaVersion", 2);
assertThrows(IllegalArgumentException.class, () -> OfflineGoalTaskReplay.parse(version.toString().getBytes(StandardCharsets.UTF_8)));
}
@Test
void wrongExpectationIsReportedAndDoesNotStopRemainingCases() throws Exception {
ObjectNode root = (ObjectNode) OfflineGoalTaskReplay.JSON.readTree(fixture());
var tasks = root.withArray("tasks");
ObjectNode expected = (ObjectNode) tasks.get(0).get("expected");
expected.put("completed", !expected.get("completed").asBoolean());
var report = OfflineGoalTaskReplay.run(root.toString().getBytes(StandardCharsets.UTF_8), "test-revision");
assertEquals(1, report.mismatchedCases());
assertEquals(tasks.size(), report.cases().size());
assertFalse(report.cases().getFirst().matched());
assertNotNull(report.cases().getFirst().actual());
}
}

View File

@ -0,0 +1,62 @@
# Offline Agent task policy replay (v1)
This harness replays **synthetic evaluator replies** against the production
`GoalEvaluationService` and `GoalCriteriaCodec`. It executes neither a task-solving
Agent nor workspace commands, and it cannot establish model success rates,
artifact truth, prompt quality, online latency or cost. The mock ChatModel is the
only provider; no credentials or network are required by this harness. Maven may
need network access to resolve missing build dependencies.
From the repository root (Java 21 and Maven):
```sh
mvn -pl mateclaw-server -am -Dtest=OfflineGoalTaskReplayTest \
-Dsurefire.failIfNoSpecifiedTests=false \
-Dagent.eval.revision="$(git rev-parse HEAD)" test
```
The report is written to `mateclaw-server/target/agent-evaluation/goal-baseline.json`.
`-Dagent.eval.suite=/absolute/path/suite.json` selects an alternate suite;
`-Dagent.eval.report=/absolute/path/report.json` changes the output. Use separate
output paths when comparing versions. Without a revision argument the report
says `unrecorded`; do not use that report for revision comparisons. Record a dirty
working tree separately; a supplied Git revision identifies committed source,
not uncommitted modifications. The initial committed baseline's revision names
the production code from cycle-001; this new test harness was added afterward.
Schema v1 has `schemaVersion`, `suiteId`, and 1100 `tasks`. Each task declares:
| Field | Meaning |
| --- | --- |
| `id` | Unique nonblank case ID, stable across revisions |
| `task` | Human-readable user task |
| `source` | Regression/test/design provenance; synthetic inputs are identified |
| `boundary` | Distinct behavior or known limitation under inspection |
| `persistent` | Whether cumulative Goal semantics apply |
| `criteria` | Initial Goal checklist, or `[]` for bootstrap |
| `terminalAnswer` | Fixed assistant final answer (empty is a valid boundary) |
| `evaluatorResponse` | Exact synthetic model response string, including malformed JSON cases |
| `expected` | Required completed flag, score, decision, ordered remaining criterion IDs and fixture call count |
Expected scores must be finite in [0,1], decisions are `completed`, `continue`, or
`fallback`, and fixture call counts are 0 or 1. Empty suites, duplicate IDs,
missing expected values, unknown fields and trailing JSON are rejected. The
whole suite is validated before replay. Expectations are independent fixed
inputs, not generated by copying the current runtime result. For a deliberate
behavior change, review and explain each changed expectation.
Reports bind the exact suite bytes with SHA-256 and record the supplied source
revision, execution mode, every expected/actual result, and mismatches. A mismatch
writes the report then fails the test/Maven command; all cases are still run.
Invalid suites fail before a new report is written, so an existing report may be
stale: always check the command exit status. `matchedCases` counts policy replay
agreement, **not successful Agent tasks**. `onlineModelCalls` is zero;
`agentTaskSuccessRate` and `onlineCost` are `not_measured`.
Ten initial cases cover distinct completion boundaries. The forged-text case
intentionally expects semantic completion: no report file is created or checked.
That known limitation demonstrates why nonblank evidence and Observe are not
strong acceptance. A future trusted file recipe must use a separate execution
mode and record its actual filesystem/version checks. A real model comparison
must additionally pin runtime/skill/model versions and measure real calls; it
cannot reuse these matched counts as its success rate.

View File

@ -0,0 +1,223 @@
{
"schemaVersion" : 1,
"suiteId" : "goal-boundaries-v1",
"suiteSha256" : "3399b3a40ea1f8ad2ac4307b85f4e045c3dba0b87d7474c9367486b3d6b6f78d",
"codeRevision" : "f606668565f5e340f79c3468d29551a5975a9e63",
"executionMode" : "offline_synthetic_evaluator_replay",
"onlineModelCalls" : 0,
"agentTaskSuccessRate" : "not_measured",
"onlineCost" : "not_measured",
"matchedCases" : 10,
"mismatchedCases" : 0,
"cases" : [ {
"id" : "empty-evidence",
"task" : "Deliver a report and verify its contents",
"source" : "cycle-001 reproduced failure; GoalCriteriaCodecTest.blankEvidenceCannotPassNewOrPersistedCriteria",
"boundary" : "Claimed pass without evidence is rejected",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1", "C2" ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1", "C2" ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "inherited-empty-pass",
"task" : "Finish a report after resuming a persisted checklist",
"source" : "cycle-001 persisted blank-evidence boundary",
"boundary" : "An omitted historical pass with blank evidence cannot complete",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "cumulative-progress",
"task" : "Verify a report created in a previous turn",
"source" : "GoalEvaluationServiceTest",
"boundary" : "Previously evidenced criterion survives omitted delta",
"expected" : {
"completed" : true,
"score" : 1.0,
"decision" : "completed",
"remainingIds" : [ ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : true,
"score" : 1.0,
"decision" : "completed",
"remainingIds" : [ ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "unknown-criterion",
"task" : "Create report.md with a fixed checklist",
"source" : "GoalCriteriaCodecTest.merge_unknownVerdictId_isIgnored",
"boundary" : "A model inventing C99 cannot pass C1",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "contradicted-prior",
"task" : "Recheck a previously created report",
"source" : "GoalEvaluationServiceTest",
"boundary" : "Explicit contradiction revokes the prior pass",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "bootstrap-only",
"task" : "Create a report with no checklist yet",
"source" : "GoalEvaluationServiceTest",
"boundary" : "Bootstrap defines criteria and cannot complete",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "continue",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "malformed-response",
"task" : "Verify the report after evaluator output corruption",
"source" : "GoalEvaluationServiceTest",
"boundary" : "Malformed model output degrades to fallback",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "fallback",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "fallback",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "valid-semantic-pass",
"task" : "Create report and verify a supplied excerpt",
"source" : "GoalEvaluationServiceTest",
"boundary" : "Nonblank semantic evidence retains legacy completion",
"expected" : {
"completed" : true,
"score" : 1.0,
"decision" : "completed",
"remainingIds" : [ ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : true,
"score" : 1.0,
"decision" : "completed",
"remainingIds" : [ ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
}, {
"id" : "empty-terminal-answer",
"task" : "Continue report task without a final answer",
"source" : "GoalEvaluationServiceTest",
"boundary" : "No answer means no evaluator fixture call",
"expected" : {
"completed" : false,
"score" : 0.0,
"decision" : "fallback",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 0
},
"actual" : {
"completed" : false,
"score" : 0.0,
"decision" : "fallback",
"remainingIds" : [ "C1" ],
"fixtureCalls" : 0
},
"matched" : true,
"error" : null
}, {
"id" : "forged-text-is-not-strong-acceptance",
"task" : "Deliver a file that exists in the workspace",
"source" : "RFC-096 text/Observe trust boundary; synthetic limitation probe, not observed model behavior",
"boundary" : "Known limitation: fabricated nonblank evidence still passes semantic policy; no file execution is performed",
"expected" : {
"completed" : true,
"score" : 1.0,
"decision" : "completed",
"remainingIds" : [ ],
"fixtureCalls" : 1
},
"actual" : {
"completed" : true,
"score" : 1.0,
"decision" : "completed",
"remainingIds" : [ ],
"fixtureCalls" : 1
},
"matched" : true,
"error" : null
} ]
}

View File

@ -0,0 +1,272 @@
{
"schemaVersion": 1,
"suiteId": "goal-boundaries-v1",
"tasks": [
{
"id": "empty-evidence",
"task": "Deliver a report and verify its contents",
"source": "cycle-001 reproduced failure; GoalCriteriaCodecTest.blankEvidenceCannotPassNewOrPersistedCriteria",
"boundary": "Claimed pass without evidence is rejected",
"persistent": false,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": false,
"evidence": ""
},
{
"id": "C2",
"text": "Verify report contents",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"\"}, {\"id\": \"C2\", \"passed\": true, \"evidence\": null}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": false,
"score": 0,
"decision": "continue",
"remainingIds": [
"C1",
"C2"
],
"fixtureCalls": 1
}
},
{
"id": "inherited-empty-pass",
"task": "Finish a report after resuming a persisted checklist",
"source": "cycle-001 persisted blank-evidence boundary",
"boundary": "An omitted historical pass with blank evidence cannot complete",
"persistent": true,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": true,
"evidence": ""
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criterionVerdicts\": [], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": false,
"score": 0,
"decision": "continue",
"remainingIds": [
"C1"
],
"fixtureCalls": 1
}
},
{
"id": "cumulative-progress",
"task": "Verify a report created in a previous turn",
"source": "GoalEvaluationServiceTest",
"boundary": "Previously evidenced criterion survives omitted delta",
"persistent": true,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": true,
"evidence": "report snapshot from earlier turn"
},
{
"id": "C2",
"text": "Verify report contents",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C2\", \"passed\": true, \"evidence\": \"Read report: required heading present\"}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": true,
"score": 1,
"decision": "completed",
"remainingIds": [],
"fixtureCalls": 1
}
},
{
"id": "unknown-criterion",
"task": "Create report.md with a fixed checklist",
"source": "GoalCriteriaCodecTest.merge_unknownVerdictId_isIgnored",
"boundary": "A model inventing C99 cannot pass C1",
"persistent": false,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C99\", \"passed\": true, \"evidence\": \"done\"}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": false,
"score": 0,
"decision": "continue",
"remainingIds": [
"C1"
],
"fixtureCalls": 1
}
},
{
"id": "contradicted-prior",
"task": "Recheck a previously created report",
"source": "GoalEvaluationServiceTest",
"boundary": "Explicit contradiction revokes the prior pass",
"persistent": true,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": true,
"evidence": "file observed"
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": false, \"evidence\": \"report.md was removed\"}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": false,
"score": 0,
"decision": "continue",
"remainingIds": [
"C1"
],
"fixtureCalls": 1
}
},
{
"id": "bootstrap-only",
"task": "Create a report with no checklist yet",
"source": "GoalEvaluationServiceTest",
"boundary": "Bootstrap defines criteria and cannot complete",
"persistent": false,
"criteria": [],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criteria\": [{\"id\": \"ignored\", \"text\": \"Create report.md\", \"passed\": true, \"evidence\": \"model says done\"}]}",
"expected": {
"completed": false,
"score": 0,
"decision": "continue",
"remainingIds": [
"C1"
],
"fixtureCalls": 1
}
},
{
"id": "malformed-response",
"task": "Verify the report after evaluator output corruption",
"source": "GoalEvaluationServiceTest",
"boundary": "Malformed model output degrades to fallback",
"persistent": false,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{not JSON",
"expected": {
"completed": false,
"score": 0,
"decision": "fallback",
"remainingIds": [
"C1"
],
"fixtureCalls": 1
}
},
{
"id": "valid-semantic-pass",
"task": "Create report and verify a supplied excerpt",
"source": "GoalEvaluationServiceTest",
"boundary": "Nonblank semantic evidence retains legacy completion",
"persistent": false,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": false,
"evidence": ""
},
{
"id": "C2",
"text": "Verify report contents",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "Report work is finished.",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"report.md excerpt: Results\"}, {\"id\": \"C2\", \"passed\": true, \"evidence\": \"Required Results section present\"}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": true,
"score": 1,
"decision": "completed",
"remainingIds": [],
"fixtureCalls": 1
}
},
{
"id": "empty-terminal-answer",
"task": "Continue report task without a final answer",
"source": "GoalEvaluationServiceTest",
"boundary": "No answer means no evaluator fixture call",
"persistent": false,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"done\"}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": false,
"score": 0,
"decision": "fallback",
"remainingIds": [
"C1"
],
"fixtureCalls": 0
}
},
{
"id": "forged-text-is-not-strong-acceptance",
"task": "Deliver a file that exists in the workspace",
"source": "RFC-096 text/Observe trust boundary; synthetic limitation probe, not observed model behavior",
"boundary": "Known limitation: fabricated nonblank evidence still passes semantic policy; no file execution is performed",
"persistent": false,
"criteria": [
{
"id": "C1",
"text": "Create report.md",
"passed": false,
"evidence": ""
}
],
"terminalAnswer": "I saved report.md.",
"evaluatorResponse": "{\"criterionVerdicts\": [{\"id\": \"C1\", \"passed\": true, \"evidence\": \"report.md saved and verified\"}], \"summary\": \"fixed synthetic reply\"}",
"expected": {
"completed": true,
"score": 1,
"decision": "completed",
"remainingIds": [],
"fixtureCalls": 1
}
}
]
}