feat(agent): deterministic Markdown normalization for final answers (#275)

LLMs routinely emit malformed Markdown (missing heading spaces, glued `---`, unaligned table pipes) that prompt rules cannot reliably prevent. Add a zero-token, regex-only MarkdownNormalizer applied on the FinalAnswerNode convergence path before persistence / channel delivery. It is code-fence aware, idempotent, and conservative (em-dash `---`, `#5`-style refs, stray prose pipes are left untouched). RETURN_DIRECT verbatim output and approval-wait paths return earlier and are unaffected.

Closes #274
This commit is contained in:
倪程伟 2026-06-07 19:10:53 +08:00 committed by GitHub
parent a9698dbed3
commit 398d7a2d80
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
3 changed files with 486 additions and 0 deletions

View File

@ -8,6 +8,7 @@ import vip.mate.agent.graph.state.DirectToolOutput;
import vip.mate.agent.graph.state.FinishReason;
import vip.mate.agent.graph.state.MateClawStateAccessor;
import vip.mate.agent.graph.state.SourceEvidenceLedger;
import vip.mate.common.text.MarkdownNormalizer;
import vip.mate.tool.document.GeneratedFileCache;
import java.util.List;
@ -170,6 +171,14 @@ public class FinalAnswerNode implements NodeAction {
validation.unsupportedReferences());
}
// Deterministic Markdown cleanup on the model-generated answer body. LLMs
// routinely emit malformed Markdown (missing heading spaces, glued `---`,
// unaligned table pipes) that prompt rules fail to prevent; this fixes the
// mechanical defects before the answer is persisted / sent to channels.
// Verbatim tool output (RETURN_DIRECT) and approval-wait paths return early
// above and are intentionally left untouched.
finalAnswer = MarkdownNormalizer.normalize(finalAnswer);
// Build the event list. Always carries the finish_reason event so
// downstream consumers (memory gate, channel accumulator, message
// metadata persistence) see a machine-readable status. When the

View File

@ -0,0 +1,293 @@
package vip.mate.common.text;
import java.util.ArrayList;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
/**
* 确定性 Markdown 规范化工具
*
* <p>用于在 Agent 最终答案落库 / 发渠道前修复 LLM 原始输出中常见的机械排版缺陷纯本地正则处理
* 不调用任何模型 token设计目标是修畸形而不改语义因此遵循以下原则</p>
*
* <ul>
* <li><b>代码块感知</b>先按 ``` / ~~~ 围栏切分围栏内的内容原样保留避免破坏代码里的
* {@code #} / {@code |} / {@code ---}</li>
* <li><b>幂等</b>{@code normalize(normalize(x)).equals(normalize(x))}</li>
* <li><b>保守</b>只在能高置信判断为畸形时才改写散文中的散落管道符行内 {@code #} 不动</li>
* </ul>
*
* <p>覆盖的修复ATX 标题补空格{@code ---} 与后续内容粘连时拆行含行首与 mid-line 后接标题两种
* 表格块单元格与分隔行对齐标题与表格粘连时拆行标题/表格块边界补空行</p>
*
* <p>不在范围内属语义判断正则无法安全自动化保留在提示词约束Emoji 位置代码块语言标注补全</p>
*/
public final class MarkdownNormalizer {
private MarkdownNormalizer() {}
/** 行首 ATX 标题但紧跟非空格、非 # 字符(缺少标题空格)。 */
private static final Pattern HEADING_NO_SPACE = Pattern.compile("^(#{1,6})([^#\\s].*)$");
/** 已规范的标题行:#{1,6} + 空白。用于块边界判断。 */
private static final Pattern HEADING_LINE = Pattern.compile("^#{1,6}\\s.*$");
/** 主题分隔线与后续内容粘连:行首 3+ 短横,后面紧跟非短横的可见内容。 */
private static final Pattern HR_GLUED = Pattern.compile("^(-{3,})([^-\\s].*)$");
/**
* 主题分隔线粘连在行内容之后mid-line且后面紧跟一个 ATX 标题
* 形如 {@code *来源2026-06-01*---### } {@code **90%**---### 综合判断}
* 仅在 {@code ---} 后紧跟 {@code #} 标题时才拆避免误伤散文里 em-dash 风格的 {@code ---}
*/
private static final Pattern HR_MID_HEADING =
Pattern.compile("^(.*?\\S)\\s*(-{3,})\\s*(#{1,6}.*)$");
/** 标题行尾粘连了表格:#{1,6} 标题文字(不含管道符)+ 管道符起始的尾部。 */
private static final Pattern HEADING_TABLE = Pattern.compile("^(#{1,6}[^|\\n]*?)\\s*(\\|.+)$");
/** GFM 表格分隔行:由短横/冒号组成的单元格,用管道符分隔(必须同时含 - 与 |)。 */
private static final Pattern SEPARATOR_ROW =
Pattern.compile("^\\s*\\|?\\s*:?-{1,}:?\\s*(\\|\\s*:?-{1,}:?\\s*)*\\|?\\s*$");
/**
* 规范化 Markdown 文本{@code null} 或空串原样返回
*/
public static String normalize(String md) {
if (md == null || md.isEmpty()) {
return md;
}
String normalized = md.replace("\r\n", "\n").replace("\r", "\n");
String[] lines = normalized.split("\n", -1);
List<String> out = new ArrayList<>();
List<String> textBuf = new ArrayList<>();
boolean inFence = false;
String fenceMarker = null;
for (String line : lines) {
String lead = line.stripLeading();
if (!inFence && (lead.startsWith("```") || lead.startsWith("~~~"))) {
flushText(textBuf, out);
textBuf.clear();
inFence = true;
fenceMarker = lead.startsWith("```") ? "```" : "~~~";
out.add(line);
} else if (inFence && lead.startsWith(fenceMarker)) {
inFence = false;
fenceMarker = null;
out.add(line);
} else if (inFence) {
out.add(line);
} else {
textBuf.add(line);
}
}
flushText(textBuf, out);
String result = String.join("\n", out);
// 仅清理文档首尾多余空行不触碰代码块内部
return result.replaceAll("^\\n+", "").replaceAll("\\n+$", "");
}
// ==================== 非代码段处理 ====================
private static void flushText(List<String> textLines, List<String> out) {
if (textLines.isEmpty()) {
return;
}
// 1. 行级展开HR 粘连拆行标题粘表格拆行标题补空格
List<String> expanded = new ArrayList<>();
for (String l : textLines) {
expanded.addAll(expandLine(l));
}
// 2. 表格块识别与规范化
List<String> normalizedLines = new ArrayList<>();
List<Boolean> isTable = new ArrayList<>();
normalizeTables(expanded, normalizedLines, isTable);
// 3. 标题/表格块边界补空行 + 折叠多余空行
out.addAll(insertBoundaryBlanks(normalizedLines, isTable));
}
private static List<String> expandLine(String line) {
List<String> result = new ArrayList<>();
Matcher hrMid = HR_MID_HEADING.matcher(line);
if (hrMid.matches()) {
result.addAll(expandLine(hrMid.group(1)));
result.add("");
result.add("---");
result.add("");
result.addAll(expandLine(hrMid.group(3)));
return result;
}
Matcher hr = HR_GLUED.matcher(line);
if (hr.matches()) {
result.add("---");
result.add("");
result.addAll(expandLine(hr.group(2)));
return result;
}
Matcher ht = HEADING_TABLE.matcher(line);
if (ht.matches() && countPipes(ht.group(2)) >= 2) {
result.add(fixHeadingSpace(ht.group(1).strip()));
result.add("");
result.add(ht.group(2).strip());
return result;
}
result.add(fixHeadingSpace(line));
return result;
}
/**
* 行首 ATX 标题缺空格时补一个空格为避免误伤 {@code #5}{@code #1} 这类引用
* 紧跟数字的不处理
*/
private static String fixHeadingSpace(String line) {
Matcher m = HEADING_NO_SPACE.matcher(line);
if (m.matches()) {
String hashes = m.group(1);
String rest = m.group(2);
if (!Character.isDigit(rest.charAt(0))) {
return hashes + " " + rest;
}
}
return line;
}
// ==================== 表格规范化 ====================
private static void normalizeTables(List<String> lines, List<String> out, List<Boolean> isTable) {
int n = lines.size();
boolean[] tbl = new boolean[n];
for (int i = 0; i < n; i++) {
if (isSeparatorRow(lines.get(i)) && i > 0 && containsPipe(lines.get(i - 1))) {
int start = i - 1;
int end = i;
int j = i + 1;
while (j < n && !lines.get(j).isBlank() && containsPipe(lines.get(j))
&& !HEADING_LINE.matcher(lines.get(j)).matches()) {
end = j;
j++;
}
for (int k = start; k <= end; k++) {
tbl[k] = true;
}
i = end;
}
}
for (int i = 0; i < n; i++) {
if (tbl[i]) {
out.add(normalizeTableRow(lines.get(i), isSeparatorRow(lines.get(i))));
} else {
out.add(lines.get(i));
}
isTable.add(tbl[i]);
}
}
private static String normalizeTableRow(String line, boolean separator) {
String s = line.strip();
if (s.startsWith("|")) {
s = s.substring(1);
}
if (s.endsWith("|")) {
s = s.substring(0, s.length() - 1);
}
String[] cells = s.split("(?<!\\\\)\\|", -1);
StringBuilder sb = new StringBuilder("|");
for (String cell : cells) {
String c = cell.strip();
if (separator) {
c = normalizeDelimiterCell(c);
}
sb.append(' ').append(c).append(" |");
}
return sb.toString();
}
private static String normalizeDelimiterCell(String cell) {
boolean left = cell.startsWith(":");
boolean right = cell.endsWith(":");
return (left ? ":" : "") + "---" + (right ? ":" : "");
}
private static boolean isSeparatorRow(String line) {
return line.indexOf('-') >= 0 && line.indexOf('|') >= 0
&& SEPARATOR_ROW.matcher(line).matches();
}
private static boolean containsPipe(String line) {
return line.indexOf('|') >= 0;
}
private static int countPipes(String s) {
int count = 0;
for (int i = 0; i < s.length(); i++) {
if (s.charAt(i) == '|') {
count++;
}
}
return count;
}
// ==================== 块边界空行 ====================
private static final int BLANK = 0;
private static final int HEADING = 1;
private static final int TABLE = 2;
private static final int OTHER = 3;
private static final int NONE = -1;
private static List<String> insertBoundaryBlanks(List<String> lines, List<Boolean> isTable) {
List<String> res = new ArrayList<>();
int lastType = NONE;
for (int i = 0; i < lines.size(); i++) {
String cur = lines.get(i);
int curType = classify(cur, isTable.get(i));
if (curType == BLANK) {
if (lastType == BLANK || lastType == NONE) {
continue; // 折叠连续空行 / 去掉前导空行
}
res.add(cur);
lastType = BLANK;
continue;
}
if (lastType != NONE && lastType != BLANK && needsBlankBetween(lastType, curType)) {
res.add("");
}
res.add(cur);
lastType = curType;
}
return res;
}
private static boolean needsBlankBetween(int prev, int cur) {
if (cur == HEADING || prev == HEADING) {
return true;
}
if (cur == TABLE && prev != TABLE) {
return true;
}
return prev == TABLE && cur != TABLE;
}
private static int classify(String line, boolean tableFlag) {
if (line.isBlank()) {
return BLANK;
}
if (tableFlag) {
return TABLE;
}
if (HEADING_LINE.matcher(line).matches()) {
return HEADING;
}
return OTHER;
}
}

View File

@ -0,0 +1,184 @@
package vip.mate.common.text;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
/**
* {@link MarkdownNormalizer} 单元测试
*/
class MarkdownNormalizerTest {
@Test
@DisplayName("null / 空串原样返回")
void nullAndEmpty() {
assertNull(MarkdownNormalizer.normalize(null));
assertEquals("", MarkdownNormalizer.normalize(""));
}
@Test
@DisplayName("ATX 标题缺空格补空格")
void headingMissingSpace() {
assertEquals("## 二、美股", MarkdownNormalizer.normalize("##二、美股"));
assertEquals("### 已完成 ✅", MarkdownNormalizer.normalize("###已完成 ✅"));
}
@Test
@DisplayName("已规范标题保持不变")
void compliantHeadingUnchanged() {
assertEquals("## 一、核心结论", MarkdownNormalizer.normalize("## 一、核心结论"));
}
@Test
@DisplayName("数字开头的 #5 / #1 视为引用,不补空格")
void headingDigitGuard() {
assertEquals("#5 bolt", MarkdownNormalizer.normalize("#5 bolt"));
assertEquals("#1 优先级", MarkdownNormalizer.normalize("#1 优先级"));
}
@Test
@DisplayName("--- 与后续内容粘连时拆行")
void thematicBreakGlued() {
assertEquals("---\n\n# 全球", MarkdownNormalizer.normalize("---#全球"));
assertEquals("---\n\n## 二、美股", MarkdownNormalizer.normalize("---##二、美股"));
}
@Test
@DisplayName("--- 粘连在行内容之后mid-line且后接标题时拆行")
void thematicBreakGluedMidLine() {
assertEquals(
"*来源:雪球 · 2026-06-01*\n\n---\n\n### 二、供应链与产能",
MarkdownNormalizer.normalize("*来源:雪球 · 2026-06-01*---### 二、供应链与产能"));
}
@Test
@DisplayName("行内容 + --- + 标题 + 表格四重粘连全部拆开")
void midLineHrHeadingTableChain() {
String input = "- 数据中心收入逾 **90%**---### 综合判断🔍| 维度 |信号 | 评级|\n"
+ "|------|------|\n"
+ "| 产品 | 强 |";
String out = MarkdownNormalizer.normalize(input);
assertTrue(out.contains("- 数据中心收入逾 **90%**"), "前缀正文应保留");
assertTrue(out.contains("\n---\n"), "--- 应独占一行");
assertTrue(out.contains("### 综合判断🔍"), "标题应从表格拆出");
assertTrue(out.contains("| 维度 | 信号 | 评级 |"), "表头应对齐");
assertFalse(out.contains("**90%**---"), "--- 不应再粘连前缀");
assertFalse(out.contains("🔍| 维度"), "标题不应再粘连表格");
}
@Test
@DisplayName("散文中的 em-dash 风格 --- 不被误拆(无后接标题)")
void midLineHrWithoutHeadingUntouched() {
String input = "他停顿了一下---然后继续说。";
assertEquals(input, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("表格单元格与分隔行对齐")
void tableCellAndSeparator() {
String input = "|指数 |涨跌| 解读 |\n"
+ "|---| --- | --- |\n"
+ "| 道琼斯 | +1.73% | 强势 |";
String expected = "| 指数 | 涨跌 | 解读 |\n"
+ "| --- | --- | --- |\n"
+ "| 道琼斯 | +1.73% | 强势 |";
assertEquals(expected, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("分隔行保留对齐冒号")
void separatorAlignmentColons() {
String input = "| a | b | c |\n"
+ "|:--|:-:|--:|\n"
+ "| 1 | 2 | 3 |";
String expected = "| a | b | c |\n"
+ "| :--- | :---: | ---: |\n"
+ "| 1 | 2 | 3 |";
assertEquals(expected, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("标题与表格粘连时拆行")
void headingGluedToTable() {
String input = "## 五、大宗商品:回调| 商品 |最新价 |涨跌 |\n"
+ "| --- | --- | ---|";
String expected = "## 五、大宗商品:回调\n"
+ "\n"
+ "| 商品 | 最新价 | 涨跌 |\n"
+ "| --- | --- | --- |";
assertEquals(expected, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("标题与表格之间补空行")
void blankLineBetweenHeadingAndTable() {
String input = "## 表格\n"
+ "| a | b |\n"
+ "| --- | --- |\n"
+ "| 1 | 2 |";
String expected = "## 表格\n"
+ "\n"
+ "| a | b |\n"
+ "| --- | --- |\n"
+ "| 1 | 2 |";
assertEquals(expected, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("代码块内部原样保留,不被规范化")
void codeFenceProtected() {
String input = "```python\n"
+ "##notheading\n"
+ "x = a|b|c\n"
+ "---glued\n"
+ "```";
assertEquals(input, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("散文中的散落管道符不被当作表格")
void prosePipesUntouched() {
String input = "这是 a | b | c 的一句话。\n另一行普通文本。";
assertEquals(input, MarkdownNormalizer.normalize(input));
}
@Test
@DisplayName("幂等:规范化两次结果一致")
void idempotent() {
String sample = "---#全球资产行情整体分析\n"
+ "##二、美股\n"
+ "|指数 |涨跌| 解读 |\n"
+ "|---| --- | --- |\n"
+ "| 道琼斯 | +1.73% | 强势 |\n"
+ "## 五、大宗商品:回调| 商品 |最新价 |\n"
+ "| --- | --- |\n"
+ "```\n"
+ "##code\n"
+ "|x|y|\n"
+ "```";
String once = MarkdownNormalizer.normalize(sample);
String twice = MarkdownNormalizer.normalize(once);
assertEquals(once, twice);
}
@Test
@DisplayName("综合样本:关键缺陷被修复")
void realWorldSampleProperties() {
String sample = "---#全球资产行情整体分析\n"
+ "---##二、美股:道指强、纳指弱\n"
+ "|指数 |涨跌| 解读 |\n"
+ "|---| --- | --- |\n"
+ "| 道琼斯 | +1.73% | 强势 |";
String out = MarkdownNormalizer.normalize(sample);
assertFalse(out.contains("---#"), "--- 不应再与标题粘连");
assertTrue(out.contains("# 全球资产行情整体分析"), "一级标题应补空格");
assertTrue(out.contains("## 二、美股"), "二级标题应补空格");
assertTrue(out.contains("| 指数 | 涨跌 | 解读 |"), "表头应对齐");
assertTrue(out.contains("| --- | --- | --- |"), "分隔行应规范");
}
}