From 398d7a2d801146ba38f407a4cddb5ad4d318ecd5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=80=AA=E7=A8=8B=E4=BC=9F?= Date: Sun, 7 Jun 2026 19:10:53 +0800 Subject: [PATCH] feat(agent): deterministic Markdown normalization for final answers (#275) LLMs routinely emit malformed Markdown (missing heading spaces, glued `---`, unaligned table pipes) that prompt rules cannot reliably prevent. Add a zero-token, regex-only MarkdownNormalizer applied on the FinalAnswerNode convergence path before persistence / channel delivery. It is code-fence aware, idempotent, and conservative (em-dash `---`, `#5`-style refs, stray prose pipes are left untouched). RETURN_DIRECT verbatim output and approval-wait paths return earlier and are unaffected. Closes #274 --- .../agent/graph/node/FinalAnswerNode.java | 9 + .../mate/common/text/MarkdownNormalizer.java | 293 ++++++++++++++++++ .../common/text/MarkdownNormalizerTest.java | 184 +++++++++++ 3 files changed, 486 insertions(+) create mode 100644 mateclaw-server/src/main/java/vip/mate/common/text/MarkdownNormalizer.java create mode 100644 mateclaw-server/src/test/java/vip/mate/common/text/MarkdownNormalizerTest.java diff --git a/mateclaw-server/src/main/java/vip/mate/agent/graph/node/FinalAnswerNode.java b/mateclaw-server/src/main/java/vip/mate/agent/graph/node/FinalAnswerNode.java index 6792e0a6..ea478714 100644 --- a/mateclaw-server/src/main/java/vip/mate/agent/graph/node/FinalAnswerNode.java +++ b/mateclaw-server/src/main/java/vip/mate/agent/graph/node/FinalAnswerNode.java @@ -8,6 +8,7 @@ import vip.mate.agent.graph.state.DirectToolOutput; import vip.mate.agent.graph.state.FinishReason; import vip.mate.agent.graph.state.MateClawStateAccessor; import vip.mate.agent.graph.state.SourceEvidenceLedger; +import vip.mate.common.text.MarkdownNormalizer; import vip.mate.tool.document.GeneratedFileCache; import java.util.List; @@ -170,6 +171,14 @@ public class FinalAnswerNode implements NodeAction { validation.unsupportedReferences()); } + // Deterministic Markdown cleanup on the model-generated answer body. LLMs + // routinely emit malformed Markdown (missing heading spaces, glued `---`, + // unaligned table pipes) that prompt rules fail to prevent; this fixes the + // mechanical defects before the answer is persisted / sent to channels. + // Verbatim tool output (RETURN_DIRECT) and approval-wait paths return early + // above and are intentionally left untouched. + finalAnswer = MarkdownNormalizer.normalize(finalAnswer); + // Build the event list. Always carries the finish_reason event so // downstream consumers (memory gate, channel accumulator, message // metadata persistence) see a machine-readable status. When the diff --git a/mateclaw-server/src/main/java/vip/mate/common/text/MarkdownNormalizer.java b/mateclaw-server/src/main/java/vip/mate/common/text/MarkdownNormalizer.java new file mode 100644 index 00000000..c643a80c --- /dev/null +++ b/mateclaw-server/src/main/java/vip/mate/common/text/MarkdownNormalizer.java @@ -0,0 +1,293 @@ +package vip.mate.common.text; + +import java.util.ArrayList; +import java.util.List; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +/** + * 确定性 Markdown 规范化工具。 + * + *

用于在 Agent 最终答案落库 / 发渠道前,修复 LLM 原始输出中常见的机械排版缺陷。纯本地正则处理, + * 不调用任何模型(零 token)。设计目标是「修畸形而不改语义」,因此遵循以下原则:

+ * + * + * + *

覆盖的修复:ATX 标题补空格、{@code ---} 与后续内容粘连时拆行(含行首与 mid-line 后接标题两种)、 + * 表格块单元格与分隔行对齐、标题与表格粘连时拆行、标题/表格块边界补空行。

+ * + *

不在范围内(属语义判断,正则无法安全自动化,保留在提示词约束):Emoji 位置、代码块语言标注补全。

+ */ +public final class MarkdownNormalizer { + + private MarkdownNormalizer() {} + + /** 行首 ATX 标题但紧跟非空格、非 # 字符(缺少标题空格)。 */ + private static final Pattern HEADING_NO_SPACE = Pattern.compile("^(#{1,6})([^#\\s].*)$"); + + /** 已规范的标题行:#{1,6} + 空白。用于块边界判断。 */ + private static final Pattern HEADING_LINE = Pattern.compile("^#{1,6}\\s.*$"); + + /** 主题分隔线与后续内容粘连:行首 3+ 短横,后面紧跟非短横的可见内容。 */ + private static final Pattern HR_GLUED = Pattern.compile("^(-{3,})([^-\\s].*)$"); + + /** + * 主题分隔线粘连在「行内容之后」(mid-line),且后面紧跟一个 ATX 标题。 + * 形如 {@code *来源…2026-06-01*---### 二、…} 或 {@code **90%**---### 综合判断}。 + * 仅在 {@code ---} 后紧跟 {@code #} 标题时才拆,避免误伤散文里 em-dash 风格的 {@code ---}。 + */ + private static final Pattern HR_MID_HEADING = + Pattern.compile("^(.*?\\S)\\s*(-{3,})\\s*(#{1,6}.*)$"); + + /** 标题行尾粘连了表格:#{1,6} 标题文字(不含管道符)+ 管道符起始的尾部。 */ + private static final Pattern HEADING_TABLE = Pattern.compile("^(#{1,6}[^|\\n]*?)\\s*(\\|.+)$"); + + /** GFM 表格分隔行:由短横/冒号组成的单元格,用管道符分隔(必须同时含 - 与 |)。 */ + private static final Pattern SEPARATOR_ROW = + Pattern.compile("^\\s*\\|?\\s*:?-{1,}:?\\s*(\\|\\s*:?-{1,}:?\\s*)*\\|?\\s*$"); + + /** + * 规范化 Markdown 文本。{@code null} 或空串原样返回。 + */ + public static String normalize(String md) { + if (md == null || md.isEmpty()) { + return md; + } + String normalized = md.replace("\r\n", "\n").replace("\r", "\n"); + String[] lines = normalized.split("\n", -1); + + List out = new ArrayList<>(); + List textBuf = new ArrayList<>(); + boolean inFence = false; + String fenceMarker = null; + + for (String line : lines) { + String lead = line.stripLeading(); + if (!inFence && (lead.startsWith("```") || lead.startsWith("~~~"))) { + flushText(textBuf, out); + textBuf.clear(); + inFence = true; + fenceMarker = lead.startsWith("```") ? "```" : "~~~"; + out.add(line); + } else if (inFence && lead.startsWith(fenceMarker)) { + inFence = false; + fenceMarker = null; + out.add(line); + } else if (inFence) { + out.add(line); + } else { + textBuf.add(line); + } + } + flushText(textBuf, out); + + String result = String.join("\n", out); + // 仅清理文档首尾多余空行,不触碰代码块内部 + return result.replaceAll("^\\n+", "").replaceAll("\\n+$", ""); + } + + // ==================== 非代码段处理 ==================== + + private static void flushText(List textLines, List out) { + if (textLines.isEmpty()) { + return; + } + // 1. 行级展开:HR 粘连拆行、标题粘表格拆行、标题补空格 + List expanded = new ArrayList<>(); + for (String l : textLines) { + expanded.addAll(expandLine(l)); + } + // 2. 表格块识别与规范化 + List normalizedLines = new ArrayList<>(); + List isTable = new ArrayList<>(); + normalizeTables(expanded, normalizedLines, isTable); + // 3. 标题/表格块边界补空行 + 折叠多余空行 + out.addAll(insertBoundaryBlanks(normalizedLines, isTable)); + } + + private static List expandLine(String line) { + List result = new ArrayList<>(); + + Matcher hrMid = HR_MID_HEADING.matcher(line); + if (hrMid.matches()) { + result.addAll(expandLine(hrMid.group(1))); + result.add(""); + result.add("---"); + result.add(""); + result.addAll(expandLine(hrMid.group(3))); + return result; + } + + Matcher hr = HR_GLUED.matcher(line); + if (hr.matches()) { + result.add("---"); + result.add(""); + result.addAll(expandLine(hr.group(2))); + return result; + } + + Matcher ht = HEADING_TABLE.matcher(line); + if (ht.matches() && countPipes(ht.group(2)) >= 2) { + result.add(fixHeadingSpace(ht.group(1).strip())); + result.add(""); + result.add(ht.group(2).strip()); + return result; + } + + result.add(fixHeadingSpace(line)); + return result; + } + + /** + * 行首 ATX 标题缺空格时补一个空格。为避免误伤 {@code #5}、{@code #1} 这类引用, + * 紧跟数字的不处理。 + */ + private static String fixHeadingSpace(String line) { + Matcher m = HEADING_NO_SPACE.matcher(line); + if (m.matches()) { + String hashes = m.group(1); + String rest = m.group(2); + if (!Character.isDigit(rest.charAt(0))) { + return hashes + " " + rest; + } + } + return line; + } + + // ==================== 表格规范化 ==================== + + private static void normalizeTables(List lines, List out, List isTable) { + int n = lines.size(); + boolean[] tbl = new boolean[n]; + for (int i = 0; i < n; i++) { + if (isSeparatorRow(lines.get(i)) && i > 0 && containsPipe(lines.get(i - 1))) { + int start = i - 1; + int end = i; + int j = i + 1; + while (j < n && !lines.get(j).isBlank() && containsPipe(lines.get(j)) + && !HEADING_LINE.matcher(lines.get(j)).matches()) { + end = j; + j++; + } + for (int k = start; k <= end; k++) { + tbl[k] = true; + } + i = end; + } + } + for (int i = 0; i < n; i++) { + if (tbl[i]) { + out.add(normalizeTableRow(lines.get(i), isSeparatorRow(lines.get(i)))); + } else { + out.add(lines.get(i)); + } + isTable.add(tbl[i]); + } + } + + private static String normalizeTableRow(String line, boolean separator) { + String s = line.strip(); + if (s.startsWith("|")) { + s = s.substring(1); + } + if (s.endsWith("|")) { + s = s.substring(0, s.length() - 1); + } + String[] cells = s.split("(?= 0 && line.indexOf('|') >= 0 + && SEPARATOR_ROW.matcher(line).matches(); + } + + private static boolean containsPipe(String line) { + return line.indexOf('|') >= 0; + } + + private static int countPipes(String s) { + int count = 0; + for (int i = 0; i < s.length(); i++) { + if (s.charAt(i) == '|') { + count++; + } + } + return count; + } + + // ==================== 块边界空行 ==================== + + private static final int BLANK = 0; + private static final int HEADING = 1; + private static final int TABLE = 2; + private static final int OTHER = 3; + private static final int NONE = -1; + + private static List insertBoundaryBlanks(List lines, List isTable) { + List res = new ArrayList<>(); + int lastType = NONE; + for (int i = 0; i < lines.size(); i++) { + String cur = lines.get(i); + int curType = classify(cur, isTable.get(i)); + + if (curType == BLANK) { + if (lastType == BLANK || lastType == NONE) { + continue; // 折叠连续空行 / 去掉前导空行 + } + res.add(cur); + lastType = BLANK; + continue; + } + + if (lastType != NONE && lastType != BLANK && needsBlankBetween(lastType, curType)) { + res.add(""); + } + res.add(cur); + lastType = curType; + } + return res; + } + + private static boolean needsBlankBetween(int prev, int cur) { + if (cur == HEADING || prev == HEADING) { + return true; + } + if (cur == TABLE && prev != TABLE) { + return true; + } + return prev == TABLE && cur != TABLE; + } + + private static int classify(String line, boolean tableFlag) { + if (line.isBlank()) { + return BLANK; + } + if (tableFlag) { + return TABLE; + } + if (HEADING_LINE.matcher(line).matches()) { + return HEADING; + } + return OTHER; + } +} diff --git a/mateclaw-server/src/test/java/vip/mate/common/text/MarkdownNormalizerTest.java b/mateclaw-server/src/test/java/vip/mate/common/text/MarkdownNormalizerTest.java new file mode 100644 index 00000000..0b85ba0b --- /dev/null +++ b/mateclaw-server/src/test/java/vip/mate/common/text/MarkdownNormalizerTest.java @@ -0,0 +1,184 @@ +package vip.mate.common.text; + +import org.junit.jupiter.api.DisplayName; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * {@link MarkdownNormalizer} 单元测试。 + */ +class MarkdownNormalizerTest { + + @Test + @DisplayName("null / 空串原样返回") + void nullAndEmpty() { + assertNull(MarkdownNormalizer.normalize(null)); + assertEquals("", MarkdownNormalizer.normalize("")); + } + + @Test + @DisplayName("ATX 标题缺空格补空格") + void headingMissingSpace() { + assertEquals("## 二、美股", MarkdownNormalizer.normalize("##二、美股")); + assertEquals("### 已完成 ✅", MarkdownNormalizer.normalize("###已完成 ✅")); + } + + @Test + @DisplayName("已规范标题保持不变") + void compliantHeadingUnchanged() { + assertEquals("## 一、核心结论", MarkdownNormalizer.normalize("## 一、核心结论")); + } + + @Test + @DisplayName("数字开头的 #5 / #1 视为引用,不补空格") + void headingDigitGuard() { + assertEquals("#5 bolt", MarkdownNormalizer.normalize("#5 bolt")); + assertEquals("#1 优先级", MarkdownNormalizer.normalize("#1 优先级")); + } + + @Test + @DisplayName("--- 与后续内容粘连时拆行") + void thematicBreakGlued() { + assertEquals("---\n\n# 全球", MarkdownNormalizer.normalize("---#全球")); + assertEquals("---\n\n## 二、美股", MarkdownNormalizer.normalize("---##二、美股")); + } + + @Test + @DisplayName("--- 粘连在行内容之后(mid-line)且后接标题时拆行") + void thematicBreakGluedMidLine() { + assertEquals( + "*来源:雪球 · 2026-06-01*\n\n---\n\n### 二、供应链与产能", + MarkdownNormalizer.normalize("*来源:雪球 · 2026-06-01*---### 二、供应链与产能")); + } + + @Test + @DisplayName("行内容 + --- + 标题 + 表格四重粘连全部拆开") + void midLineHrHeadingTableChain() { + String input = "- 数据中心收入逾 **90%**---### 综合判断🔍| 维度 |信号 | 评级|\n" + + "|------|------|\n" + + "| 产品 | 强 |"; + String out = MarkdownNormalizer.normalize(input); + assertTrue(out.contains("- 数据中心收入逾 **90%**"), "前缀正文应保留"); + assertTrue(out.contains("\n---\n"), "--- 应独占一行"); + assertTrue(out.contains("### 综合判断🔍"), "标题应从表格拆出"); + assertTrue(out.contains("| 维度 | 信号 | 评级 |"), "表头应对齐"); + assertFalse(out.contains("**90%**---"), "--- 不应再粘连前缀"); + assertFalse(out.contains("🔍| 维度"), "标题不应再粘连表格"); + } + + @Test + @DisplayName("散文中的 em-dash 风格 --- 不被误拆(无后接标题)") + void midLineHrWithoutHeadingUntouched() { + String input = "他停顿了一下---然后继续说。"; + assertEquals(input, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("表格单元格与分隔行对齐") + void tableCellAndSeparator() { + String input = "|指数 |涨跌| 解读 |\n" + + "|---| --- | --- |\n" + + "| 道琼斯 | +1.73% | 强势 |"; + String expected = "| 指数 | 涨跌 | 解读 |\n" + + "| --- | --- | --- |\n" + + "| 道琼斯 | +1.73% | 强势 |"; + assertEquals(expected, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("分隔行保留对齐冒号") + void separatorAlignmentColons() { + String input = "| a | b | c |\n" + + "|:--|:-:|--:|\n" + + "| 1 | 2 | 3 |"; + String expected = "| a | b | c |\n" + + "| :--- | :---: | ---: |\n" + + "| 1 | 2 | 3 |"; + assertEquals(expected, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("标题与表格粘连时拆行") + void headingGluedToTable() { + String input = "## 五、大宗商品:回调| 商品 |最新价 |涨跌 |\n" + + "| --- | --- | ---|"; + String expected = "## 五、大宗商品:回调\n" + + "\n" + + "| 商品 | 最新价 | 涨跌 |\n" + + "| --- | --- | --- |"; + assertEquals(expected, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("标题与表格之间补空行") + void blankLineBetweenHeadingAndTable() { + String input = "## 表格\n" + + "| a | b |\n" + + "| --- | --- |\n" + + "| 1 | 2 |"; + String expected = "## 表格\n" + + "\n" + + "| a | b |\n" + + "| --- | --- |\n" + + "| 1 | 2 |"; + assertEquals(expected, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("代码块内部原样保留,不被规范化") + void codeFenceProtected() { + String input = "```python\n" + + "##notheading\n" + + "x = a|b|c\n" + + "---glued\n" + + "```"; + assertEquals(input, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("散文中的散落管道符不被当作表格") + void prosePipesUntouched() { + String input = "这是 a | b | c 的一句话。\n另一行普通文本。"; + assertEquals(input, MarkdownNormalizer.normalize(input)); + } + + @Test + @DisplayName("幂等:规范化两次结果一致") + void idempotent() { + String sample = "---#全球资产行情整体分析\n" + + "##二、美股\n" + + "|指数 |涨跌| 解读 |\n" + + "|---| --- | --- |\n" + + "| 道琼斯 | +1.73% | 强势 |\n" + + "## 五、大宗商品:回调| 商品 |最新价 |\n" + + "| --- | --- |\n" + + "```\n" + + "##code\n" + + "|x|y|\n" + + "```"; + String once = MarkdownNormalizer.normalize(sample); + String twice = MarkdownNormalizer.normalize(once); + assertEquals(once, twice); + } + + @Test + @DisplayName("综合样本:关键缺陷被修复") + void realWorldSampleProperties() { + String sample = "---#全球资产行情整体分析\n" + + "---##二、美股:道指强、纳指弱\n" + + "|指数 |涨跌| 解读 |\n" + + "|---| --- | --- |\n" + + "| 道琼斯 | +1.73% | 强势 |"; + String out = MarkdownNormalizer.normalize(sample); + + assertFalse(out.contains("---#"), "--- 不应再与标题粘连"); + assertTrue(out.contains("# 全球资产行情整体分析"), "一级标题应补空格"); + assertTrue(out.contains("## 二、美股"), "二级标题应补空格"); + assertTrue(out.contains("| 指数 | 涨跌 | 解读 |"), "表头应对齐"); + assertTrue(out.contains("| --- | --- | --- |"), "分隔行应规范"); + } +}