From 4fb6a83eb605cb687f8a16d907b49cd1f8bcba9e Mon Sep 17 00:00:00 2001 From: matevip Date: Tue, 14 Apr 2026 17:39:50 +0800 Subject: [PATCH] =?UTF-8?q?feat(wiki):=20two-phase=20digest=20=E2=80=94=20?= =?UTF-8?q?route=20+=20per-page=20merge=20(RFC-012=20M2)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../java/vip/mate/wiki/WikiProperties.java | 4 +- .../wiki/service/WikiProcessingService.java | 139 +++++++++++++++++- .../prompts/wiki/merge-page-system.txt | 47 ++++++ .../prompts/wiki/merge-page-user.txt | 25 ++++ .../resources/prompts/wiki/route-system.txt | 52 +++++++ .../resources/prompts/wiki/route-user.txt | 21 +++ 6 files changed, 283 insertions(+), 5 deletions(-) create mode 100644 mateclaw-server/src/main/resources/prompts/wiki/merge-page-system.txt create mode 100644 mateclaw-server/src/main/resources/prompts/wiki/merge-page-user.txt create mode 100644 mateclaw-server/src/main/resources/prompts/wiki/route-system.txt create mode 100644 mateclaw-server/src/main/resources/prompts/wiki/route-user.txt diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/WikiProperties.java b/mateclaw-server/src/main/java/vip/mate/wiki/WikiProperties.java index 3ce58c20..1a54c25c 100644 --- a/mateclaw-server/src/main/java/vip/mate/wiki/WikiProperties.java +++ b/mateclaw-server/src/main/java/vip/mate/wiki/WikiProperties.java @@ -78,7 +78,7 @@ public class WikiProperties { * 是否启用两阶段消化(路由 → 逐页 merge)。 *

* RFC-012 M2:true 时单 chunk 的 LLM 输出量大幅缩减,避免 nginx 60s 网关超时。 - * 默认 false 保持向后兼容;M2 实现完成后切到 true。 + * 默认 true(M2 上线);遇问题可在 application.yml 配 mate.wiki.use-two-phase-digest=false 回退到旧行为。 */ - private boolean useTwoPhaseDigest = false; + private boolean useTwoPhaseDigest = true; } diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiProcessingService.java b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiProcessingService.java index f97f5b89..658d5ae9 100644 --- a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiProcessingService.java +++ b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiProcessingService.java @@ -326,8 +326,12 @@ public class WikiProcessingService { */ private int processChunk(WikiKnowledgeBaseEntity kb, WikiRawMaterialEntity raw, String textContent, String existingPagesIndex) { + // RFC-012 M2:两阶段消化(路由 → 逐页 merge),单次 LLM 调用输出量大幅缩减,避免 nginx 60s 网关超时 + if (properties.isUseTwoPhaseDigest()) { + return processChunkTwoPhase(kb, raw, textContent, existingPagesIndex); + } - // 加载 prompt 模板 + // 旧路径:单次调用让 LLM 同时处理新建 + 全量 merge(输出爆炸,易触发 504) String systemPrompt = PromptLoader.loadPrompt("wiki/digest-system"); String userTemplate = PromptLoader.loadPrompt("wiki/digest-user"); @@ -337,17 +341,146 @@ public class WikiProcessingService { .replace("{raw_title}", raw.getTitle()) .replace("{raw_content}", textContent); - // 调用 LLM(带无限重试,仅在模型不可用时终止) Prompt prompt = new Prompt(List.of( new SystemMessage(systemPrompt), new UserMessage(userPrompt) )); String llmResponse = callLlmWithResilientRetry(prompt, "chunk of raw=" + raw.getId()); - // 解析并持久化页面 return applyLlmResponse(kb.getId(), raw.getId(), llmResponse); } + /** + * RFC-012 M2 两阶段消化: + *

+ * 阶段 A(route):一次 LLM 调用决定要 create 哪些新页 + 要 update 哪些已有页(仅 slug 列表)。 + * 输入小、输出短,单次稳定在 30s 内返回。 + *

+ * 阶段 B(merge):对 update 列表里的每个 slug 单独发 LLM 调用,输入只塞这一页的现有正文 + 当前 + * chunk 文本,输出该页 merge 后的完整内容。每次调用单页规模,远不会触发 nginx 60s 超时。 + *

+ * 新建页直接落库;merge 页因互不依赖,可在当前 chunk 的 virtual thread 内顺序处理(chunk 之间 + * 已通过 maxParallelChunks Semaphore 拿到了并行度)。 + * + * @return 创建+更新的页面数 + */ + private int processChunkTwoPhase(WikiKnowledgeBaseEntity kb, WikiRawMaterialEntity raw, + String textContent, String existingPagesIndex) { + Long kbId = kb.getId(); + Long rawId = raw.getId(); + String configContent = kb.getConfigContent() != null ? kb.getConfigContent() : ""; + String rawTitle = raw.getTitle(); + + // ─── 阶段 A:路由 ─── + String routeSystem = PromptLoader.loadPrompt("wiki/route-system"); + String routeUserTemplate = PromptLoader.loadPrompt("wiki/route-user"); + String routeUser = routeUserTemplate + .replace("{config}", configContent) + .replace("{existing_pages}", existingPagesIndex) + .replace("{raw_title}", rawTitle) + .replace("{raw_content}", textContent); + Prompt routePrompt = new Prompt(List.of( + new SystemMessage(routeSystem), + new UserMessage(routeUser) + )); + String routeResponse = callLlmWithResilientRetry(routePrompt, "route chunk of raw=" + rawId); + JsonNode routeJson = parseJsonResponse(routeResponse); + if (routeJson == null) { + log.warn("[Wiki] Route phase: failed to parse JSON for kbId={}, rawId={}, responseLen={}, first200={}", + kbId, rawId, routeResponse != null ? routeResponse.length() : 0, + routeResponse != null ? routeResponse.substring(0, Math.min(200, routeResponse.length())) : "null"); + return 0; + } + + String sourceRawIds = "[" + rawId + "]"; + int created = 0; + int updated = 0; + + // 应用 create 列表(直接落库,无需第二轮 LLM) + JsonNode createNode = routeJson.path("create"); + if (createNode.isArray()) { + for (JsonNode pageNode : createNode) { + String slug = pageNode.path("slug").asText(""); + String title = pageNode.path("title").asText(""); + String content = pageNode.path("content").asText(""); + String summary = pageNode.path("summary").asText(""); + if (slug.isBlank() || title.isBlank()) continue; + WikiPageEntity existing = pageService.getBySlug(kbId, slug); + if (existing != null) { + // LLM 误把已存在 slug 放进 create —— 走 update 路径兜底 + pageService.updatePageByAi(kbId, slug, content, summary, rawId); + updated++; + } else { + pageService.createPage(kbId, slug, title, content, summary, sourceRawIds); + created++; + } + } + } + + // 收集 update 列表(仅 slug) + List updateSlugs = new ArrayList<>(); + JsonNode updateNode = routeJson.path("update"); + if (updateNode.isArray()) { + for (JsonNode slugNode : updateNode) { + String slug = slugNode.asText(""); + if (!slug.isBlank()) updateSlugs.add(slug); + } + } + log.info("[Wiki] Route phase: kbId={}, rawId={}, planned create={}, planned update={}", + kbId, rawId, createNode.isArray() ? createNode.size() : 0, updateSlugs.size()); + + // ─── 阶段 B:逐页 merge ─── + if (!updateSlugs.isEmpty()) { + String mergeSystem = PromptLoader.loadPrompt("wiki/merge-page-system"); + String mergeUserTemplate = PromptLoader.loadPrompt("wiki/merge-page-user"); + for (String slug : updateSlugs) { + WikiPageEntity existing = pageService.getBySlug(kbId, slug); + if (existing == null) { + log.warn("[Wiki] Merge phase: slug '{}' planned for update but not found in DB, skipping", slug); + continue; + } + String mergeUser = mergeUserTemplate + .replace("{config}", configContent) + .replace("{page_slug}", existing.getSlug() != null ? existing.getSlug() : slug) + .replace("{page_title}", existing.getTitle() != null ? existing.getTitle() : "") + .replace("{page_last_updated_by}", existing.getLastUpdatedBy() != null ? existing.getLastUpdatedBy() : "ai") + .replace("{page_content}", existing.getContent() != null ? existing.getContent() : "") + .replace("{raw_title}", rawTitle) + .replace("{raw_content}", textContent); + Prompt mergePrompt = new Prompt(List.of( + new SystemMessage(mergeSystem), + new UserMessage(mergeUser) + )); + String mergeResponse; + try { + mergeResponse = callLlmWithResilientRetry(mergePrompt, + "merge page slug=" + slug + " of raw=" + rawId); + } catch (RuntimeException e) { + // 单页 merge 失败不影响其他页面,整 chunk 继续 + log.warn("[Wiki] Merge phase: slug '{}' failed: {}", slug, e.getMessage()); + continue; + } + JsonNode mergeJson = parseJsonResponse(mergeResponse); + if (mergeJson == null) { + log.warn("[Wiki] Merge phase: slug '{}' returned unparseable JSON, skipping", slug); + continue; + } + String content = mergeJson.path("content").asText(""); + String summary = mergeJson.path("summary").asText(""); + if (content.isBlank()) { + log.warn("[Wiki] Merge phase: slug '{}' returned blank content, skipping", slug); + continue; + } + pageService.updatePageByAi(kbId, slug, content, summary, rawId); + updated++; + } + } + + log.info("[Wiki] Two-phase digest applied: kbId={}, rawId={}, created={}, updated={}", + kbId, rawId, created, updated); + return created + updated; + } + /** * 解析 LLM 响应并创建/更新 Wiki 页面 * diff --git a/mateclaw-server/src/main/resources/prompts/wiki/merge-page-system.txt b/mateclaw-server/src/main/resources/prompts/wiki/merge-page-system.txt new file mode 100644 index 00000000..e1a0641e --- /dev/null +++ b/mateclaw-server/src/main/resources/prompts/wiki/merge-page-system.txt @@ -0,0 +1,47 @@ +你是一个知识库 Wiki 单页合并助手。你的唯一任务是:把一段新材料合并进**一个已有的 Wiki 页面**,输出该页面更新后的完整内容。 + +## 你做什么 + +- 读懂现有页面的内容结构 +- 从新材料里抽取与该页面主题相关的信息 +- **合并**新旧信息(不是简单替换、不是追加),保留双向链接 +- 输出**这一个页面**更新后的**完整 markdown** + +## 你不做什么 + +- **不要创建其他页面** —— 你只负责一个页面 +- **不要输出 diff 或增量** —— 必须输出完整内容 +- **不要输出 markdown 代码块包裹** —— 直接输出 JSON +- 不要删除已有页面中仍然有效的信息 + +## 合并规则 + +- 新旧信息有冲突 → 在新内容中明确标注矛盾点 +- 新材料对该页面无新增信息 → 输出原 content 即可(保持不变) +- 已有页面 lastUpdatedBy=manual → 仍然合并,但优先保留手动编辑的措辞和结构,仅追加新事实 + +## 链接 + +- 沿用已有页面里的 [[页面标题]] 双向链接 +- 如果新材料引出了对其他已知概念的引用,新增 [[…]] 链接 + +## 语言 + +- 跟随原始材料和已有页面的语言 + +## 输出格式 + +严格输出 JSON,不要 markdown 代码块: + +{ + "slug": "existing-slug", + "title": "页面标题(可微调)", + "content": "## 标题\n\n摘要...\n\n### 章节...\n\n参见:[[相关]]", + "summary": "更新后的一段话摘要" +} + +字段说明: +- `slug`:保持与输入一致 +- `title`:通常保持不变;只有当新材料明确改变了页面主题时才改 +- `content`:完整 markdown,包含原内容中仍然有效的部分 + 来自新材料的新增信息 +- `summary`:更新后的一段话摘要 diff --git a/mateclaw-server/src/main/resources/prompts/wiki/merge-page-user.txt b/mateclaw-server/src/main/resources/prompts/wiki/merge-page-user.txt new file mode 100644 index 00000000..061f6455 --- /dev/null +++ b/mateclaw-server/src/main/resources/prompts/wiki/merge-page-user.txt @@ -0,0 +1,25 @@ +## 知识库处理规则 + +{config} + +## 待合并的现有 Wiki 页面 + +slug:`{page_slug}` +标题:{page_title} +最近编辑者:{page_last_updated_by} + +### 现有内容 + +{page_content} + +## 新材料 + +标题:{raw_title} + +{raw_content} + +--- + +请把新材料中与本页面主题相关的信息合并进**这一个页面**,输出完整的 JSON(按 system 规则)。 +- 如果新材料对本页面无新增信息:原 content 原样返回即可。 +- 不要输出其他页面,不要 diff,不要追加。 diff --git a/mateclaw-server/src/main/resources/prompts/wiki/route-system.txt b/mateclaw-server/src/main/resources/prompts/wiki/route-system.txt new file mode 100644 index 00000000..8aaaec4a --- /dev/null +++ b/mateclaw-server/src/main/resources/prompts/wiki/route-system.txt @@ -0,0 +1,52 @@ +你是一个知识库 Wiki 路由助手。你的唯一任务是:阅读一段原始材料,决定**哪些新页面需要创建**、**哪些已有页面需要合并更新**。 + +## 你做什么 + +1. **阅读原始材料**,识别其中包含的概念、实体、流程 +2. **比对已有页面索引**(仅含 slug + title + summary,不含正文): + - 材料中出现的概念**已经被某个已有页面充分覆盖** → 放入 `update` 列表(仅写 slug,**不要**写正文) + - 材料中出现的概念**没有任何已有页面覆盖** → 放入 `create` 列表,**给出完整页面正文** +3. 不创建浅薄页面(< 3 句实质内容的概念跳过,不进任何列表) + +## 你不做什么 + +- **不要重写或合并已有页面的内容** —— update 列表只写 slug,正文合并由后续步骤完成 +- **不要为已存在概念在 create 中复制一份** +- 不要输出 markdown 代码块包裹 + +## 创建新页面(仅 `create` 列表使用)的格式规则 + +- 使用 Markdown 标题(## / ###)组织 +- 使用 [[页面标题]] 双向链接到其他页面(无论是新页面还是已有页面) +- slug 用小写字母 + 连字符 +- summary 一段话简短摘要 +- content 完整 markdown 正文,开头先一段摘要 + +## 语言 + +- 跟随原始材料的语言(中文材料 → 中文输出) +- 术语保持一致 + +## 输出格式 + +严格输出 JSON,不要包含 markdown 代码块: + +{ + "create": [ + { + "slug": "concept-name", + "title": "概念名称", + "content": "## 概念名称\n\n摘要段落...\n\n### 详细内容\n...\n\n参见:[[相关主题]]", + "summary": "一段话摘要" + } + ], + "update": [ + "existing-slug-1", + "existing-slug-2" + ] +} + +字段说明: +- `create`:完全新增的页面,给出完整内容 +- `update`:已存在但需要根据新材料合并更新的页面,**只列 slug 字符串数组** +- 两个数组都可以为空(材料完全无价值时全空,材料只更新已有页时 create 为空) diff --git a/mateclaw-server/src/main/resources/prompts/wiki/route-user.txt b/mateclaw-server/src/main/resources/prompts/wiki/route-user.txt new file mode 100644 index 00000000..b2b119e8 --- /dev/null +++ b/mateclaw-server/src/main/resources/prompts/wiki/route-user.txt @@ -0,0 +1,21 @@ +## 知识库处理规则 + +{config} + +## 已有 Wiki 页面索引(slug + 摘要) + +{existing_pages} + +## 待消化的原始材料 + +标题:{raw_title} + +{raw_content} + +--- + +请按 system 中规定的 JSON 格式输出: +- `create`:完全没有对应页面的新概念(含完整内容) +- `update`:已有页面但需根据本材料合并更新的(**仅 slug 列表,不要正文**) + +不要试图自己合并 update 列表里页面的内容 —— 那是下一阶段独立完成的工作。