From b41496ed46197950c52478a85e66e8607058c774 Mon Sep 17 00:00:00 2001 From: matevip Date: Sat, 25 Apr 2026 09:56:19 +0800 Subject: [PATCH] feat(wiki): PR-4 on-demand compile + multi-page read tools --- .../wiki/service/WikiCitationService.java | 30 +++ .../mate/wiki/service/WikiCompileService.java | 188 ++++++++++++++++++ .../java/vip/mate/wiki/tool/WikiTool.java | 88 ++++++++ 3 files changed, 306 insertions(+) create mode 100644 mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCompileService.java diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCitationService.java b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCitationService.java index d2560f89..49d1426f 100644 --- a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCitationService.java +++ b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCitationService.java @@ -62,6 +62,36 @@ public class WikiCitationService { log.debug("[WikiCitation] Built citations for pageId={}, kbId={}", pageId, kbId); } + /** + * RFC-051 PR-4: rebuild citations from a specific list of evidence chunks + * rather than the union of all chunks for the page's source raws. Used by + * {@code WikiCompileService} so on-demand pages cite only the chunks the + * compile prompt actually saw — keeping relation/citation signals clean. + *

+ * Falls back to the raw-level rebuild if the evidence list is null/empty, + * so callers don't have to special-case "no evidence found". + */ + public void buildCitations(Long pageId, Long kbId, List evidenceChunkIds) { + if (evidenceChunkIds == null || evidenceChunkIds.isEmpty()) { + buildCitations(pageId, kbId); + return; + } + WikiPageEntity page = pageMapper.selectById(pageId); + if (page == null) return; + + citationMapper.softDeleteByPageId(pageId); + for (Long chunkId : evidenceChunkIds) { + WikiPageCitationEntity citation = new WikiPageCitationEntity(); + citation.setPageId(pageId); + citation.setChunkId(chunkId); + citation.setConfidence(BigDecimal.ONE); + citation.setCreatedBy("compile"); + citationMapper.insert(citation); + } + log.info("[WikiCitation] Built {} evidence citations for pageId={}, kbId={}", + evidenceChunkIds.size(), pageId, kbId); + } + private List parseRawIds(String json) { if (json == null || json.isBlank()) return List.of(); try { diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCompileService.java b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCompileService.java new file mode 100644 index 00000000..787545f7 --- /dev/null +++ b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiCompileService.java @@ -0,0 +1,188 @@ +package vip.mate.wiki.service; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import lombok.RequiredArgsConstructor; +import lombok.extern.slf4j.Slf4j; +import org.springframework.ai.chat.messages.SystemMessage; +import org.springframework.ai.chat.messages.UserMessage; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.prompt.Prompt; +import org.springframework.beans.factory.annotation.Autowired; +import org.springframework.stereotype.Service; +import vip.mate.wiki.job.WikiJobStep; +import vip.mate.wiki.job.WikiModelRoutingService; +import vip.mate.wiki.model.WikiChunkEntity; +import vip.mate.wiki.model.WikiPageEntity; +import vip.mate.wiki.repository.WikiChunkMapper; + +import java.util.ArrayList; +import java.util.List; + +/** + * RFC-051 PR-4: on-demand page compilation. + *

+ * Given a topic (or an explicit {@code slug}), search for the most relevant + * chunks via {@link HybridRetriever}, build a short evidence pack from the + * top hits, and ask the LLM for a single Markdown page. The resulting page + * is persisted and its citations are bound to the evidence chunk IDs only — + * not to every chunk of the source raw — so relation signals stay clean. + *

+ * This is the bridge between lazy ingest (where 0 pages is the steady state) + * and the user/agent saying "now produce a page about X". + */ +@Slf4j +@Service +@RequiredArgsConstructor +public class WikiCompileService { + + private final HybridRetriever hybridRetriever; + private final WikiChunkMapper chunkMapper; + private final WikiPageService pageService; + private final WikiCitationService citationService; + private final ObjectMapper objectMapper; + + /** + * Optional. When wired we use the routing chain (stepModels[CREATE_PAGE] + * -> wikiDefaultModelId -> system default); otherwise the caller's + * pre-built {@link ChatModel} factory wins. + */ + @Autowired(required = false) + private WikiModelRoutingService modelRoutingService; + + public record CompileResult(Long pageId, String slug, String title, int evidenceChunkCount, + boolean created) {} + + /** + * Compile or update a single page on the topic. + * + * @param kbId knowledge base id + * @param topic natural-language topic; required + * @param slug optional explicit slug; auto-derived from topic when null/blank + * @param maxEvidenceChunks evidence pack cap (defaults to 8 when null) + * @return result with the persisted page id and how many chunks were cited + */ + public CompileResult compilePage(Long kbId, String topic, String slug, Integer maxEvidenceChunks) { + if (kbId == null) throw new IllegalArgumentException("kbId is required"); + if (topic == null || topic.isBlank()) throw new IllegalArgumentException("topic is required"); + int cap = (maxEvidenceChunks == null || maxEvidenceChunks <= 0) + ? 8 : Math.min(20, maxEvidenceChunks); + + // 1. Retrieve evidence chunks via semantic search (hybrid retriever). + List hits = hybridRetriever.searchChunks(kbId, topic, cap); + if (hits.isEmpty()) { + throw new IllegalStateException("No evidence chunks found for topic: " + topic); + } + + List evidenceChunkIds = new ArrayList<>(hits.size()); + StringBuilder evidenceBlock = new StringBuilder(); + for (int i = 0; i < hits.size(); i++) { + HybridRetriever.ChunkHit hit = hits.get(i); + evidenceChunkIds.add(hit.chunkId()); + WikiChunkEntity chunk = chunkMapper.selectById(hit.chunkId()); + if (chunk == null || chunk.getContent() == null) continue; + evidenceBlock.append("### Evidence ").append(i + 1) + .append(" (chunk=").append(hit.chunkId()).append(")"); + if (hit.headerBreadcrumb() != null && !hit.headerBreadcrumb().isBlank()) { + evidenceBlock.append(" — ").append(hit.headerBreadcrumb()); + } + if (hit.pageNumber() != null) { + evidenceBlock.append(" (page ").append(hit.pageNumber()).append(")"); + } + evidenceBlock.append("\n\n").append(chunk.getContent()).append("\n\n"); + } + + // 2. Resolve slug + title. + String resolvedSlug = (slug == null || slug.isBlank()) ? WikiPageService.toSlug(topic) : slug; + if (resolvedSlug == null || resolvedSlug.isBlank()) { + resolvedSlug = "page-" + System.currentTimeMillis(); + } + WikiPageEntity existing = pageService.getBySlug(kbId, resolvedSlug); + if (existing != null && WikiPageService.isProtected(existing)) { + throw new IllegalStateException("Refusing to compile over protected page: " + resolvedSlug); + } + + // 3. Build LLM prompt. + String system = """ + You are a wiki page compiler. Produce a single Markdown page that + synthesizes the supplied evidence. Constraints: + - Output ONLY a JSON object: {"title": "...", "summary": "...", "content": "..."} + - Do not invent facts beyond the evidence. + - Use [[wikilinks]] when an evidence breadcrumb names a related concept. + - Keep summary <= 300 characters. + - content must be Markdown with ## headers. + """; + String user = "## Topic\n\n" + topic + "\n\n## Evidence\n\n" + evidenceBlock; + + ChatModel chatModel = resolveChatModel(kbId); + ChatResponse resp = chatModel.call(new Prompt(List.of( + new SystemMessage(system), new UserMessage(user)))); + if (resp == null || resp.getResult() == null + || resp.getResult().getOutput() == null + || resp.getResult().getOutput().getText() == null) { + throw new IllegalStateException("LLM returned no compile output"); + } + String body = resp.getResult().getOutput().getText(); + JsonNode parsed = parseJson(body); + if (parsed == null) { + throw new IllegalStateException("LLM compile output was not valid JSON"); + } + String title = parsed.path("title").asText(topic); + String summary = parsed.path("summary").asText(""); + String content = parsed.path("content").asText(""); + if (content.isBlank()) { + throw new IllegalStateException("LLM compile output missing content"); + } + + // 4. Persist (create or update via AI path). + WikiPageEntity persisted; + boolean created; + if (existing == null) { + persisted = pageService.createPage(kbId, resolvedSlug, title, content, summary, null); + created = true; + } else { + persisted = pageService.updatePageByAi(kbId, resolvedSlug, content, summary, null); + created = false; + if (persisted == null) persisted = existing; + } + + // 5. Bind evidence citations only. + try { + citationService.buildCitations(persisted.getId(), kbId, evidenceChunkIds); + } catch (Exception e) { + log.warn("[WikiCompile] Failed to attach evidence citations for pageId={}: {}", + persisted.getId(), e.getMessage()); + } + + log.info("[WikiCompile] {} page slug={} title='{}' from {} evidence chunks (kbId={})", + created ? "Created" : "Updated", resolvedSlug, title, evidenceChunkIds.size(), kbId); + return new CompileResult(persisted.getId(), resolvedSlug, title, evidenceChunkIds.size(), created); + } + + private ChatModel resolveChatModel(Long kbId) { + if (modelRoutingService == null) { + throw new IllegalStateException("ModelRoutingService unavailable; cannot compile"); + } + Long modelId = modelRoutingService.selectModelId(kbId, "compile_page", WikiJobStep.CREATE_PAGE); + return modelRoutingService.buildChatModel(modelId); + } + + private JsonNode parseJson(String text) { + if (text == null) return null; + try { + return objectMapper.readTree(text); + } catch (Exception ignored) { + int s = text.indexOf('{'); + int e = text.lastIndexOf('}'); + if (s >= 0 && e > s) { + try { + return objectMapper.readTree(text.substring(s, e + 1)); + } catch (Exception ignored2) { + return null; + } + } + return null; + } + } +} diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/tool/WikiTool.java b/mateclaw-server/src/main/java/vip/mate/wiki/tool/WikiTool.java index 44621d27..38805edf 100644 --- a/mateclaw-server/src/main/java/vip/mate/wiki/tool/WikiTool.java +++ b/mateclaw-server/src/main/java/vip/mate/wiki/tool/WikiTool.java @@ -55,6 +55,10 @@ public class WikiTool { @Autowired(required = false) private WikiRawMaterialMapper rawMaterialMapper; + /** RFC-051 PR-4: optional on-demand compile. Tool surface skipped when missing. */ + @Autowired(required = false) + private WikiCompileService compileService; + public WikiTool(WikiPageService pageService, WikiKnowledgeBaseService kbService, WikiRawMaterialService rawService, @@ -362,6 +366,90 @@ public class WikiTool { .toString(); } + // ==================== RFC-051 PR-4: on-demand compile + batch read ==================== + + @Tool(description = """ + Compile (or update) a single wiki page about a topic from existing chunks. + Use this AFTER lazy ingest when search has surfaced relevant content but no + page exists yet. The page will cite only the evidence chunks the compile + prompt actually used — not every chunk of the source raw material. + Set slug to control the page slug; otherwise it's derived from the topic. + """) + public String wiki_compile_page( + @ToolParam(description = "Agent ID") Long agentId, + @ToolParam(description = "Topic to compile a page about (natural language)") String topic, + @ToolParam(description = "Optional explicit slug for the page", required = false) String slug, + @ToolParam(description = "Max evidence chunks (default 8, max 20)", required = false) Integer maxEvidenceChunks) { + + if (topic == null || topic.isBlank()) { + return error("topic is required"); + } + Long kbId = resolveKbId(agentId); + if (kbId == null) return error("No wiki knowledge base found for this agent"); + if (compileService == null) return error("Compile service not available"); + + try { + WikiCompileService.CompileResult res = compileService.compilePage(kbId, topic, slug, maxEvidenceChunks); + return JSONUtil.createObj() + .set("ok", true) + .set("slug", res.slug()) + .set("title", res.title()) + .set("evidenceChunks", res.evidenceChunkCount()) + .set("created", res.created()) + .toString(); + } catch (IllegalStateException | IllegalArgumentException e) { + return error(e.getMessage()); + } catch (Exception e) { + log.warn("[WikiTool] wiki_compile_page failed: {}", e.getMessage()); + return error("Compile failed: " + e.getMessage()); + } + } + + @Tool(description = """ + Read multiple wiki pages in one call. Prefer this over multiple wiki_read_page + calls when you already know the slugs you need. The response is capped per + page; protected/system pages can still be read explicitly here. + """) + public String wiki_read_many( + @ToolParam(description = "Agent ID") Long agentId, + @ToolParam(description = "Comma-separated slugs (max 10)") String slugs, + @ToolParam(description = "Max chars returned per page (default 2000, max 8000)", required = false) Integer maxCharsPerPage) { + + if (slugs == null || slugs.isBlank()) return error("slugs is required"); + Long kbId = resolveKbId(agentId); + if (kbId == null) return error("No wiki knowledge base found for this agent"); + + int cap = (maxCharsPerPage == null || maxCharsPerPage <= 0) ? 2000 : Math.min(8000, maxCharsPerPage); + List slugList = Arrays.stream(slugs.split(",")) + .map(String::trim).filter(s -> !s.isEmpty()).limit(10).toList(); + if (slugList.isEmpty()) return error("No valid slugs supplied"); + + JSONArray arr = new JSONArray(); + for (String s : slugList) { + WikiPageEntity page = pageService.getBySlug(kbId, s); + if (page == null) { + arr.add(JSONUtil.createObj().set("slug", s).set("found", false)); + continue; + } + String content = page.getContent() == null ? "" : page.getContent(); + boolean truncated = content.length() > cap; + if (truncated) content = content.substring(0, cap) + "\n…(truncated)"; + arr.add(JSONUtil.createObj() + .set("slug", s) + .set("found", true) + .set("title", page.getTitle()) + .set("summary", page.getSummary()) + .set("content", content) + .set("truncated", truncated)); + pageService.trackReference(kbId, s); + } + return JSONUtil.createObj() + .set("kbId", kbId) + .set("requestedCount", slugList.size()) + .set("pages", arr) + .toString(); + } + @Tool(description = """ Delete an AI-generated wiki page. Cannot delete manually curated pages. """)