mirror of
https://gitee.com/mateos/mateclaw.git
synced 2026-09-13 03:13:41 +08:00
feat(tool/document): generate xlsx, pptx, pdf documents from markdown
This commit is contained in:
parent
a65aaa7aa4
commit
c82bf4d18f
@ -351,6 +351,51 @@
|
||||
<version>3.0.0</version>
|
||||
</dependency>
|
||||
|
||||
<!-- ===== Markdown -> PDF rendering =====
|
||||
Flying Saucer 9.13 ships a single `flying-saucer-pdf` artifact that
|
||||
writes PDF via OpenPDF (LGPL fork of iText 5). It does NOT depend on
|
||||
PDFBox, so it sidesteps a version conflict with the existing
|
||||
pdfbox:3.0.3 dependency. CSS3 paged-media features (@page,
|
||||
counter(page), counter(pages), @top-center / @bottom-center) are
|
||||
supported, which the cover / header / footer rendering relies on.
|
||||
|
||||
commonmark-java is the reference CommonMark implementation,
|
||||
actively maintained on a monthly cadence (vs. flexmark, whose
|
||||
upstream stalled at 0.64.8 in 2023). It parses markdown into the
|
||||
XHTML Flying Saucer consumes. The alternative LibreOffice path in
|
||||
PdfRenderTool reuses MarkdownDocxRenderer + a soffice subprocess
|
||||
and adds no dependencies of its own. -->
|
||||
<dependency>
|
||||
<groupId>org.xhtmlrenderer</groupId>
|
||||
<artifactId>flying-saucer-pdf</artifactId>
|
||||
<version>9.13.0</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.commonmark</groupId>
|
||||
<artifactId>commonmark</artifactId>
|
||||
<version>0.28.0</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.commonmark</groupId>
|
||||
<artifactId>commonmark-ext-gfm-tables</artifactId>
|
||||
<version>0.28.0</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.commonmark</groupId>
|
||||
<artifactId>commonmark-ext-yaml-front-matter</artifactId>
|
||||
<version>0.28.0</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.commonmark</groupId>
|
||||
<artifactId>commonmark-ext-gfm-strikethrough</artifactId>
|
||||
<version>0.28.0</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.commonmark</groupId>
|
||||
<artifactId>commonmark-ext-autolink</artifactId>
|
||||
<version>0.28.0</version>
|
||||
</dependency>
|
||||
|
||||
<!-- ===== Database Migration (Flyway) ===== -->
|
||||
<dependency>
|
||||
<groupId>org.flywaydb</groupId>
|
||||
|
||||
@ -5,14 +5,14 @@ import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.ai.tool.annotation.Tool;
|
||||
import org.springframework.ai.tool.annotation.ToolParam;
|
||||
import org.springframework.stereotype.Component;
|
||||
import vip.mate.tool.document.FilenameSanitizer;
|
||||
import vip.mate.tool.document.GeneratedFileCache;
|
||||
import vip.mate.tool.document.GeneratedFileLink;
|
||||
import vip.mate.tool.document.MarkdownDocxRenderer;
|
||||
import vip.mate.tool.guard.WorkspacePathGuard;
|
||||
import vip.mate.tool.document.MarkdownInputResolver;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.Resolved;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.ResolveException;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
@ -39,8 +39,10 @@ public class DocxRenderTool {
|
||||
private final GeneratedFileCache cache;
|
||||
|
||||
@Tool(description = """
|
||||
Render a new .docx file from Markdown text and return a one-time download URL.
|
||||
Use for creating NEW documents: reports, memos, contracts, letters, resumes.
|
||||
Render a new .docx (Microsoft Word) file from Markdown text and return a
|
||||
one-time download URL. Use for creating EDITABLE Word documents the user
|
||||
will continue to revise — reports, memos, contracts, letters, resumes.
|
||||
|
||||
Supports: headings (# ## ###), bold (**text**), bullet lists (- item),
|
||||
numbered lists (1. item), tables (| col | col |), plain paragraphs,
|
||||
images () — SVG is rasterized
|
||||
@ -50,6 +52,11 @@ public class DocxRenderTool {
|
||||
disk) — passing huge markdown as a tool argument burns LLM tokens needlessly.
|
||||
|
||||
Do NOT use for:
|
||||
- **Anything the user asked for in PDF / .pdf format — use `renderPdf` /
|
||||
`renderPdfFromFile` instead. PDF is a separate non-editable deliverable
|
||||
format; don't silently substitute docx for it.**
|
||||
- Spreadsheets / workbooks — use `renderXlsx` / `renderXlsxFromFile`.
|
||||
- Slide decks / presentations — use `renderPptx` / `renderPptxFromFile`.
|
||||
- Editing an existing .docx file (use run_skill_script with unpack/edit/pack)
|
||||
- Adding tracked changes or comments (use run_skill_script)
|
||||
- GB/T 9704 official documents (use writeGongwen tool, BmacClaw only)
|
||||
@ -69,26 +76,15 @@ public class DocxRenderTool {
|
||||
return "错误:markdown 参数为空,无法生成文档。";
|
||||
}
|
||||
|
||||
String safeName = sanitizeFilename(filename);
|
||||
String displayName = safeName + ".docx";
|
||||
String size = (pageSize == null || pageSize.isBlank()) ? "A4" : pageSize.trim();
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "document", ".docx") + ".docx";
|
||||
String size = resolveSize(pageSize);
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(markdown, size);
|
||||
String id = cache.put(bytes, displayName, DOCX_MIME);
|
||||
long elapsed = System.currentTimeMillis() - t0;
|
||||
log.info("[DocxRender] generated {} ({} bytes, {}ms, id={})",
|
||||
displayName, bytes.length, elapsed, id);
|
||||
|
||||
String url = "/api/v1/files/generated/" + id;
|
||||
// Explicit instruction to suppress LLM hallucinating an absolute host.
|
||||
// DeepSeek/Claude have been observed prepending placeholder domains
|
||||
// (e.g. https://ai-tools-system.com) when echoing the URL back to the user,
|
||||
// breaking the download link. Repeat the path verbatim with no host.
|
||||
return "文档已生成:[" + displayName + "](" + url + ")(链接 10 分钟内有效)。\n"
|
||||
+ "重要:回答用户时**必须**使用上述相对路径 `" + url + "`,"
|
||||
+ "**不要**添加任何 https://、http:// 域名前缀,前端会自动拼接当前主机。";
|
||||
log.info("[DocxRender] generated {} ({} bytes, {}ms)",
|
||||
displayName, bytes.length, System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultZh(bytes, displayName, DOCX_MIME, cache, "文档");
|
||||
} catch (Exception e) {
|
||||
log.error("[DocxRender] render failed for {}: {}", displayName, e.getMessage(), e);
|
||||
return "渲染失败:" + e.getMessage();
|
||||
@ -107,7 +103,13 @@ public class DocxRenderTool {
|
||||
* rendered from disk in one IO call. Token cost ≈ 50 (just the path).
|
||||
*/
|
||||
@Tool(description = """
|
||||
Render a .docx file from a markdown FILE on disk and return a one-time download URL.
|
||||
Render a .docx (Microsoft Word) file from a markdown FILE on disk and return
|
||||
a one-time download URL. Use this for EDITABLE Word documents only.
|
||||
|
||||
**If the user asked for PDF / .pdf in any wording, use `renderPdfFromFile`
|
||||
instead. Do not silently substitute docx for PDF.** Same for spreadsheets
|
||||
(`renderXlsxFromFile`) and slide decks (`renderPptxFromFile`).
|
||||
|
||||
Use this instead of `renderDocx` when the markdown body is large (>5 KB) — the
|
||||
LLM does not need to repeat its own previous output as a tool argument.
|
||||
|
||||
@ -132,56 +134,25 @@ public class DocxRenderTool {
|
||||
@ToolParam(description = "Page size: A4 or LETTER (default: A4)", required = false)
|
||||
String pageSize) {
|
||||
|
||||
if (filePath == null || filePath.isBlank()) {
|
||||
return "Error: filePath parameter is empty.";
|
||||
}
|
||||
|
||||
Path resolved;
|
||||
Resolved input;
|
||||
try {
|
||||
resolved = WorkspacePathGuard.validatePath(filePath);
|
||||
} catch (Exception e) {
|
||||
return "Error: path validation failed — " + e.getMessage();
|
||||
}
|
||||
if (!Files.exists(resolved)) {
|
||||
return "Error: file not found at " + resolved;
|
||||
}
|
||||
if (!Files.isRegularFile(resolved) || !Files.isReadable(resolved)) {
|
||||
return "Error: path is not a readable regular file " + resolved;
|
||||
input = MarkdownInputResolver.readSingle(filePath);
|
||||
} catch (ResolveException e) {
|
||||
return "Error: " + e.getMessage();
|
||||
}
|
||||
|
||||
String markdown;
|
||||
long mdBytes;
|
||||
try {
|
||||
mdBytes = Files.size(resolved);
|
||||
markdown = Files.readString(resolved, StandardCharsets.UTF_8);
|
||||
} catch (Exception e) {
|
||||
log.error("[DocxRender] read markdown failed for {}: {}", resolved, e.getMessage(), e);
|
||||
return "Error: failed to read markdown — " + e.getMessage();
|
||||
}
|
||||
if (markdown.isBlank()) {
|
||||
return "Error: markdown file is empty " + resolved;
|
||||
}
|
||||
|
||||
String safeName = sanitizeFilename(filename);
|
||||
String displayName = safeName + ".docx";
|
||||
String size = (pageSize == null || pageSize.isBlank()) ? "A4" : pageSize.trim();
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "document", ".docx") + ".docx";
|
||||
String size = resolveSize(pageSize);
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(markdown, size);
|
||||
String id = cache.put(bytes, displayName, DOCX_MIME);
|
||||
long elapsed = System.currentTimeMillis() - t0;
|
||||
log.info("[DocxRender] generated {} ({} bytes from {} bytes md, {}ms, id={})",
|
||||
displayName, bytes.length, mdBytes, elapsed, id);
|
||||
|
||||
String url = "/api/v1/files/generated/" + id;
|
||||
return "Document generated: [" + displayName + "](" + url + ") (link valid for 10 minutes).\n"
|
||||
+ "IMPORTANT: when replying to the user you **must** use the relative path `"
|
||||
+ url + "` verbatim. Do **not** prepend any https://, http:// or domain — "
|
||||
+ "the frontend will resolve the current host automatically.";
|
||||
byte[] bytes = renderer.render(input.markdown(), size);
|
||||
log.info("[DocxRender] generated {} ({} bytes from {} bytes md, {}ms)",
|
||||
displayName, bytes.length, input.totalBytes(), System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultEn(bytes, displayName, DOCX_MIME, cache, "Document", 1);
|
||||
} catch (Exception e) {
|
||||
log.error("[DocxRender] render failed for {} (source: {}): {}",
|
||||
displayName, resolved, e.getMessage(), e);
|
||||
displayName, input.sources().get(0), e.getMessage(), e);
|
||||
return "Render failed: " + e.getMessage();
|
||||
}
|
||||
}
|
||||
@ -222,91 +193,32 @@ public class DocxRenderTool {
|
||||
@ToolParam(description = "Page size: A4 or LETTER (default: A4)", required = false)
|
||||
String pageSize) {
|
||||
|
||||
if (filePaths == null || filePaths.isEmpty()) {
|
||||
return "Error: filePaths is empty.";
|
||||
Resolved input;
|
||||
try {
|
||||
input = MarkdownInputResolver.readManyJoined(filePaths);
|
||||
} catch (ResolveException e) {
|
||||
return "Error: " + e.getMessage();
|
||||
}
|
||||
|
||||
StringBuilder combined = new StringBuilder();
|
||||
long totalBytes = 0;
|
||||
List<String> resolvedPaths = new ArrayList<>();
|
||||
for (int idx = 0; idx < filePaths.size(); idx++) {
|
||||
String raw = filePaths.get(idx);
|
||||
if (raw == null || raw.isBlank()) {
|
||||
return "Error: filePaths[" + idx + "] is empty.";
|
||||
}
|
||||
Path resolved;
|
||||
try {
|
||||
resolved = WorkspacePathGuard.validatePath(raw);
|
||||
} catch (Exception e) {
|
||||
return "Error: filePaths[" + idx + "] validation failed — " + e.getMessage();
|
||||
}
|
||||
if (!Files.exists(resolved)) {
|
||||
return "Error: filePaths[" + idx + "] not found at " + resolved;
|
||||
}
|
||||
if (!Files.isRegularFile(resolved) || !Files.isReadable(resolved)) {
|
||||
return "Error: filePaths[" + idx + "] is not a readable regular file " + resolved;
|
||||
}
|
||||
String content;
|
||||
try {
|
||||
totalBytes += Files.size(resolved);
|
||||
content = Files.readString(resolved, StandardCharsets.UTF_8);
|
||||
} catch (Exception e) {
|
||||
log.error("[DocxRender] read failed for {}: {}", resolved, e.getMessage(), e);
|
||||
return "Error: read failed for " + resolved + " — " + e.getMessage();
|
||||
}
|
||||
if (content.isBlank()) {
|
||||
return "Error: filePaths[" + idx + "] is blank " + resolved;
|
||||
}
|
||||
if (combined.length() > 0) combined.append("\n\n");
|
||||
combined.append(content);
|
||||
resolvedPaths.add(resolved.toString());
|
||||
}
|
||||
|
||||
String safeName = sanitizeFilename(filename);
|
||||
String displayName = safeName + ".docx";
|
||||
String size = (pageSize == null || pageSize.isBlank()) ? "A4" : pageSize.trim();
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "document", ".docx") + ".docx";
|
||||
String size = resolveSize(pageSize);
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(combined.toString(), size);
|
||||
String id = cache.put(bytes, displayName, DOCX_MIME);
|
||||
long elapsed = System.currentTimeMillis() - t0;
|
||||
log.info("[DocxRender] generated {} ({} bytes from {} files / {} bytes md, {}ms, id={})",
|
||||
displayName, bytes.length, resolvedPaths.size(), totalBytes, elapsed, id);
|
||||
|
||||
String url = "/api/v1/files/generated/" + id;
|
||||
return "Document generated from " + resolvedPaths.size() + " files: ["
|
||||
+ displayName + "](" + url + ") (link valid for 10 minutes).\n"
|
||||
+ "IMPORTANT: when replying to the user you **must** use the relative path `"
|
||||
+ url + "` verbatim. Do **not** prepend any https://, http:// or domain — "
|
||||
+ "the frontend will resolve the current host automatically.";
|
||||
byte[] bytes = renderer.render(input.markdown(), size);
|
||||
log.info("[DocxRender] generated {} ({} bytes from {} files / {} bytes md, {}ms)",
|
||||
displayName, bytes.length, input.fileCount(), input.totalBytes(),
|
||||
System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultEn(bytes, displayName, DOCX_MIME, cache,
|
||||
"Document", input.fileCount());
|
||||
} catch (Exception e) {
|
||||
log.error("[DocxRender] render failed for {} (sources: {}): {}",
|
||||
displayName, resolvedPaths, e.getMessage(), e);
|
||||
displayName, input.sources(), e.getMessage(), e);
|
||||
return "Render failed: " + e.getMessage();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip path separators and other unsafe characters from a user-supplied
|
||||
* filename. Falls back to a generic name when nothing usable remains.
|
||||
*/
|
||||
private String sanitizeFilename(String name) {
|
||||
if (name == null) return "document";
|
||||
String trimmed = name.trim();
|
||||
if (trimmed.toLowerCase().endsWith(".docx")) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - 5);
|
||||
}
|
||||
StringBuilder sb = new StringBuilder(trimmed.length());
|
||||
for (char c : trimmed.toCharArray()) {
|
||||
if (c == '/' || c == '\\' || c == ':' || c == '*' || c == '?'
|
||||
|| c == '"' || c == '<' || c == '>' || c == '|' || c < 0x20) {
|
||||
sb.append('_');
|
||||
} else {
|
||||
sb.append(c);
|
||||
}
|
||||
}
|
||||
String cleaned = sb.toString().strip();
|
||||
return cleaned.isEmpty() ? "document" : cleaned;
|
||||
private static String resolveSize(String pageSize) {
|
||||
return (pageSize == null || pageSize.isBlank()) ? "A4" : pageSize.trim();
|
||||
}
|
||||
}
|
||||
|
||||
@ -0,0 +1,172 @@
|
||||
package vip.mate.tool.builtin;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.ai.tool.annotation.Tool;
|
||||
import org.springframework.ai.tool.annotation.ToolParam;
|
||||
import org.springframework.stereotype.Component;
|
||||
import vip.mate.tool.document.FilenameSanitizer;
|
||||
import vip.mate.tool.document.GeneratedFileCache;
|
||||
import vip.mate.tool.document.GeneratedFileLink;
|
||||
import vip.mate.tool.document.MarkdownInputResolver;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.Resolved;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.ResolveException;
|
||||
import vip.mate.tool.document.pdf.MarkdownPdfRenderer;
|
||||
import vip.mate.tool.document.pdf.PdfProperties;
|
||||
|
||||
import java.util.Locale;
|
||||
|
||||
/**
|
||||
* Render a brand-new .pdf from Markdown. Two backends sit behind this tool:
|
||||
* a LibreOffice subprocess (preferred when {@code soffice} is available, best
|
||||
* Chinese typography) and an in-process OpenHTMLtoPDF path (always available,
|
||||
* supports cover / page header / page footer driven by YAML frontmatter).
|
||||
* The orchestrator picks one per call; see {@link MarkdownPdfRenderer}.
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class PdfRenderTool {
|
||||
|
||||
private static final String PDF_MIME = "application/pdf";
|
||||
|
||||
private final MarkdownPdfRenderer renderer;
|
||||
private final GeneratedFileCache cache;
|
||||
|
||||
@Tool(description = """
|
||||
Render a NEW .pdf file from Markdown and return a one-time download URL.
|
||||
|
||||
**MUST use this tool (NOT renderDocx / renderDocxFromFile) whenever the user
|
||||
says any of: "PDF", ".pdf", "导出 PDF", "生成 pdf", "另存为 PDF", "出一份 PDF",
|
||||
"save as PDF", "export to PDF".** PDF is a final, non-editable deliverable
|
||||
format; if the user asked for it explicitly, do not silently substitute docx.
|
||||
|
||||
**Do NOT bypass this tool by shelling out to `chrome --headless --print-to-pdf`,
|
||||
`wkhtmltopdf`, `weasyprint`, or any markdown-to-PDF Python skill. Those produce
|
||||
a PDF on local disk that is NOT registered in mateclaw's download cache, so
|
||||
the user has no clickable download link and the file leaks into the workspace.
|
||||
Always use this tool instead — it returns a `/api/v1/files/generated/<id>` URL
|
||||
the user can download from chat.**
|
||||
|
||||
Use for FINAL deliverables — reports, white-papers, contracts, briefings —
|
||||
where the recipient should not edit the document.
|
||||
|
||||
Markdown convention:
|
||||
- Standard subset: headings (# ## ###), bold, italic, lists, tables,
|
||||
blockquotes, code blocks, links.
|
||||
- Optional YAML frontmatter at the top of the markdown drives cover
|
||||
page and page header / footer:
|
||||
|
||||
---
|
||||
title: 季度业务回顾
|
||||
subtitle: Q1 2026
|
||||
header: 内部资料 - 仅限分发
|
||||
footer: Mate Inc. © 2026
|
||||
---
|
||||
|
||||
# 第一章
|
||||
...
|
||||
|
||||
- Without frontmatter, the first `# H1` heading is used as the cover
|
||||
title and pages are numbered automatically with no header / footer.
|
||||
|
||||
For markdown bodies larger than ~5 KB, prefer renderPdfFromFile.
|
||||
|
||||
Returns a markdown link the user can click to download the file.
|
||||
The link is valid for 10 minutes.
|
||||
""")
|
||||
public String renderPdf(
|
||||
@ToolParam(description = "Document content in Markdown format (optional YAML frontmatter for cover / header / footer)")
|
||||
String markdown,
|
||||
@ToolParam(description = "Output filename without extension, e.g. 'q1-review'")
|
||||
String filename,
|
||||
@ToolParam(description = "Page size: A4 or LETTER (default: A4)", required = false)
|
||||
String pageSize,
|
||||
@ToolParam(description = "Engine: 'auto' (default), 'html' (force in-process), or 'libreoffice' (force soffice)", required = false)
|
||||
String engine) {
|
||||
|
||||
if (markdown == null || markdown.isBlank()) {
|
||||
return "错误:markdown 参数为空,无法生成 PDF。";
|
||||
}
|
||||
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "document", ".pdf") + ".pdf";
|
||||
String size = resolveSize(pageSize);
|
||||
PdfProperties.Engine eng = resolveEngine(engine);
|
||||
|
||||
try {
|
||||
MarkdownPdfRenderer.Result result = renderer.render(markdown, size, eng);
|
||||
log.info("[PdfRender] generated {} ({} bytes via {})",
|
||||
displayName, result.bytes().length, result.backend());
|
||||
return GeneratedFileLink.resultZh(result.bytes(), displayName, PDF_MIME, cache, "PDF");
|
||||
} catch (Exception e) {
|
||||
log.error("[PdfRender] render failed for {}: {}", displayName, e.getMessage(), e);
|
||||
return "渲染失败:" + e.getMessage();
|
||||
}
|
||||
}
|
||||
|
||||
@Tool(description = """
|
||||
Render a .pdf from a markdown FILE on disk and return a one-time download URL.
|
||||
|
||||
**MUST use this tool (NOT renderDocxFromFile) whenever the user asks for a
|
||||
PDF / .pdf / 导出 PDF / 生成 pdf and the markdown body is already on disk.**
|
||||
Do not silently substitute docx when the user explicitly requested PDF.
|
||||
|
||||
Use this instead of `renderPdf` when the markdown body is large (>5 KB) — the
|
||||
LLM does not need to repeat its own previous output as a tool argument.
|
||||
|
||||
Typical workflow:
|
||||
1. write_file(path="report.md", content="---\\ntitle: ...\\n---\\n# ...")
|
||||
2. renderPdfFromFile(filePath="report.md", filename="q1-review")
|
||||
3. return the download link to the user
|
||||
|
||||
The markdown file is read with UTF-8. Path resolution honors the workspace
|
||||
boundary (same rules as read_file / write_file).
|
||||
|
||||
Same supported markdown subset and frontmatter convention as renderPdf.
|
||||
""")
|
||||
public String renderPdfFromFile(
|
||||
@ToolParam(description = "Absolute or workspace-relative path to a markdown file")
|
||||
String filePath,
|
||||
@ToolParam(description = "Output filename without extension, e.g. 'q1-review'")
|
||||
String filename,
|
||||
@ToolParam(description = "Page size: A4 or LETTER (default: A4)", required = false)
|
||||
String pageSize,
|
||||
@ToolParam(description = "Engine: 'auto' (default), 'html', or 'libreoffice'", required = false)
|
||||
String engine) {
|
||||
|
||||
Resolved input;
|
||||
try {
|
||||
input = MarkdownInputResolver.readSingle(filePath);
|
||||
} catch (ResolveException e) {
|
||||
return "Error: " + e.getMessage();
|
||||
}
|
||||
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "document", ".pdf") + ".pdf";
|
||||
String size = resolveSize(pageSize);
|
||||
PdfProperties.Engine eng = resolveEngine(engine);
|
||||
|
||||
try {
|
||||
MarkdownPdfRenderer.Result result = renderer.render(input.markdown(), size, eng);
|
||||
log.info("[PdfRender] generated {} ({} bytes via {} from {} bytes md)",
|
||||
displayName, result.bytes().length, result.backend(), input.totalBytes());
|
||||
return GeneratedFileLink.resultEn(result.bytes(), displayName, PDF_MIME, cache, "Document", 1);
|
||||
} catch (Exception e) {
|
||||
log.error("[PdfRender] render failed for {} (source: {}): {}",
|
||||
displayName, input.sources().get(0), e.getMessage(), e);
|
||||
return "Render failed: " + e.getMessage();
|
||||
}
|
||||
}
|
||||
|
||||
private static String resolveSize(String pageSize) {
|
||||
return (pageSize == null || pageSize.isBlank()) ? "A4" : pageSize.trim();
|
||||
}
|
||||
|
||||
private static PdfProperties.Engine resolveEngine(String engine) {
|
||||
if (engine == null || engine.isBlank()) return PdfProperties.Engine.AUTO;
|
||||
try {
|
||||
return PdfProperties.Engine.valueOf(engine.trim().toUpperCase(Locale.ROOT));
|
||||
} catch (IllegalArgumentException e) {
|
||||
return PdfProperties.Engine.AUTO;
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,149 @@
|
||||
package vip.mate.tool.builtin;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.ai.tool.annotation.Tool;
|
||||
import org.springframework.ai.tool.annotation.ToolParam;
|
||||
import org.springframework.stereotype.Component;
|
||||
import vip.mate.tool.document.FilenameSanitizer;
|
||||
import vip.mate.tool.document.GeneratedFileCache;
|
||||
import vip.mate.tool.document.GeneratedFileLink;
|
||||
import vip.mate.tool.document.MarkdownInputResolver;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.Resolved;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.ResolveException;
|
||||
import vip.mate.tool.document.MarkdownPptxRenderer;
|
||||
|
||||
/**
|
||||
* Render a brand-new .pptx deck from Markdown, in-process via Apache POI.
|
||||
* The LLM produces a Marp-style markdown body where {@code ---} separates
|
||||
* slides, {@code # / ## / ###} is the slide title, and {@code - item} are
|
||||
* bullets.
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class PptxRenderTool {
|
||||
|
||||
private static final String PPTX_MIME =
|
||||
"application/vnd.openxmlformats-officedocument.presentationml.presentation";
|
||||
|
||||
private final MarkdownPptxRenderer renderer;
|
||||
private final GeneratedFileCache cache;
|
||||
|
||||
@Tool(description = """
|
||||
Render a NEW .pptx slide deck from Markdown and return a one-time download URL.
|
||||
Use for creating presentations: pitch decks, project plans, talks, briefings.
|
||||
|
||||
Markdown convention (Marp-style):
|
||||
- `---` on its own line separates slides.
|
||||
- The first `# / ## / ###` of a slide becomes its title.
|
||||
- Lines starting with `-` or `*` become bullet points.
|
||||
- Other non-blank lines become plain paragraphs.
|
||||
- `<!-- speaker note -->` HTML comments become speaker notes.
|
||||
|
||||
Example:
|
||||
# My Presentation
|
||||
|
||||
By Author Name
|
||||
|
||||
---
|
||||
|
||||
## Topic 1
|
||||
|
||||
- Point one
|
||||
- Point two
|
||||
- Point three
|
||||
|
||||
<!-- Remember to mention the timeline here. -->
|
||||
|
||||
---
|
||||
|
||||
## Conclusion
|
||||
|
||||
Thanks!
|
||||
|
||||
For markdown bodies larger than ~5 KB, prefer renderPptxFromFile (read
|
||||
from disk) — passing huge markdown as a tool argument burns LLM tokens.
|
||||
|
||||
Returns a markdown link the user can click to download the file.
|
||||
The link is valid for 10 minutes.
|
||||
""")
|
||||
public String renderPptx(
|
||||
@ToolParam(description = "Slide content in Marp-style Markdown ('---' between slides)")
|
||||
String markdown,
|
||||
@ToolParam(description = "Output filename without extension, e.g. 'pitch-deck'")
|
||||
String filename,
|
||||
@ToolParam(description = "Aspect ratio: '16:9' (default, widescreen) or '4:3' (legacy)", required = false)
|
||||
String aspectRatio) {
|
||||
|
||||
if (markdown == null || markdown.isBlank()) {
|
||||
return "错误:markdown 参数为空,无法生成演示文稿。";
|
||||
}
|
||||
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "presentation", ".pptx") + ".pptx";
|
||||
String ratio = resolveRatio(aspectRatio);
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(markdown, ratio);
|
||||
log.info("[PptxRender] generated {} ({} bytes, {}ms)",
|
||||
displayName, bytes.length, System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultZh(bytes, displayName, PPTX_MIME, cache, "演示文稿");
|
||||
} catch (Exception e) {
|
||||
log.error("[PptxRender] render failed for {}: {}", displayName, e.getMessage(), e);
|
||||
return "渲染失败:" + e.getMessage();
|
||||
}
|
||||
}
|
||||
|
||||
@Tool(description = """
|
||||
Render a .pptx deck from a markdown FILE on disk and return a one-time download URL.
|
||||
Use this instead of `renderPptx` when the markdown body is large (>5 KB) — the
|
||||
LLM does not need to repeat its own previous output as a tool argument.
|
||||
|
||||
Typical workflow:
|
||||
1. write_file(path="deck.md", content="# Title\\n\\n---\\n\\n## Topic\\n\\n- ...")
|
||||
2. renderPptxFromFile(filePath="deck.md", filename="pitch-deck")
|
||||
3. return the download link to the user
|
||||
|
||||
The markdown file is read with UTF-8. Path resolution honors the workspace
|
||||
boundary (same rules as read_file / write_file).
|
||||
|
||||
Same supported Marp-style markdown subset as renderPptx (`---` slide breaks,
|
||||
`# / ##` titles, `-` / `*` bullets, `<!-- ... -->` speaker notes).
|
||||
""")
|
||||
public String renderPptxFromFile(
|
||||
@ToolParam(description = "Absolute or workspace-relative path to a markdown file")
|
||||
String filePath,
|
||||
@ToolParam(description = "Output filename without extension, e.g. 'pitch-deck'")
|
||||
String filename,
|
||||
@ToolParam(description = "Aspect ratio: '16:9' (default) or '4:3'", required = false)
|
||||
String aspectRatio) {
|
||||
|
||||
Resolved input;
|
||||
try {
|
||||
input = MarkdownInputResolver.readSingle(filePath);
|
||||
} catch (ResolveException e) {
|
||||
return "Error: " + e.getMessage();
|
||||
}
|
||||
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "presentation", ".pptx") + ".pptx";
|
||||
String ratio = resolveRatio(aspectRatio);
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(input.markdown(), ratio);
|
||||
log.info("[PptxRender] generated {} ({} bytes from {} bytes md, {}ms)",
|
||||
displayName, bytes.length, input.totalBytes(),
|
||||
System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultEn(bytes, displayName, PPTX_MIME, cache, "Presentation", 1);
|
||||
} catch (Exception e) {
|
||||
log.error("[PptxRender] render failed for {} (source: {}): {}",
|
||||
displayName, input.sources().get(0), e.getMessage(), e);
|
||||
return "Render failed: " + e.getMessage();
|
||||
}
|
||||
}
|
||||
|
||||
private static String resolveRatio(String aspectRatio) {
|
||||
return (aspectRatio == null || aspectRatio.isBlank()) ? "16:9" : aspectRatio.trim();
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,133 @@
|
||||
package vip.mate.tool.builtin;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.ai.tool.annotation.Tool;
|
||||
import org.springframework.ai.tool.annotation.ToolParam;
|
||||
import org.springframework.stereotype.Component;
|
||||
import vip.mate.tool.document.FilenameSanitizer;
|
||||
import vip.mate.tool.document.GeneratedFileCache;
|
||||
import vip.mate.tool.document.GeneratedFileLink;
|
||||
import vip.mate.tool.document.MarkdownInputResolver;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.Resolved;
|
||||
import vip.mate.tool.document.MarkdownInputResolver.ResolveException;
|
||||
import vip.mate.tool.document.MarkdownXlsxRenderer;
|
||||
|
||||
/**
|
||||
* Render a brand-new .xlsx workbook from a Markdown body, in-process via
|
||||
* Apache POI. Mirrors {@link DocxRenderTool}'s shape: the LLM produces a
|
||||
* Markdown body where each {@code # Heading} starts a sheet and the pipe-style
|
||||
* table beneath it becomes the sheet content.
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class XlsxRenderTool {
|
||||
|
||||
private static final String XLSX_MIME =
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet";
|
||||
|
||||
private final MarkdownXlsxRenderer renderer;
|
||||
private final GeneratedFileCache cache;
|
||||
|
||||
@Tool(description = """
|
||||
Render a NEW .xlsx workbook from Markdown and return a one-time download URL.
|
||||
Use for creating spreadsheets: financial reports, data tables, comparison
|
||||
matrices, plans, schedules.
|
||||
|
||||
Markdown convention:
|
||||
- Each `# Sheet Name` starts a new sheet.
|
||||
- The pipe-style table under the heading becomes the sheet body.
|
||||
- The first table row is rendered as the header (bold, light-grey fill,
|
||||
frozen). Numeric cells are auto-detected and stored as numbers so
|
||||
Excel can sort / sum them; non-numeric cells stay as strings.
|
||||
- Sub-headings (## / ###) and free-form prose are ignored — xlsx is
|
||||
tabular and there is nowhere sensible to put them.
|
||||
|
||||
Example:
|
||||
# Q1 Sales
|
||||
| Region | Revenue | Growth |
|
||||
| --- | --- | --- |
|
||||
| North | 12000 | 0.15 |
|
||||
| South | 8500 | 0.08 |
|
||||
|
||||
# Q2 Sales
|
||||
| Region | Revenue |
|
||||
| --- | --- |
|
||||
| North | 14000 |
|
||||
|
||||
For markdown bodies larger than ~5 KB, prefer renderXlsxFromFile (read
|
||||
from disk) — passing huge markdown as a tool argument burns LLM tokens.
|
||||
|
||||
Returns a markdown link the user can click to download the file.
|
||||
The link is valid for 10 minutes.
|
||||
""")
|
||||
public String renderXlsx(
|
||||
@ToolParam(description = "Workbook content in Markdown format (sheets as `# Heading`, tables as `| ... |`)")
|
||||
String markdown,
|
||||
@ToolParam(description = "Output filename without extension, e.g. 'q1-sales'")
|
||||
String filename) {
|
||||
|
||||
if (markdown == null || markdown.isBlank()) {
|
||||
return "错误:markdown 参数为空,无法生成工作簿。";
|
||||
}
|
||||
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "workbook", ".xlsx") + ".xlsx";
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(markdown);
|
||||
log.info("[XlsxRender] generated {} ({} bytes, {}ms)",
|
||||
displayName, bytes.length, System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultZh(bytes, displayName, XLSX_MIME, cache, "工作簿");
|
||||
} catch (Exception e) {
|
||||
log.error("[XlsxRender] render failed for {}: {}", displayName, e.getMessage(), e);
|
||||
return "渲染失败:" + e.getMessage();
|
||||
}
|
||||
}
|
||||
|
||||
@Tool(description = """
|
||||
Render a .xlsx workbook from a markdown FILE on disk and return a one-time download URL.
|
||||
Use this instead of `renderXlsx` when the markdown body is large (>5 KB) — the
|
||||
LLM does not need to repeat its own previous output as a tool argument.
|
||||
|
||||
Typical workflow:
|
||||
1. write_file(path="report.md", content="# Q1\\n| ... |\\n...")
|
||||
2. renderXlsxFromFile(filePath="report.md", filename="quarterly-report")
|
||||
3. return the download link to the user
|
||||
|
||||
The markdown file is read with UTF-8. Path resolution honors the workspace
|
||||
boundary (same rules as read_file / write_file).
|
||||
|
||||
Same supported markdown subset as renderXlsx (`# Heading` per sheet,
|
||||
pipe-style tables; numeric cells auto-detected).
|
||||
""")
|
||||
public String renderXlsxFromFile(
|
||||
@ToolParam(description = "Absolute or workspace-relative path to a markdown file")
|
||||
String filePath,
|
||||
@ToolParam(description = "Output filename without extension, e.g. 'quarterly-report'")
|
||||
String filename) {
|
||||
|
||||
Resolved input;
|
||||
try {
|
||||
input = MarkdownInputResolver.readSingle(filePath);
|
||||
} catch (ResolveException e) {
|
||||
return "Error: " + e.getMessage();
|
||||
}
|
||||
|
||||
String displayName = FilenameSanitizer.sanitize(filename, "workbook", ".xlsx") + ".xlsx";
|
||||
|
||||
try {
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = renderer.render(input.markdown());
|
||||
log.info("[XlsxRender] generated {} ({} bytes from {} bytes md, {}ms)",
|
||||
displayName, bytes.length, input.totalBytes(),
|
||||
System.currentTimeMillis() - t0);
|
||||
return GeneratedFileLink.resultEn(bytes, displayName, XLSX_MIME, cache, "Workbook", 1);
|
||||
} catch (Exception e) {
|
||||
log.error("[XlsxRender] render failed for {} (source: {}): {}",
|
||||
displayName, input.sources().get(0), e.getMessage(), e);
|
||||
return "Render failed: " + e.getMessage();
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,42 @@
|
||||
package vip.mate.tool.document;
|
||||
|
||||
import java.util.Locale;
|
||||
|
||||
/**
|
||||
* Strip path separators and other characters that are illegal in download
|
||||
* filenames from an LLM-supplied name. The LLM is allowed to suffix the
|
||||
* extension itself (e.g. "report.docx") — {@link #sanitize} drops a known
|
||||
* extension before sanitizing so callers can re-append it consistently.
|
||||
*/
|
||||
public final class FilenameSanitizer {
|
||||
|
||||
private FilenameSanitizer() {}
|
||||
|
||||
/**
|
||||
* @param name candidate name from the LLM (may be null / blank / contain ext)
|
||||
* @param fallback name to use when {@code name} is null, blank, or sanitizes to empty
|
||||
* @param dropExt optional trailing extension to strip case-insensitively
|
||||
* before sanitizing (e.g. {@code ".docx"}); pass {@code null}
|
||||
* to skip
|
||||
* @return a non-blank base name with no path separators or shell metacharacters
|
||||
*/
|
||||
public static String sanitize(String name, String fallback, String dropExt) {
|
||||
if (name == null) return fallback;
|
||||
String trimmed = name.trim();
|
||||
if (dropExt != null && !dropExt.isEmpty()
|
||||
&& trimmed.toLowerCase(Locale.ROOT).endsWith(dropExt.toLowerCase(Locale.ROOT))) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - dropExt.length());
|
||||
}
|
||||
StringBuilder sb = new StringBuilder(trimmed.length());
|
||||
for (char c : trimmed.toCharArray()) {
|
||||
if (c == '/' || c == '\\' || c == ':' || c == '*' || c == '?'
|
||||
|| c == '"' || c == '<' || c == '>' || c == '|' || c < 0x20) {
|
||||
sb.append('_');
|
||||
} else {
|
||||
sb.append(c);
|
||||
}
|
||||
}
|
||||
String cleaned = sb.toString().strip();
|
||||
return cleaned.isEmpty() ? fallback : cleaned;
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,58 @@
|
||||
package vip.mate.tool.document;
|
||||
|
||||
/**
|
||||
* Stash freshly-rendered bytes into the {@link GeneratedFileCache} and format
|
||||
* the markdown link the tool returns to the LLM.
|
||||
*
|
||||
* <p>Two locales are exposed because mateclaw's existing convention has the
|
||||
* inline render tools speak Chinese and the file-driven render tools speak
|
||||
* English. Each variant carries the "do NOT prepend a host" instruction
|
||||
* because some models hallucinate a placeholder domain in front of the
|
||||
* relative URL when echoing it back.
|
||||
*/
|
||||
public final class GeneratedFileLink {
|
||||
|
||||
private GeneratedFileLink() {}
|
||||
|
||||
/**
|
||||
* Chinese-language tool result for inline render entry points
|
||||
* ({@code renderDocx} / {@code renderXlsx} / {@code renderPptx}).
|
||||
*
|
||||
* @param typeLabel "文档" / "工作簿" / "演示文稿"
|
||||
*/
|
||||
public static String resultZh(byte[] bytes, String displayName, String mimeType,
|
||||
GeneratedFileCache cache, String typeLabel) {
|
||||
String url = stash(bytes, displayName, mimeType, cache);
|
||||
return typeLabel + "已生成:[" + displayName + "](" + url + ")(链接 10 分钟内有效)。\n"
|
||||
+ "重要:回答用户时**必须**使用上述相对路径 `" + url + "`,"
|
||||
+ "**不要**添加任何 https://、http:// 域名前缀,前端会自动拼接当前主机。";
|
||||
}
|
||||
|
||||
/**
|
||||
* English-language tool result for file-driven render entry points
|
||||
* ({@code renderDocxFromFile} / {@code renderDocxFromFiles} / etc.).
|
||||
*
|
||||
* @param typeLabel "Document" / "Workbook" / "Presentation"
|
||||
* @param sourceFileCount number of source markdown files combined into the
|
||||
* artifact; values {@code > 1} produce a "from N files"
|
||||
* prefix, {@code 1} produces the plain "generated" prefix
|
||||
*/
|
||||
public static String resultEn(byte[] bytes, String displayName, String mimeType,
|
||||
GeneratedFileCache cache, String typeLabel,
|
||||
int sourceFileCount) {
|
||||
String url = stash(bytes, displayName, mimeType, cache);
|
||||
String prefix = sourceFileCount > 1
|
||||
? typeLabel + " generated from " + sourceFileCount + " files"
|
||||
: typeLabel + " generated";
|
||||
return prefix + ": [" + displayName + "](" + url + ") (link valid for 10 minutes).\n"
|
||||
+ "IMPORTANT: when replying to the user you **must** use the relative path `"
|
||||
+ url + "` verbatim. Do **not** prepend any https://, http:// or domain — "
|
||||
+ "the frontend will resolve the current host automatically.";
|
||||
}
|
||||
|
||||
private static String stash(byte[] bytes, String displayName, String mimeType,
|
||||
GeneratedFileCache cache) {
|
||||
String id = cache.put(bytes, displayName, mimeType);
|
||||
return "/api/v1/files/generated/" + id;
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,116 @@
|
||||
package vip.mate.tool.document;
|
||||
|
||||
import vip.mate.tool.guard.WorkspacePathGuard;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Read one or more markdown files from the workspace, returning a single
|
||||
* resolved record that the document-render tools can hand straight to a
|
||||
* markdown-to-bytes renderer.
|
||||
*
|
||||
* <p>All path validation goes through {@link WorkspacePathGuard} so the LLM
|
||||
* cannot escape the workspace boundary by passing {@code ../}-prefixed paths.
|
||||
* Errors are signalled via {@link ResolveException} carrying a short message
|
||||
* the tool layer surfaces verbatim to the model.
|
||||
*/
|
||||
public final class MarkdownInputResolver {
|
||||
|
||||
private MarkdownInputResolver() {}
|
||||
|
||||
public record Resolved(String markdown, List<Path> sources, long totalBytes) {
|
||||
public int fileCount() {
|
||||
return sources.size();
|
||||
}
|
||||
}
|
||||
|
||||
public static class ResolveException extends Exception {
|
||||
public ResolveException(String message) { super(message); }
|
||||
}
|
||||
|
||||
/** Read a single markdown file. */
|
||||
public static Resolved readSingle(String filePath) throws ResolveException {
|
||||
if (filePath == null || filePath.isBlank()) {
|
||||
throw new ResolveException("filePath parameter is empty.");
|
||||
}
|
||||
Path resolved = validate(filePath, -1);
|
||||
long size;
|
||||
String content;
|
||||
try {
|
||||
size = Files.size(resolved);
|
||||
content = Files.readString(resolved, StandardCharsets.UTF_8);
|
||||
} catch (Exception e) {
|
||||
throw new ResolveException("failed to read markdown — " + e.getMessage());
|
||||
}
|
||||
if (content.isBlank()) {
|
||||
throw new ResolveException("markdown file is empty " + resolved);
|
||||
}
|
||||
return new Resolved(content, List.of(resolved), size);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read multiple markdown files in order and join them with one blank line
|
||||
* between each. Used by the multi-chapter docx renderer so a long report
|
||||
* can live in {@code cover.md} / {@code ch1.md} / {@code ch2.md} and still
|
||||
* compile to a single document.
|
||||
*/
|
||||
public static Resolved readManyJoined(List<String> filePaths) throws ResolveException {
|
||||
if (filePaths == null || filePaths.isEmpty()) {
|
||||
throw new ResolveException("filePaths is empty.");
|
||||
}
|
||||
StringBuilder combined = new StringBuilder();
|
||||
long totalBytes = 0;
|
||||
List<Path> resolvedPaths = new ArrayList<>(filePaths.size());
|
||||
for (int idx = 0; idx < filePaths.size(); idx++) {
|
||||
String raw = filePaths.get(idx);
|
||||
if (raw == null || raw.isBlank()) {
|
||||
throw new ResolveException("filePaths[" + idx + "] is empty.");
|
||||
}
|
||||
Path resolved = validate(raw, idx);
|
||||
String content;
|
||||
try {
|
||||
totalBytes += Files.size(resolved);
|
||||
content = Files.readString(resolved, StandardCharsets.UTF_8);
|
||||
} catch (Exception e) {
|
||||
throw new ResolveException(
|
||||
"filePaths[" + idx + "] read failed — " + e.getMessage());
|
||||
}
|
||||
if (content.isBlank()) {
|
||||
throw new ResolveException("filePaths[" + idx + "] is blank " + resolved);
|
||||
}
|
||||
if (combined.length() > 0) combined.append("\n\n");
|
||||
combined.append(content);
|
||||
resolvedPaths.add(resolved);
|
||||
}
|
||||
return new Resolved(combined.toString(), List.copyOf(resolvedPaths), totalBytes);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve and validate a single path. {@code idx >= 0} formats errors as
|
||||
* {@code filePaths[idx]: ...} for the multi-file caller; {@code idx < 0}
|
||||
* uses the bare message form for the single-file caller.
|
||||
*/
|
||||
private static Path validate(String raw, int idx) throws ResolveException {
|
||||
Path resolved;
|
||||
try {
|
||||
resolved = WorkspacePathGuard.validatePath(raw);
|
||||
} catch (Exception e) {
|
||||
throw new ResolveException(prefix(idx) + "path validation failed — " + e.getMessage());
|
||||
}
|
||||
if (!Files.exists(resolved)) {
|
||||
throw new ResolveException(prefix(idx) + "file not found at " + resolved);
|
||||
}
|
||||
if (!Files.isRegularFile(resolved) || !Files.isReadable(resolved)) {
|
||||
throw new ResolveException(prefix(idx) + "path is not a readable regular file " + resolved);
|
||||
}
|
||||
return resolved;
|
||||
}
|
||||
|
||||
private static String prefix(int idx) {
|
||||
return idx < 0 ? "" : "filePaths[" + idx + "] ";
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,206 @@
|
||||
package vip.mate.tool.document;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.apache.poi.xslf.usermodel.XMLSlideShow;
|
||||
import org.apache.poi.xslf.usermodel.XSLFSlide;
|
||||
import org.apache.poi.xslf.usermodel.XSLFTextBox;
|
||||
import org.apache.poi.xslf.usermodel.XSLFTextParagraph;
|
||||
import org.apache.poi.xslf.usermodel.XSLFTextRun;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.awt.Dimension;
|
||||
import java.awt.Rectangle;
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Render a Markdown string into a PowerPoint .pptx byte array using Apache POI.
|
||||
*
|
||||
* <p>Convention (Marp-compatible subset):
|
||||
* <ul>
|
||||
* <li>{@code ---} on its own line separates slides.</li>
|
||||
* <li>The first {@code # / ## / ###} of a slide becomes the slide title.</li>
|
||||
* <li>Lines starting with {@code - } or {@code * } become bullets.</li>
|
||||
* <li>Other non-blank lines become plain paragraphs.</li>
|
||||
* <li>{@code <!-- ... -->} HTML comments become speaker notes.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Page size: 16:9 widescreen by default (960pt x 540pt). Pass
|
||||
* {@code "4:3"} or {@code "STANDARD"} to {@link #render(String, String)} for
|
||||
* legacy 4:3 (720pt x 540pt).
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
public class MarkdownPptxRenderer {
|
||||
|
||||
/** {@code ---} alone on a line separates slides (Marp / commonmark thematic break). */
|
||||
private static final Pattern SLIDE_BREAK = Pattern.compile("^-{3,}\\s*$");
|
||||
|
||||
/** {@code # / ## / ###} title at the start of a slide. */
|
||||
private static final Pattern HEADING = Pattern.compile("^(#{1,3})\\s+(.+)$");
|
||||
|
||||
/** Bullet item: {@code - foo} or {@code * foo}. */
|
||||
private static final Pattern BULLET = Pattern.compile("^\\s*[-*]\\s+(.*)$");
|
||||
|
||||
/** Speaker note marker: {@code <!-- text -->}. */
|
||||
private static final Pattern SPEAKER_NOTE = Pattern.compile("^<!--\\s*(.*?)\\s*-->\\s*$");
|
||||
|
||||
private static final double TITLE_FONT_SIZE = 32.0;
|
||||
private static final double BULLET_FONT_SIZE = 20.0;
|
||||
private static final double PARAGRAPH_FONT_SIZE = 18.0;
|
||||
|
||||
public byte[] render(String markdown, String aspectRatio) throws IOException {
|
||||
if (markdown == null) markdown = "";
|
||||
|
||||
try (XMLSlideShow ppt = new XMLSlideShow();
|
||||
ByteArrayOutputStream baos = new ByteArrayOutputStream()) {
|
||||
|
||||
ppt.setPageSize(resolvePageSize(aspectRatio));
|
||||
|
||||
List<SlideSpec> slides = parseSlides(markdown);
|
||||
if (slides.isEmpty()) {
|
||||
// Always produce at least one slide so the file is openable.
|
||||
slides.add(new SlideSpec(null, List.of(), null));
|
||||
}
|
||||
|
||||
int width = (int) ppt.getPageSize().getWidth();
|
||||
int height = (int) ppt.getPageSize().getHeight();
|
||||
for (SlideSpec spec : slides) {
|
||||
writeSlide(ppt, spec, width, height);
|
||||
}
|
||||
|
||||
ppt.write(baos);
|
||||
return baos.toByteArray();
|
||||
}
|
||||
}
|
||||
|
||||
private record SlideSpec(String title, List<BodyLine> body, String speakerNote) {}
|
||||
|
||||
private record BodyLine(boolean bullet, String text) {}
|
||||
|
||||
private List<SlideSpec> parseSlides(String markdown) {
|
||||
List<SlideSpec> result = new ArrayList<>();
|
||||
String[] lines = markdown.split("\\R", -1);
|
||||
|
||||
String currentTitle = null;
|
||||
List<BodyLine> currentBody = new ArrayList<>();
|
||||
StringBuilder currentNote = new StringBuilder();
|
||||
|
||||
for (String rawLine : lines) {
|
||||
String line = rawLine.strip();
|
||||
if (SLIDE_BREAK.matcher(line).matches()) {
|
||||
if (currentTitle != null || !currentBody.isEmpty() || currentNote.length() > 0) {
|
||||
result.add(new SlideSpec(
|
||||
currentTitle, currentBody,
|
||||
currentNote.length() == 0 ? null : currentNote.toString().strip()));
|
||||
}
|
||||
currentTitle = null;
|
||||
currentBody = new ArrayList<>();
|
||||
currentNote = new StringBuilder();
|
||||
continue;
|
||||
}
|
||||
|
||||
var noteMatch = SPEAKER_NOTE.matcher(line);
|
||||
if (noteMatch.matches()) {
|
||||
if (currentNote.length() > 0) currentNote.append('\n');
|
||||
currentNote.append(noteMatch.group(1));
|
||||
continue;
|
||||
}
|
||||
|
||||
if (line.isEmpty()) {
|
||||
if (!currentBody.isEmpty()) {
|
||||
currentBody.add(new BodyLine(false, ""));
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
var headingMatch = HEADING.matcher(line);
|
||||
if (headingMatch.matches() && currentTitle == null && currentBody.isEmpty()) {
|
||||
currentTitle = headingMatch.group(2).strip();
|
||||
continue;
|
||||
}
|
||||
|
||||
var bulletMatch = BULLET.matcher(line);
|
||||
if (bulletMatch.matches()) {
|
||||
currentBody.add(new BodyLine(true, bulletMatch.group(1).strip()));
|
||||
continue;
|
||||
}
|
||||
|
||||
currentBody.add(new BodyLine(false, line));
|
||||
}
|
||||
|
||||
if (currentTitle != null || !currentBody.isEmpty() || currentNote.length() > 0) {
|
||||
result.add(new SlideSpec(
|
||||
currentTitle, currentBody,
|
||||
currentNote.length() == 0 ? null : currentNote.toString().strip()));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
private void writeSlide(XMLSlideShow ppt, SlideSpec spec, int slideW, int slideH) {
|
||||
XSLFSlide slide = ppt.createSlide();
|
||||
|
||||
int margin = 48;
|
||||
int titleY = 36;
|
||||
int titleH = spec.title() != null ? 80 : 0;
|
||||
int bodyY = titleY + (titleH > 0 ? titleH + 12 : 0);
|
||||
int bodyH = slideH - bodyY - margin;
|
||||
|
||||
if (spec.title() != null) {
|
||||
XSLFTextBox titleBox = slide.createTextBox();
|
||||
titleBox.setAnchor(new Rectangle(margin, titleY, slideW - margin * 2, titleH));
|
||||
// POI creates text boxes with one empty paragraph + run; reuse it for the title.
|
||||
XSLFTextParagraph titleP = titleBox.getTextParagraphs().get(0);
|
||||
XSLFTextRun titleR = titleP.getTextRuns().isEmpty()
|
||||
? titleP.addNewTextRun()
|
||||
: titleP.getTextRuns().get(0);
|
||||
titleR.setText(spec.title());
|
||||
titleR.setFontSize(TITLE_FONT_SIZE);
|
||||
titleR.setBold(true);
|
||||
}
|
||||
|
||||
if (!spec.body().isEmpty()) {
|
||||
XSLFTextBox bodyBox = slide.createTextBox();
|
||||
bodyBox.setAnchor(new Rectangle(margin + 12, bodyY, slideW - margin * 2 - 12, bodyH));
|
||||
// Drop the default empty paragraph so our first body line lines up at the top.
|
||||
bodyBox.clearText();
|
||||
|
||||
for (BodyLine bl : spec.body()) {
|
||||
XSLFTextParagraph p = bodyBox.addNewTextParagraph();
|
||||
if (bl.bullet()) {
|
||||
p.setBullet(true);
|
||||
p.setIndentLevel(0);
|
||||
}
|
||||
XSLFTextRun r = p.addNewTextRun();
|
||||
r.setText(bl.text());
|
||||
r.setFontSize(bl.bullet() ? BULLET_FONT_SIZE : PARAGRAPH_FONT_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
if (spec.speakerNote() != null && !spec.speakerNote().isBlank()) {
|
||||
try {
|
||||
slide.getNotes().getPlaceholder(0).setText(spec.speakerNote());
|
||||
} catch (Exception e) {
|
||||
log.debug("Failed to attach speaker note: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a user-supplied aspect-ratio string to a POI {@link Dimension}
|
||||
* in points. The default (and value for any unrecognized input) is 16:9.
|
||||
*/
|
||||
private Dimension resolvePageSize(String aspectRatio) {
|
||||
if (aspectRatio == null) return new Dimension(960, 540);
|
||||
String normalized = aspectRatio.trim().toUpperCase(Locale.ROOT);
|
||||
return switch (normalized) {
|
||||
case "4:3", "STANDARD" -> new Dimension(720, 540);
|
||||
case "16:9", "WIDE", "WIDESCREEN", "" -> new Dimension(960, 540);
|
||||
default -> new Dimension(960, 540);
|
||||
};
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,234 @@
|
||||
package vip.mate.tool.document;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.apache.poi.ss.usermodel.BorderStyle;
|
||||
import org.apache.poi.ss.usermodel.Cell;
|
||||
import org.apache.poi.ss.usermodel.CellStyle;
|
||||
import org.apache.poi.ss.usermodel.FillPatternType;
|
||||
import org.apache.poi.ss.usermodel.Font;
|
||||
import org.apache.poi.ss.usermodel.HorizontalAlignment;
|
||||
import org.apache.poi.ss.usermodel.IndexedColors;
|
||||
import org.apache.poi.ss.usermodel.Row;
|
||||
import org.apache.poi.ss.usermodel.Sheet;
|
||||
import org.apache.poi.xssf.usermodel.XSSFWorkbook;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Render a Markdown string into an Excel .xlsx byte array using Apache POI.
|
||||
*
|
||||
* <p>Convention: each ATX H1 ({@code # Sheet Name}) starts a new sheet. The
|
||||
* pipe-style table that follows becomes the sheet body. The first table row
|
||||
* is treated as the header (bold, light-grey fill, frozen). Numeric-looking
|
||||
* cells are stored as numbers; everything else is stored as a string.
|
||||
*
|
||||
* <p>Markdown without an explicit {@code # heading} produces a single sheet
|
||||
* named {@code Sheet1}. Markdown without any {@code | table |} rows produces
|
||||
* an empty workbook with one blank sheet (rendering still succeeds).
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
public class MarkdownXlsxRenderer {
|
||||
|
||||
/** Detects the markdown table separator row, e.g. {@code | --- | :---: |}. */
|
||||
private static final Pattern TABLE_SEPARATOR =
|
||||
Pattern.compile("^\\s*\\|?\\s*:?-{3,}:?\\s*(\\|\\s*:?-{3,}:?\\s*)+\\|?\\s*$");
|
||||
|
||||
/** Detects a sheet boundary {@code # Sheet Name}. ## / ### are NOT boundaries. */
|
||||
private static final Pattern SHEET_BOUNDARY = Pattern.compile("^#\\s+(.+)$");
|
||||
|
||||
/** Cells that look like numbers (optional sign, digits, optional decimal). */
|
||||
private static final Pattern NUMERIC = Pattern.compile("^-?\\d+(\\.\\d+)?$");
|
||||
|
||||
public byte[] render(String markdown) throws IOException {
|
||||
if (markdown == null) markdown = "";
|
||||
|
||||
try (XSSFWorkbook wb = new XSSFWorkbook();
|
||||
ByteArrayOutputStream baos = new ByteArrayOutputStream()) {
|
||||
|
||||
CellStyle headerStyle = buildHeaderStyle(wb);
|
||||
|
||||
List<SheetSpec> sheets = parseSheets(markdown);
|
||||
if (sheets.isEmpty()) {
|
||||
// Always produce a non-empty workbook so the file is openable.
|
||||
wb.createSheet("Sheet1");
|
||||
} else {
|
||||
// Track names lowercased — Excel sheet uniqueness is
|
||||
// case-insensitive ("Sales" and "sales" collide).
|
||||
Set<String> usedLower = new HashSet<>(sheets.size());
|
||||
int seq = 1;
|
||||
for (SheetSpec spec : sheets) {
|
||||
String safe = sanitizeSheetName(spec.name(), seq++);
|
||||
String unique = uniqueSheetName(safe, usedLower);
|
||||
Sheet sheet = wb.createSheet(unique);
|
||||
writeSheetBody(sheet, spec.rows(), headerStyle);
|
||||
}
|
||||
}
|
||||
|
||||
wb.write(baos);
|
||||
return baos.toByteArray();
|
||||
}
|
||||
}
|
||||
|
||||
private record SheetSpec(String name, List<List<String>> rows) {}
|
||||
|
||||
private List<SheetSpec> parseSheets(String markdown) {
|
||||
List<SheetSpec> sheets = new ArrayList<>();
|
||||
String currentName = null;
|
||||
List<List<String>> currentRows = new ArrayList<>();
|
||||
|
||||
for (String rawLine : markdown.split("\\R", -1)) {
|
||||
String line = rawLine.strip();
|
||||
if (line.isEmpty()) continue;
|
||||
|
||||
var sheetMatch = SHEET_BOUNDARY.matcher(line);
|
||||
if (sheetMatch.matches()) {
|
||||
if (currentName != null || !currentRows.isEmpty()) {
|
||||
sheets.add(new SheetSpec(currentName, currentRows));
|
||||
}
|
||||
currentName = sheetMatch.group(1).strip();
|
||||
currentRows = new ArrayList<>();
|
||||
continue;
|
||||
}
|
||||
|
||||
if (TABLE_SEPARATOR.matcher(line).matches()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Strict markdown-table detection: a row must be wrapped in pipes,
|
||||
// otherwise prose lines like "A | B 是数据库主键" or file paths like
|
||||
// "src/main/java/Foo|Bar" would be silently swallowed into the sheet.
|
||||
// GFM technically allows pipe-less leading/trailing pipes for tables,
|
||||
// but the rendered LLM output overwhelmingly uses the wrapped form,
|
||||
// and being strict avoids false positives that pollute the workbook.
|
||||
if (line.startsWith("|") && line.endsWith("|") && line.length() >= 2) {
|
||||
List<String> cells = splitTableRow(line);
|
||||
if (!cells.isEmpty()) {
|
||||
currentRows.add(cells);
|
||||
}
|
||||
}
|
||||
// Other content (paragraphs, sub-headings) is intentionally ignored —
|
||||
// xlsx is tabular and there is nowhere sensible to render free prose.
|
||||
}
|
||||
|
||||
if (currentName != null || !currentRows.isEmpty()) {
|
||||
sheets.add(new SheetSpec(currentName, currentRows));
|
||||
}
|
||||
return sheets;
|
||||
}
|
||||
|
||||
private List<String> splitTableRow(String line) {
|
||||
String trimmed = line.strip();
|
||||
if (trimmed.startsWith("|")) trimmed = trimmed.substring(1);
|
||||
if (trimmed.endsWith("|")) trimmed = trimmed.substring(0, trimmed.length() - 1);
|
||||
String[] parts = trimmed.split("\\|", -1);
|
||||
List<String> cells = new ArrayList<>(parts.length);
|
||||
for (String p : parts) cells.add(p.strip());
|
||||
return cells;
|
||||
}
|
||||
|
||||
private void writeSheetBody(Sheet sheet, List<List<String>> rows, CellStyle headerStyle) {
|
||||
if (rows.isEmpty()) return;
|
||||
|
||||
int maxCols = 0;
|
||||
for (int r = 0; r < rows.size(); r++) {
|
||||
List<String> rowData = rows.get(r);
|
||||
Row row = sheet.createRow(r);
|
||||
for (int c = 0; c < rowData.size(); c++) {
|
||||
Cell cell = row.createCell(c);
|
||||
String value = rowData.get(c);
|
||||
if (NUMERIC.matcher(value).matches()) {
|
||||
cell.setCellValue(Double.parseDouble(value));
|
||||
} else {
|
||||
cell.setCellValue(value);
|
||||
}
|
||||
if (r == 0) cell.setCellStyle(headerStyle);
|
||||
}
|
||||
if (rowData.size() > maxCols) maxCols = rowData.size();
|
||||
}
|
||||
|
||||
// Freeze the header row and auto-size columns. autoSizeColumn is O(n*m)
|
||||
// but agent-generated workbooks are small, so the cost is negligible.
|
||||
sheet.createFreezePane(0, 1);
|
||||
for (int c = 0; c < maxCols; c++) {
|
||||
try {
|
||||
sheet.autoSizeColumn(c);
|
||||
} catch (Exception e) {
|
||||
log.debug("autoSizeColumn({}) failed (likely missing fonts on a headless host): {}",
|
||||
c, e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private CellStyle buildHeaderStyle(XSSFWorkbook wb) {
|
||||
CellStyle style = wb.createCellStyle();
|
||||
Font font = wb.createFont();
|
||||
font.setBold(true);
|
||||
style.setFont(font);
|
||||
style.setFillForegroundColor(IndexedColors.GREY_25_PERCENT.getIndex());
|
||||
style.setFillPattern(FillPatternType.SOLID_FOREGROUND);
|
||||
style.setAlignment(HorizontalAlignment.LEFT);
|
||||
style.setBorderBottom(BorderStyle.THIN);
|
||||
return style;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve duplicate sheet names by appending {@code (2)}, {@code (3)}…
|
||||
* within the 31-char Excel limit. POI throws on collision, which would
|
||||
* otherwise abort the entire render when an LLM emits two sheets with the
|
||||
* same heading or two long headings whose first 31 chars happen to match.
|
||||
*
|
||||
* <p>Excel sheet uniqueness is case-INsensitive, so {@code "Sales"} and
|
||||
* {@code "sales"} collide. We track names lowercased while still passing
|
||||
* the original casing into {@link Sheet#createSheet(String)} — so the
|
||||
* displayed tab keeps the user's casing.
|
||||
*/
|
||||
private String uniqueSheetName(String candidate, Set<String> usedLower) {
|
||||
if (usedLower.add(candidate.toLowerCase(Locale.ROOT))) return candidate;
|
||||
for (int i = 2; i < 1000; i++) {
|
||||
String suffix = " (" + i + ")";
|
||||
int maxBase = 31 - suffix.length();
|
||||
String base = candidate.length() > maxBase
|
||||
? candidate.substring(0, maxBase)
|
||||
: candidate;
|
||||
String trial = base + suffix;
|
||||
if (usedLower.add(trial.toLowerCase(Locale.ROOT))) return trial;
|
||||
}
|
||||
// Pathological: 1000 collisions. Fall back to a guaranteed-unique tag
|
||||
// built from nanoTime so the render still succeeds.
|
||||
String fallback = ("Sheet_" + System.nanoTime());
|
||||
if (fallback.length() > 31) fallback = fallback.substring(0, 31);
|
||||
usedLower.add(fallback.toLowerCase(Locale.ROOT));
|
||||
return fallback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Excel sheet names are limited to 31 chars and cannot contain {@code : / \ ? * [ ]},
|
||||
* cannot be blank, and must be unique. Uniqueness is enforced separately by
|
||||
* {@link #uniqueSheetName(String, Set)} so this method stays single-shot.
|
||||
*/
|
||||
private String sanitizeSheetName(String raw, int seq) {
|
||||
if (raw == null || raw.isBlank()) return "Sheet" + seq;
|
||||
StringBuilder sb = new StringBuilder(raw.length());
|
||||
for (char ch : raw.toCharArray()) {
|
||||
if (ch == ':' || ch == '/' || ch == '\\' || ch == '?'
|
||||
|| ch == '*' || ch == '[' || ch == ']') {
|
||||
sb.append('_');
|
||||
} else {
|
||||
sb.append(ch);
|
||||
}
|
||||
}
|
||||
String cleaned = sb.toString().strip();
|
||||
if (cleaned.isEmpty()) cleaned = "Sheet" + seq;
|
||||
if (cleaned.length() > 31) cleaned = cleaned.substring(0, 31);
|
||||
return cleaned;
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,112 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.Paths;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Locate a font file capable of rendering CJK text for {@link OpenHtmlToPdfBackend}.
|
||||
*
|
||||
* <p>OpenHTMLtoPDF renders any glyph the registered font does not cover as a
|
||||
* blank {@code .notdef} box, so a CJK-capable font is mandatory whenever the
|
||||
* markdown contains Chinese, Japanese, or Korean text. We try in this order:
|
||||
* <ol>
|
||||
* <li>An explicit {@code mateclaw.pdf.font-path} configuration value.</li>
|
||||
* <li>A short list of OS-default paths that ship with macOS / Windows / common
|
||||
* Linux distributions. The first existing file wins.</li>
|
||||
* <li>{@link Optional#empty()} — the renderer falls back to PDFBox's built-in
|
||||
* Latin-only fonts, which renders Chinese as boxes; logged as a warning.</li>
|
||||
* </ol>
|
||||
*/
|
||||
@Slf4j
|
||||
public final class CjkFontResolver {
|
||||
|
||||
// .ttf candidates are listed FIRST because OpenPDF 2.0.5 (used by the
|
||||
// FlyingSaucer PDF backend) cannot reliably read Apple-style .ttc font
|
||||
// collections — it loads them without throwing, but the resulting
|
||||
// BaseFont has an empty cmap and reports `charExists` as false even for
|
||||
// ASCII. The PDF then renders as a blank page. .ttf collections do not
|
||||
// share that limitation, so we try them first and only fall through to
|
||||
// .ttc when nothing else is available. The runtime charExists check in
|
||||
// FlyingSaucerPdfBackend will reject any candidate that loads but
|
||||
// cannot actually render glyphs.
|
||||
|
||||
private static final String USER_HOME = System.getProperty("user.home", "");
|
||||
|
||||
private static final List<String> CANDIDATES_MACOS = List.of(
|
||||
// Popular open-source CJK .ttf fonts that users commonly install
|
||||
USER_HOME + "/Library/Fonts/HarmonyOS_SansSC_Regular.ttf",
|
||||
"/Library/Fonts/HarmonyOS_SansSC_Regular.ttf",
|
||||
USER_HOME + "/Library/Fonts/SourceHanSansSC-Regular.otf",
|
||||
"/Library/Fonts/SourceHanSansSC-Regular.otf",
|
||||
USER_HOME + "/Library/Fonts/NotoSansSC-Regular.ttf",
|
||||
"/Library/Fonts/NotoSansSC-Regular.ttf",
|
||||
USER_HOME + "/Library/Fonts/Arial Unicode.ttf",
|
||||
"/Library/Fonts/Arial Unicode.ttf",
|
||||
// .ttc fallbacks — known to be lossy under OpenPDF on macOS,
|
||||
// but listed so the resolver can still warn about them.
|
||||
"/System/Library/Fonts/PingFang.ttc",
|
||||
"/System/Library/Fonts/STHeiti Light.ttc",
|
||||
"/System/Library/Fonts/STHeiti Medium.ttc",
|
||||
"/Library/Fonts/Songti.ttc");
|
||||
|
||||
private static final List<String> CANDIDATES_WINDOWS = List.of(
|
||||
// Plain .ttf first, .ttc / .otf later
|
||||
"C:/Windows/Fonts/msyh.ttf",
|
||||
"C:/Windows/Fonts/simhei.ttf", // 黑体
|
||||
"C:/Windows/Fonts/simsun.ttf",
|
||||
"C:/Windows/Fonts/HarmonyOS_SansSC_Regular.ttf",
|
||||
"C:/Windows/Fonts/NotoSansSC-Regular.ttf",
|
||||
// Collections last
|
||||
"C:/Windows/Fonts/msyh.ttc", // 微软雅黑
|
||||
"C:/Windows/Fonts/simsun.ttc"); // 宋体
|
||||
|
||||
private static final List<String> CANDIDATES_LINUX = List.of(
|
||||
// Plain .ttf / .otf first
|
||||
"/usr/share/fonts/opentype/noto/NotoSansCJKsc-Regular.otf",
|
||||
"/usr/share/fonts/truetype/noto/NotoSansSC-Regular.ttf",
|
||||
"/usr/share/fonts/truetype/harmonyos-sans/HarmonyOS_SansSC_Regular.ttf",
|
||||
"/usr/share/fonts/truetype/source-han-sans/SourceHanSansSC-Regular.otf",
|
||||
// Collections last
|
||||
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
|
||||
"/usr/share/fonts/truetype/wqy/wqy-microhei.ttc",
|
||||
"/usr/share/fonts/truetype/wqy/wqy-zenhei.ttc",
|
||||
"/usr/share/fonts/truetype/arphic/uming.ttc",
|
||||
"/usr/share/fonts/truetype/arphic/ukai.ttc");
|
||||
|
||||
private CjkFontResolver() {}
|
||||
|
||||
public static Optional<Path> resolve(String configuredPath) {
|
||||
if (configuredPath != null && !configuredPath.isBlank()) {
|
||||
Path explicit = Paths.get(configuredPath.trim());
|
||||
if (Files.isRegularFile(explicit)) {
|
||||
log.debug("[CjkFont] using configured font: {}", explicit);
|
||||
return Optional.of(explicit);
|
||||
}
|
||||
log.warn("[CjkFont] configured font path does not exist: {}", explicit);
|
||||
}
|
||||
|
||||
for (String candidate : candidatesForCurrentOs()) {
|
||||
Path p = Paths.get(candidate);
|
||||
if (Files.isRegularFile(p)) {
|
||||
log.debug("[CjkFont] auto-detected system font: {}", p);
|
||||
return Optional.of(p);
|
||||
}
|
||||
}
|
||||
log.warn("[CjkFont] no CJK font found on this host; PDF Chinese characters "
|
||||
+ "will render as blank boxes. Set mateclaw.pdf.font-path to override.");
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
private static List<String> candidatesForCurrentOs() {
|
||||
String osName = System.getProperty("os.name", "").toLowerCase(Locale.ROOT);
|
||||
if (osName.contains("mac")) return CANDIDATES_MACOS;
|
||||
if (osName.contains("win")) return CANDIDATES_WINDOWS;
|
||||
return CANDIDATES_LINUX;
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,377 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
import com.lowagie.text.pdf.BaseFont;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.commonmark.ext.autolink.AutolinkExtension;
|
||||
import org.commonmark.ext.front.matter.YamlFrontMatterExtension;
|
||||
import org.commonmark.ext.gfm.strikethrough.StrikethroughExtension;
|
||||
import org.commonmark.ext.gfm.tables.TablesExtension;
|
||||
import org.commonmark.node.Node;
|
||||
import org.commonmark.parser.Parser;
|
||||
import org.commonmark.renderer.html.HtmlRenderer;
|
||||
import org.springframework.stereotype.Component;
|
||||
import org.xhtmlrenderer.pdf.ITextFontResolver;
|
||||
import org.xhtmlrenderer.pdf.ITextRenderer;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* In-process PDF rendering: markdown → flexmark XHTML → Flying Saucer (XHTMLRenderer)
|
||||
* → OpenPDF.
|
||||
*
|
||||
* <p>This backend is always available and is the only one that supports cover
|
||||
* pages, page headers, and page footers (driven by YAML frontmatter; see
|
||||
* {@link PdfFrontmatter}). It uses CSS3 paged-media features that Flying Saucer
|
||||
* implements: {@code @page}, {@code counter(page)}, {@code counter(pages)},
|
||||
* {@code @top-center}, {@code @bottom-center}, and {@code page-break-before}.
|
||||
*
|
||||
* <p>Flying Saucer requires strict XHTML, so flexmark's HTML output is wrapped
|
||||
* in an XHTML envelope. Self-closing void elements ({@code <br>}, {@code <hr>},
|
||||
* {@code <img>}) are normalised by flexmark when generating the body, so we do
|
||||
* not need a post-processor.
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class FlyingSaucerPdfBackend implements PdfBackend {
|
||||
|
||||
private final PdfProperties properties;
|
||||
|
||||
@Override
|
||||
public String name() { return "flying-saucer"; }
|
||||
|
||||
@Override
|
||||
public byte[] render(PdfRenderRequest request) throws Exception {
|
||||
String bodyHtml = renderMarkdownToHtml(request.markdown());
|
||||
|
||||
try (ByteArrayOutputStream baos = new ByteArrayOutputStream()) {
|
||||
ITextRenderer renderer = new ITextRenderer();
|
||||
// Register the CJK font BEFORE building the HTML, because the CSS we
|
||||
// emit references the font's actual family name (read from the font
|
||||
// file). Aliases via ITextFontResolver's 5-arg overload proved
|
||||
// unreliable on .ttc collections: the API accepts the override but
|
||||
// the lookup map silently misses it, leaving the body to fall back
|
||||
// to Times-Roman and Chinese to render as .notdef boxes.
|
||||
String cjkFamily = registerCjkFont(renderer.getFontResolver());
|
||||
String fullHtml = wrapHtml(bodyHtml, request, cjkFamily);
|
||||
log.debug("[FlyingSaucerPdf] HTML length={}, body length={}, cjkFamily={}",
|
||||
fullHtml.length(), bodyHtml.length(), cjkFamily);
|
||||
try {
|
||||
renderer.setDocumentFromString(fullHtml);
|
||||
renderer.layout();
|
||||
renderer.createPDF(baos);
|
||||
} catch (Throwable t) {
|
||||
log.error("[FlyingSaucerPdf] ITextRenderer failed: {}: {}",
|
||||
t.getClass().getName(), t.getMessage(), t);
|
||||
throw t;
|
||||
}
|
||||
return baos.toByteArray();
|
||||
}
|
||||
}
|
||||
|
||||
private String renderMarkdownToHtml(String markdown) {
|
||||
List<org.commonmark.Extension> extensions = List.of(
|
||||
TablesExtension.create(),
|
||||
StrikethroughExtension.create(),
|
||||
AutolinkExtension.create(),
|
||||
YamlFrontMatterExtension.create());
|
||||
Parser parser = Parser.builder().extensions(extensions).build();
|
||||
// Flying Saucer requires strict XHTML, so void elements (<br>, <hr>,
|
||||
// <img>) must be self-closed. The xhtml renderer flavour does this.
|
||||
HtmlRenderer renderer = HtmlRenderer.builder()
|
||||
.extensions(extensions)
|
||||
.build();
|
||||
Node document = parser.parse(markdown);
|
||||
return renderer.render(document);
|
||||
}
|
||||
|
||||
/**
|
||||
* Register the resolved CJK font with Flying Saucer and return the
|
||||
* font's actual {@code font-family} name so the inline stylesheet can
|
||||
* reference it. Returns {@code null} if no font was found or the
|
||||
* registration failed — callers must tolerate Chinese rendering as
|
||||
* blank boxes in that case.
|
||||
*
|
||||
* <p>Why we read the real family name instead of using the 5-arg
|
||||
* {@code addFont(... fontFamilyNameOverride ...)} overload: that override
|
||||
* succeeds in the call but does not get added to the renderer's
|
||||
* {@code _fontFamilies} lookup map for {@code .ttc} collections, so the
|
||||
* CSS declaration {@code font-family: "CJK"} still misses and the body
|
||||
* falls back to Times-Roman. Reading the font's intrinsic family name
|
||||
* via OpenPDF's {@link BaseFont#getFamilyFontName()} sidesteps that
|
||||
* map entirely.
|
||||
*/
|
||||
private String registerCjkFont(ITextFontResolver fonts) {
|
||||
Optional<Path> fontPath = CjkFontResolver.resolve(properties.fontPath());
|
||||
if (fontPath.isEmpty()) {
|
||||
log.error("[FlyingSaucerPdf] No CJK font registered. Chinese characters "
|
||||
+ "in this PDF will render as blank boxes. Set mateclaw.pdf.font-path "
|
||||
+ "to the absolute path of a CJK-capable .ttf / .ttc / .otf file.");
|
||||
return null;
|
||||
}
|
||||
// BaseFont.IDENTITY_H + EMBEDDED is what makes CJK actually appear in
|
||||
// the output PDF — without IDENTITY_H glyph indexing, Chinese characters
|
||||
// render as blanks even when the font file is found.
|
||||
//
|
||||
// OpenPDF 2.0.5 has a known weakness with Apple-style .ttc font
|
||||
// collections (PingFang.ttc, STHeiti.ttc, Songti.ttc on macOS): the
|
||||
// load succeeds but the cmap is empty, charExists returns false even
|
||||
// for ASCII, and the rendered PDF is a blank page. We probe the font
|
||||
// with charExists below; if it cannot render the characters we need,
|
||||
// we DO NOT register it and return null so the document keeps
|
||||
// falling back to the next family in the CSS chain.
|
||||
String fontKey = fontFileWithSubfontIndex(fontPath.get());
|
||||
BaseFont probe;
|
||||
try {
|
||||
probe = BaseFont.createFont(fontKey, BaseFont.IDENTITY_H, BaseFont.EMBEDDED);
|
||||
} catch (Throwable t) {
|
||||
log.error("[FlyingSaucerPdf] BaseFont.createFont failed for {} — Chinese "
|
||||
+ "will render as blank boxes. {}: {}",
|
||||
fontKey, t.getClass().getSimpleName(), t.getMessage());
|
||||
return null;
|
||||
}
|
||||
if (!probe.charExists('你') || !probe.charExists('A')) {
|
||||
log.error("[FlyingSaucerPdf] Font {} loaded but cmap is empty "
|
||||
+ "(charExists '你'={} 'A'={}). This is the known OpenPDF Apple-.ttc "
|
||||
+ "limitation — install a .ttf CJK font (e.g. HarmonyOS Sans SC, "
|
||||
+ "Noto Sans SC) and either drop it under ~/Library/Fonts/ or set "
|
||||
+ "mateclaw.pdf.font-path to its absolute path.",
|
||||
fontKey, probe.charExists('你'), probe.charExists('A'));
|
||||
return null;
|
||||
}
|
||||
String realFamily = readFamilyName(probe, fontKey);
|
||||
try {
|
||||
fonts.addFont(fontKey, BaseFont.IDENTITY_H, BaseFont.EMBEDDED);
|
||||
log.info("[FlyingSaucerPdf] registered CJK font: {} (family=\"{}\", cmap OK)",
|
||||
fontKey, realFamily);
|
||||
return realFamily;
|
||||
} catch (Exception e) {
|
||||
log.error("[FlyingSaucerPdf] failed to register CJK font {} — Chinese "
|
||||
+ "characters in this PDF will render as blank boxes. {}: {}",
|
||||
fontKey, e.getClass().getSimpleName(), e.getMessage());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pull a usable family name out of the loaded font. Some fonts
|
||||
* (HarmonyOS Sans SC) leave {@code getFamilyFontName} empty and
|
||||
* carry the name only in {@code getPostscriptFontName}, so we fall
|
||||
* back to that.
|
||||
*/
|
||||
private static String readFamilyName(BaseFont probe, String fontKey) {
|
||||
try {
|
||||
String[][] familyNames = probe.getFamilyFontName();
|
||||
if (familyNames != null && familyNames.length > 0) {
|
||||
String fallback = null;
|
||||
for (String[] row : familyNames) {
|
||||
if (row == null || row.length < 4 || row[3] == null || row[3].isBlank()) continue;
|
||||
if (fallback == null) fallback = row[3];
|
||||
if ("3".equals(row[0]) && "1033".equals(row[2])) {
|
||||
return row[3];
|
||||
}
|
||||
}
|
||||
if (fallback != null) return fallback;
|
||||
}
|
||||
String psName = probe.getPostscriptFontName();
|
||||
if (psName != null && !psName.isBlank()) return psName;
|
||||
} catch (Throwable t) {
|
||||
log.warn("[FlyingSaucerPdf] could not read family name from {}: {}",
|
||||
fontKey, t.getMessage());
|
||||
}
|
||||
return "Helvetica"; // benign fallback
|
||||
}
|
||||
|
||||
private static String fontFileWithSubfontIndex(Path path) {
|
||||
String name = path.getFileName().toString().toLowerCase(Locale.ROOT);
|
||||
if (name.endsWith(".ttc") || name.endsWith(".otc")) {
|
||||
return path.toString() + ",0";
|
||||
}
|
||||
return path.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap the rendered markdown body in an XHTML envelope plus a CSS @page
|
||||
* stylesheet that drives cover / header / footer / page numbers.
|
||||
*
|
||||
* @param cjkFamily the actual family name of the registered CJK font as
|
||||
* reported by OpenPDF, or {@code null} if no font was
|
||||
* registered. Injected verbatim into the body
|
||||
* {@code font-family} declaration; when absent we fall
|
||||
* through directly to Helvetica.
|
||||
*/
|
||||
private String wrapHtml(String bodyHtml, PdfRenderRequest request, String cjkFamily) {
|
||||
PdfFrontmatter fm = request.frontmatter();
|
||||
String pageSize = request.pageSize();
|
||||
String cjkFamilyDecl = cjkFamily == null
|
||||
? ""
|
||||
: "\"" + cssEscape(cjkFamily) + "\", ";
|
||||
|
||||
// Only render a real cover page when the user explicitly asked for one
|
||||
// via YAML frontmatter. A synthesised cover (H1 promoted into title)
|
||||
// would otherwise duplicate the heading: once on the cover and again
|
||||
// as the first body H1.
|
||||
String coverHtml = fm.hasExplicitCover()
|
||||
? "<div class=\"cover\">"
|
||||
+ "<h1 class=\"cover-title\">" + escape(fm.title()) + "</h1>"
|
||||
+ (fm.subtitleOpt().isPresent()
|
||||
? "<p class=\"cover-subtitle\">" + escape(fm.subtitle()) + "</p>"
|
||||
: "")
|
||||
+ "</div>"
|
||||
: "";
|
||||
|
||||
// Page margin boxes do NOT inherit `font-family` from body — Flying
|
||||
// Saucer treats them as detached generated content boxes. If we don't
|
||||
// give them a CJK-capable font here, header/footer Chinese characters
|
||||
// silently drop ("Tech Daily · 每日科技精选" → "Tech Daily ·") because
|
||||
// the default Helvetica has no CJK glyphs. We thread the same family
|
||||
// we registered for body text through here so the rendering is
|
||||
// consistent across the document.
|
||||
String marginBoxFontDecl = "font-family: " + cjkFamilyDecl
|
||||
+ "\"Helvetica\", sans-serif; font-size: 9pt; color: #888;";
|
||||
String headerCss = fm.hasHeader()
|
||||
? "@top-center { content: \"" + cssEscape(fm.header()) + "\"; "
|
||||
+ marginBoxFontDecl + " }"
|
||||
: "";
|
||||
String footerCss = "@bottom-center { content: " + footerContent(fm)
|
||||
+ "; " + marginBoxFontDecl + " }";
|
||||
|
||||
// No <!DOCTYPE>: Flying Saucer's default EntityResolver tries to fetch
|
||||
// the W3C XHTML DTD over the network during setDocumentFromString().
|
||||
// On any host with no internet (or with W3C throttling) the document
|
||||
// load silently fails and we emit a 1.3 KB blank PDF. Plain XHTML
|
||||
// without a DOCTYPE renders just fine.
|
||||
return """
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<html xmlns="http://www.w3.org/1999/xhtml">
|
||||
<head>
|
||||
<title>document</title>
|
||||
<style type="text/css">
|
||||
@page {
|
||||
size: %s;
|
||||
margin: 25mm 20mm 22mm 20mm;
|
||||
%s
|
||||
%s
|
||||
}
|
||||
@page :first {
|
||||
@top-center { content: ""; }
|
||||
@bottom-center { content: ""; }
|
||||
}
|
||||
body {
|
||||
font-family: %s"Helvetica", "Arial", sans-serif;
|
||||
font-size: 11pt;
|
||||
line-height: 1.6;
|
||||
color: #222;
|
||||
}
|
||||
.cover {
|
||||
page-break-after: always;
|
||||
text-align: center;
|
||||
padding-top: 60mm;
|
||||
}
|
||||
.cover-title {
|
||||
font-size: 36pt;
|
||||
margin-bottom: 12mm;
|
||||
font-weight: bold;
|
||||
/* The cover-title element is an h1, so we must explicitly
|
||||
cancel the body-h1 rule below — otherwise its
|
||||
`page-break-before: always` and bottom border would
|
||||
split the cover into a blank leading page plus the
|
||||
title underlined like a section heading. */
|
||||
page-break-before: auto;
|
||||
border-bottom: none;
|
||||
padding-bottom: 0;
|
||||
}
|
||||
.cover-subtitle { font-size: 18pt; color: #666; margin: 0; }
|
||||
h1 {
|
||||
font-size: 22pt;
|
||||
margin: 18pt 0 10pt;
|
||||
page-break-before: always;
|
||||
border-bottom: 1pt solid #ccc;
|
||||
padding-bottom: 4pt;
|
||||
}
|
||||
/* The first H1 inside the body must NOT force a fresh page when
|
||||
we did not render a cover, otherwise the document opens on a
|
||||
blank first page. */
|
||||
.content > h1:first-child { page-break-before: auto; }
|
||||
h2 { font-size: 16pt; margin: 14pt 0 6pt; }
|
||||
h3 { font-size: 13pt; margin: 12pt 0 4pt; color: #333; }
|
||||
p { margin: 6pt 0; }
|
||||
ul, ol { margin: 6pt 0 6pt 18pt; padding: 0; }
|
||||
li { margin: 2pt 0; }
|
||||
table {
|
||||
border-collapse: collapse;
|
||||
width: 100%%;
|
||||
margin: 8pt 0;
|
||||
font-size: 10pt;
|
||||
}
|
||||
th, td {
|
||||
border: 0.5pt solid #888;
|
||||
padding: 4pt 6pt;
|
||||
vertical-align: top;
|
||||
}
|
||||
th { background: #eee; font-weight: bold; text-align: left; }
|
||||
code {
|
||||
font-family: "Menlo", "Consolas", monospace;
|
||||
background: #f6f6f6;
|
||||
padding: 1px 4px;
|
||||
font-size: 10pt;
|
||||
}
|
||||
pre {
|
||||
background: #f6f6f6;
|
||||
padding: 8pt 10pt;
|
||||
border-left: 3pt solid #ccc;
|
||||
font-size: 10pt;
|
||||
white-space: pre-wrap;
|
||||
word-break: break-all;
|
||||
}
|
||||
pre code { background: none; padding: 0; }
|
||||
blockquote {
|
||||
border-left: 3pt solid #ccc;
|
||||
padding: 0 12pt;
|
||||
color: #555;
|
||||
margin: 6pt 0;
|
||||
}
|
||||
hr { border: none; border-top: 1pt solid #ddd; margin: 12pt 0; }
|
||||
a { color: #205493; text-decoration: none; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
%s
|
||||
<div class="content">%s</div>
|
||||
</body>
|
||||
</html>
|
||||
""".formatted(pageSize, headerCss, footerCss, cjkFamilyDecl, coverHtml, bodyHtml);
|
||||
}
|
||||
|
||||
private String footerContent(PdfFrontmatter fm) {
|
||||
// Always show page numbers; concatenate user footer ahead if provided.
|
||||
String pageCounter = "\"" + cssEscape("第 ") + "\" counter(page) "
|
||||
+ "\" / \" counter(pages) \"" + cssEscape(" 页") + "\"";
|
||||
if (fm.hasFooter()) {
|
||||
return "\"" + cssEscape(fm.footer()) + " \" " + pageCounter;
|
||||
}
|
||||
return pageCounter;
|
||||
}
|
||||
|
||||
/** Escape user text for placement inside an HTML element. */
|
||||
private String escape(String s) {
|
||||
if (s == null) return "";
|
||||
return s.replace("&", "&")
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
.replace("\"", """);
|
||||
}
|
||||
|
||||
/** Escape user text for placement inside a CSS string literal. */
|
||||
private String cssEscape(String s) {
|
||||
if (s == null) return "";
|
||||
return s.replace("\\", "\\\\")
|
||||
.replace("\"", "\\\"")
|
||||
.replace("\n", " ");
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,146 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.stereotype.Component;
|
||||
import vip.mate.tool.document.MarkdownDocxRenderer;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Render PDF by routing markdown through {@link MarkdownDocxRenderer} and
|
||||
* then handing the docx to a {@code soffice --convert-to pdf} subprocess.
|
||||
* LibreOffice's typesetter beats anything we can write in-process for plain
|
||||
* narrative text, especially with mixed CJK + Latin scripts, so this is the
|
||||
* preferred path when the local install has it.
|
||||
*
|
||||
* <p>Limitations the orchestrator must respect:
|
||||
* <ul>
|
||||
* <li>The intermediate docx has no first-class cover page, page header,
|
||||
* or page footer the way {@link OpenHtmlToPdfBackend} does. Calls that
|
||||
* want those features go to the HTML path instead — see
|
||||
* {@link #supports(PdfRenderRequest)}.</li>
|
||||
* <li>Page numbers themselves come for free: LibreOffice adds them by
|
||||
* default during PDF export.</li>
|
||||
* </ul>
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class LibreOfficePdfBackend implements PdfBackend {
|
||||
|
||||
private static final long CONVERT_TIMEOUT_SECONDS = 90;
|
||||
|
||||
private final MarkdownDocxRenderer docxRenderer;
|
||||
private final PdfProperties properties;
|
||||
|
||||
@Override
|
||||
public String name() { return "libreoffice"; }
|
||||
|
||||
@Override
|
||||
public boolean isAvailable() {
|
||||
if (!properties.libreoffice().enabled()) return false;
|
||||
try {
|
||||
ProcessBuilder pb = new ProcessBuilder(properties.libreoffice().binary(), "--version");
|
||||
pb.redirectErrorStream(true);
|
||||
Process p = pb.start();
|
||||
// Drain stdout so the child can exit even on systems whose pipe buffers
|
||||
// are tiny; the version string is short, this won't block.
|
||||
p.getInputStream().readAllBytes();
|
||||
boolean finished = p.waitFor(5, TimeUnit.SECONDS);
|
||||
if (!finished) {
|
||||
p.destroyForcibly();
|
||||
return false;
|
||||
}
|
||||
return p.exitValue() == 0;
|
||||
} catch (Exception e) {
|
||||
log.debug("[LibreOfficePdf] soffice probe failed: {}", e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The docx intermediate cannot carry page headers / footers / an explicit
|
||||
* cover page, so we decline requests that need those. The orchestrator
|
||||
* routes such requests to {@link FlyingSaucerPdfBackend} instead.
|
||||
*
|
||||
* <p>Synthetic covers (an H1 that {@code parseOrSynthesise} promoted into a
|
||||
* cover title) are NOT rejected — those would otherwise force AUTO mode to
|
||||
* pick the in-process backend for almost every markdown body, since LLM
|
||||
* output overwhelmingly starts with a {@code # H1}. The H1 will simply
|
||||
* render as the document's first heading, which is what users expect when
|
||||
* they didn't ask for a cover explicitly.
|
||||
*/
|
||||
@Override
|
||||
public boolean supports(PdfRenderRequest request) {
|
||||
PdfFrontmatter fm = request.frontmatter();
|
||||
return !fm.hasExplicitCover() && !fm.hasHeader() && !fm.hasFooter();
|
||||
}
|
||||
|
||||
@Override
|
||||
public byte[] render(PdfRenderRequest request) throws Exception {
|
||||
// Use the same A4/LETTER page-size argument shape MarkdownDocxRenderer expects.
|
||||
byte[] docxBytes = docxRenderer.render(request.markdown(), request.pageSize());
|
||||
|
||||
Path tempDir = Files.createTempDirectory("mc_pdf_");
|
||||
try {
|
||||
Path docxFile = tempDir.resolve("input.docx");
|
||||
Files.write(docxFile, docxBytes);
|
||||
|
||||
ProcessBuilder pb = new ProcessBuilder(
|
||||
properties.libreoffice().binary(),
|
||||
"--headless",
|
||||
"--convert-to", "pdf",
|
||||
"--outdir", tempDir.toString(),
|
||||
docxFile.toString());
|
||||
pb.redirectErrorStream(true);
|
||||
Process p = pb.start();
|
||||
byte[] stderr = p.getInputStream().readAllBytes();
|
||||
boolean finished = p.waitFor(CONVERT_TIMEOUT_SECONDS, TimeUnit.SECONDS);
|
||||
if (!finished) {
|
||||
p.destroyForcibly();
|
||||
throw new IOException("soffice conversion timed out after " + CONVERT_TIMEOUT_SECONDS + "s");
|
||||
}
|
||||
if (p.exitValue() != 0) {
|
||||
throw new IOException("soffice exit " + p.exitValue() + ": "
|
||||
+ new String(stderr).strip());
|
||||
}
|
||||
|
||||
Path pdfFile = tempDir.resolve("input.pdf");
|
||||
if (!Files.isRegularFile(pdfFile)) {
|
||||
throw new IOException("soffice produced no PDF (stderr: "
|
||||
+ new String(stderr).strip() + ")");
|
||||
}
|
||||
return Files.readAllBytes(pdfFile);
|
||||
} finally {
|
||||
cleanup(tempDir);
|
||||
}
|
||||
}
|
||||
|
||||
private void cleanup(Path tempDir) {
|
||||
try (var stream = Files.walk(tempDir)) {
|
||||
List<Path> entries = stream.sorted(Comparator.reverseOrder()).toList();
|
||||
for (Path entry : entries) {
|
||||
try {
|
||||
Files.deleteIfExists(entry);
|
||||
} catch (IOException ignored) {
|
||||
// Best-effort cleanup; the temp dir lives inside java.io.tmpdir
|
||||
// and will be reclaimed by the OS on next reboot if we lose the race.
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// ditto
|
||||
}
|
||||
// Suppress IDE warning about unused parameter when File.delete fails silently.
|
||||
File f = tempDir.toFile();
|
||||
if (f.exists() && !f.delete()) {
|
||||
log.debug("[LibreOfficePdf] could not delete temp dir {}", tempDir);
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,74 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.boot.context.properties.EnableConfigurationProperties;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
/**
|
||||
* Orchestrate PDF rendering. Picks a {@link PdfBackend} based on the caller's
|
||||
* engine preference, the backend's {@link PdfBackend#isAvailable()} probe, and
|
||||
* its {@link PdfBackend#supports(PdfRenderRequest)} declaration. Both backends
|
||||
* receive a normalised {@link PdfRenderRequest} so they don't have to redo
|
||||
* frontmatter parsing or page-size defaulting.
|
||||
*
|
||||
* <p>Dispatch table:
|
||||
* <pre>
|
||||
* engine=AUTO + libreoffice ok + supports request → libreoffice
|
||||
* engine=AUTO + libreoffice missing OR can't do header/footer → openhtmltopdf
|
||||
* engine=LIBREOFFICE + supports → libreoffice (else throw)
|
||||
* engine=HTML → openhtmltopdf
|
||||
* </pre>
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
@EnableConfigurationProperties(PdfProperties.class)
|
||||
public class MarkdownPdfRenderer {
|
||||
|
||||
private final LibreOfficePdfBackend libreOffice;
|
||||
private final FlyingSaucerPdfBackend html;
|
||||
private final PdfProperties properties;
|
||||
|
||||
public record Result(byte[] bytes, String backend) {}
|
||||
|
||||
public Result render(String markdown, String pageSize, PdfProperties.Engine engine) throws Exception {
|
||||
if (engine == null) engine = properties.defaultEngine();
|
||||
|
||||
PdfFrontmatter fm = PdfFrontmatter.parseOrSynthesise(markdown);
|
||||
String body = PdfFrontmatter.stripFrontmatter(markdown);
|
||||
PdfRenderRequest request = new PdfRenderRequest(body, fm, pageSize, engine);
|
||||
|
||||
PdfBackend chosen = pick(request);
|
||||
long t0 = System.currentTimeMillis();
|
||||
byte[] bytes = chosen.render(request);
|
||||
log.info("[Pdf] rendered via {} ({} bytes, {}ms, frontmatter cover={} header={} footer={})",
|
||||
chosen.name(), bytes.length, System.currentTimeMillis() - t0,
|
||||
fm.hasCover(), fm.hasHeader(), fm.hasFooter());
|
||||
return new Result(bytes, chosen.name());
|
||||
}
|
||||
|
||||
private PdfBackend pick(PdfRenderRequest request) {
|
||||
return switch (request.engine()) {
|
||||
case LIBREOFFICE -> {
|
||||
if (!libreOffice.isAvailable()) {
|
||||
throw new IllegalStateException(
|
||||
"engine=libreoffice but soffice is not available on PATH");
|
||||
}
|
||||
if (!libreOffice.supports(request)) {
|
||||
throw new IllegalStateException(
|
||||
"engine=libreoffice but the request needs cover/header/footer; "
|
||||
+ "use engine=html or remove those frontmatter fields");
|
||||
}
|
||||
yield libreOffice;
|
||||
}
|
||||
case HTML -> html;
|
||||
case AUTO -> {
|
||||
if (libreOffice.isAvailable() && libreOffice.supports(request)) {
|
||||
yield libreOffice;
|
||||
}
|
||||
yield html;
|
||||
}
|
||||
};
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,29 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
/**
|
||||
* One way to turn markdown bytes into PDF bytes. {@link MarkdownPdfRenderer}
|
||||
* picks an implementation at request time based on availability and the
|
||||
* caller's {@link PdfRenderRequest#engine()} preference.
|
||||
*/
|
||||
public interface PdfBackend {
|
||||
|
||||
/** Stable identifier surfaced in the tool result and in logs. */
|
||||
String name();
|
||||
|
||||
/**
|
||||
* Whether this backend can run at all on the current host. The default
|
||||
* implementation says yes; the LibreOffice backend overrides this to
|
||||
* probe for {@code soffice}.
|
||||
*/
|
||||
default boolean isAvailable() { return true; }
|
||||
|
||||
/**
|
||||
* Whether this backend can faithfully render the request. The HTML
|
||||
* backend always returns {@code true}; the LibreOffice backend declines
|
||||
* requests that need cover / header / footer because those features
|
||||
* cannot be expressed through the docx intermediate.
|
||||
*/
|
||||
default boolean supports(PdfRenderRequest request) { return true; }
|
||||
|
||||
byte[] render(PdfRenderRequest request) throws Exception;
|
||||
}
|
||||
@ -0,0 +1,165 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
import org.commonmark.ext.front.matter.YamlFrontMatterExtension;
|
||||
import org.commonmark.ext.front.matter.YamlFrontMatterVisitor;
|
||||
import org.commonmark.node.Node;
|
||||
import org.commonmark.parser.Parser;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Extract the YAML frontmatter block at the top of a markdown body so the PDF
|
||||
* pipeline can drive cover / page header / page footer text from it. The
|
||||
* frontmatter block — when present — has the form:
|
||||
* <pre>
|
||||
* ---
|
||||
* title: 季度报告
|
||||
* subtitle: Q1 2026
|
||||
* header: 内部资料
|
||||
* footer: Mate Inc. © 2026
|
||||
* ---
|
||||
* </pre>
|
||||
*
|
||||
* <p>Markdown without frontmatter parses to {@link #empty()}; the renderer
|
||||
* then synthesises a cover from the first {@code # H1} heading and uses
|
||||
* default header / footer text.
|
||||
*/
|
||||
public record PdfFrontmatter(
|
||||
String title,
|
||||
String subtitle,
|
||||
String header,
|
||||
String footer,
|
||||
boolean explicitCover) {
|
||||
|
||||
/**
|
||||
* Backwards-compat constructor for callers that only know about the four
|
||||
* text slots; the cover-source flag defaults to {@code false} (synthetic).
|
||||
*/
|
||||
public PdfFrontmatter(String title, String subtitle, String header, String footer) {
|
||||
this(title, subtitle, header, footer, false);
|
||||
}
|
||||
|
||||
public boolean hasCover() {
|
||||
return notBlank(title) || notBlank(subtitle);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the cover came from a YAML frontmatter block (true) or was
|
||||
* synthesised by promoting a leading {@code # H1} into a cover title (false).
|
||||
* Synthetic covers are not real layout requirements — the LibreOffice
|
||||
* backend can ignore them and render the H1 inline as part of the document.
|
||||
*/
|
||||
public boolean hasExplicitCover() {
|
||||
return explicitCover && hasCover();
|
||||
}
|
||||
|
||||
public boolean hasHeader() {
|
||||
return notBlank(header);
|
||||
}
|
||||
|
||||
public boolean hasFooter() {
|
||||
return notBlank(footer);
|
||||
}
|
||||
|
||||
/** Whether ANY of the frontmatter slots is populated. */
|
||||
public boolean isPresent() {
|
||||
return hasCover() || hasHeader() || hasFooter();
|
||||
}
|
||||
|
||||
public static PdfFrontmatter empty() {
|
||||
return new PdfFrontmatter(null, null, null, null, false);
|
||||
}
|
||||
|
||||
public static PdfFrontmatter parse(String markdown) {
|
||||
if (markdown == null || markdown.isBlank()) return empty();
|
||||
|
||||
Parser parser = Parser.builder()
|
||||
.extensions(List.of(YamlFrontMatterExtension.create()))
|
||||
.build();
|
||||
Node document = parser.parse(markdown);
|
||||
|
||||
YamlFrontMatterVisitor visitor = new YamlFrontMatterVisitor();
|
||||
document.accept(visitor);
|
||||
Map<String, List<String>> data = visitor.getData();
|
||||
if (data == null || data.isEmpty()) return empty();
|
||||
|
||||
return new PdfFrontmatter(
|
||||
first(data, "title"),
|
||||
first(data, "subtitle"),
|
||||
first(data, "header"),
|
||||
first(data, "footer"),
|
||||
/* explicitCover = */ true);
|
||||
}
|
||||
|
||||
private static String first(Map<String, List<String>> data, String key) {
|
||||
List<String> values = data.get(key);
|
||||
if (values == null || values.isEmpty()) return null;
|
||||
String v = values.get(0);
|
||||
if (v == null) return null;
|
||||
// YAML scalar values come back with surrounding quotes preserved when the
|
||||
// user wrote `title: "..."`. Strip a single matching pair so the rendered
|
||||
// cover doesn't show literal quote characters.
|
||||
v = v.trim();
|
||||
if ((v.startsWith("\"") && v.endsWith("\"") && v.length() >= 2)
|
||||
|| (v.startsWith("'") && v.endsWith("'") && v.length() >= 2)) {
|
||||
v = v.substring(1, v.length() - 1);
|
||||
}
|
||||
return v;
|
||||
}
|
||||
|
||||
private static boolean notBlank(String s) {
|
||||
return s != null && !s.isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* Convenience: read frontmatter, if missing look for a leading {@code # H1}
|
||||
* to use as the cover title. The synthesised result is flagged with
|
||||
* {@code explicitCover=false} so backends that cannot render an actual
|
||||
* cover page (LibreOffice via the docx intermediate) can safely ignore it
|
||||
* — the H1 will still render as the first heading inline.
|
||||
*/
|
||||
public static PdfFrontmatter parseOrSynthesise(String markdown) {
|
||||
PdfFrontmatter fm = parse(markdown);
|
||||
if (fm.hasCover()) return fm;
|
||||
|
||||
String firstHeading = firstHeading(markdown);
|
||||
if (firstHeading != null) {
|
||||
return new PdfFrontmatter(firstHeading, fm.subtitle(), fm.header(), fm.footer(),
|
||||
/* explicitCover = */ false);
|
||||
}
|
||||
return fm;
|
||||
}
|
||||
|
||||
private static String firstHeading(String markdown) {
|
||||
for (String rawLine : markdown.split("\\R", -1)) {
|
||||
String line = rawLine.strip();
|
||||
if (line.startsWith("# ") && line.length() > 2) {
|
||||
return line.substring(2).strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Strip a leading YAML frontmatter block from a markdown body. */
|
||||
public static String stripFrontmatter(String markdown) {
|
||||
if (markdown == null) return "";
|
||||
String trimmed = markdown.stripLeading();
|
||||
if (!trimmed.startsWith("---")) return markdown;
|
||||
int firstBreak = trimmed.indexOf('\n');
|
||||
if (firstBreak < 0) return markdown;
|
||||
int closing = trimmed.indexOf("\n---", firstBreak);
|
||||
if (closing < 0) return markdown;
|
||||
int after = trimmed.indexOf('\n', closing + 4);
|
||||
return after < 0 ? "" : trimmed.substring(after + 1);
|
||||
}
|
||||
|
||||
/** Try to find {@link Optional} variant for callers preferring null-safe accessors. */
|
||||
public Optional<String> titleOpt() { return Optional.ofNullable(title).filter(PdfFrontmatter::nb); }
|
||||
public Optional<String> subtitleOpt() { return Optional.ofNullable(subtitle).filter(PdfFrontmatter::nb); }
|
||||
public Optional<String> headerOpt() { return Optional.ofNullable(header).filter(PdfFrontmatter::nb); }
|
||||
public Optional<String> footerOpt() { return Optional.ofNullable(footer).filter(PdfFrontmatter::nb); }
|
||||
|
||||
private static boolean nb(String s) { return !s.isBlank(); }
|
||||
}
|
||||
@ -0,0 +1,44 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
|
||||
/**
|
||||
* Configuration for the markdown-to-PDF rendering pipeline.
|
||||
*
|
||||
* <p>Example {@code application.yml}:
|
||||
* <pre>
|
||||
* mateclaw:
|
||||
* pdf:
|
||||
* fontPath: /Library/Fonts/Songti.ttc
|
||||
* defaultEngine: AUTO
|
||||
* libreoffice:
|
||||
* enabled: true
|
||||
* binary: soffice
|
||||
* </pre>
|
||||
*/
|
||||
@ConfigurationProperties(prefix = "mateclaw.pdf")
|
||||
public record PdfProperties(
|
||||
String fontPath,
|
||||
Engine defaultEngine,
|
||||
Libreoffice libreoffice) {
|
||||
|
||||
public PdfProperties {
|
||||
if (defaultEngine == null) defaultEngine = Engine.AUTO;
|
||||
if (libreoffice == null) libreoffice = new Libreoffice(true, "soffice");
|
||||
}
|
||||
|
||||
public enum Engine {
|
||||
/** Try LibreOffice first, fall back to OpenHTMLtoPDF. */
|
||||
AUTO,
|
||||
/** Force the LibreOffice subprocess path. Fails if soffice is missing. */
|
||||
LIBREOFFICE,
|
||||
/** Force the in-process OpenHTMLtoPDF path. */
|
||||
HTML
|
||||
}
|
||||
|
||||
public record Libreoffice(boolean enabled, String binary) {
|
||||
public Libreoffice {
|
||||
if (binary == null || binary.isBlank()) binary = "soffice";
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -0,0 +1,28 @@
|
||||
package vip.mate.tool.document.pdf;
|
||||
|
||||
/**
|
||||
* Request payload handed to a {@link PdfBackend}. The orchestrator builds
|
||||
* this once per call after parsing frontmatter and resolving page size; the
|
||||
* backends are read-only consumers.
|
||||
*
|
||||
* @param markdown markdown body with the YAML frontmatter block already stripped
|
||||
* @param frontmatter parsed (or synthesised from a leading {@code # H1}) frontmatter
|
||||
* @param pageSize "A4" or "LETTER"
|
||||
* @param engine the engine preference the caller gave; the orchestrator
|
||||
* uses this to decide which backend to ask, but each backend
|
||||
* only sees the request after the choice has been made and
|
||||
* may largely ignore the field
|
||||
*/
|
||||
public record PdfRenderRequest(
|
||||
String markdown,
|
||||
PdfFrontmatter frontmatter,
|
||||
String pageSize,
|
||||
PdfProperties.Engine engine) {
|
||||
|
||||
public PdfRenderRequest {
|
||||
if (markdown == null) markdown = "";
|
||||
if (frontmatter == null) frontmatter = PdfFrontmatter.empty();
|
||||
if (pageSize == null || pageSize.isBlank()) pageSize = "A4";
|
||||
if (engine == null) engine = PdfProperties.Engine.AUTO;
|
||||
}
|
||||
}
|
||||
@ -482,6 +482,21 @@ MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name,
|
||||
KEY (id)
|
||||
VALUES (1000000019, 'DocxRenderTool', 'DOCX Render', 'Render Markdown directly into a .docx and return a one-time download link. In-process Apache POI implementation, no Node.js subprocess; supports headings, bold, lists, tables. Preferred tool for creating new documents.', 'builtin', 'docxRenderTool', '📝', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- Built-in tool: XLSX Render (in-process Apache POI; markdown tables -> multi-sheet workbook)
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000020, 'XlsxRenderTool', 'XLSX Render', 'Render Markdown directly into a .xlsx workbook and return a one-time download link. In-process Apache POI; each # heading becomes a sheet, pipe tables become rows, numeric cells auto-detected.', 'builtin', 'xlsxRenderTool', '📊', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- Built-in tool: PPTX Render (in-process Apache POI; Marp-style markdown -> .pptx deck)
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000021, 'PptxRenderTool', 'PPTX Render', 'Render Marp-style Markdown directly into a .pptx deck and return a one-time download link. In-process Apache POI; --- separates slides, # / ## titles, - bullets, <!-- speaker notes -->.', 'builtin', 'pptxRenderTool', '🎞️', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- Built-in tool: PDF Render (dual backend: LibreOffice subprocess preferred, OpenPDF + Flying Saucer fallback)
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000022, 'PdfRenderTool', 'PDF Render', 'Render Markdown into a final-form .pdf and return a one-time download link. Two backends (LibreOffice subprocess preferred, OpenPDF + Flying Saucer fallback); supports YAML frontmatter for cover / page header / page footer.', 'builtin', 'pdfRenderTool', '📄', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- Example MCP Server: Filesystem (see MateClaw docs mcpServers.filesystem)
|
||||
MERGE INTO mate_mcp_server (
|
||||
id, name, description, transport, url, headers_json, command, args_json, env_json, cwd,
|
||||
|
||||
@ -533,6 +533,21 @@ INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name
|
||||
VALUES (1000000019, 'DocxRenderTool', 'DOCX Render', 'Render Markdown directly into a .docx and return a one-time download link. In-process Apache POI implementation, no Node.js subprocess; supports headings, bold, lists, tables. Preferred tool for creating new documents.', 'builtin', 'docxRenderTool', '📝', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- Built-in tool: XLSX Render (in-process Apache POI; markdown tables -> multi-sheet workbook)
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000020, 'XlsxRenderTool', 'XLSX Render', 'Render Markdown directly into a .xlsx workbook and return a one-time download link. In-process Apache POI; each # heading becomes a sheet, pipe tables become rows, numeric cells auto-detected.', 'builtin', 'xlsxRenderTool', '📊', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- Built-in tool: PPTX Render (in-process Apache POI; Marp-style markdown -> .pptx deck)
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000021, 'PptxRenderTool', 'PPTX Render', 'Render Marp-style Markdown directly into a .pptx deck and return a one-time download link. In-process Apache POI; --- separates slides, # / ## titles, - bullets, <!-- speaker notes -->.', 'builtin', 'pptxRenderTool', '🎞️', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- Built-in tool: PDF Render (dual backend: LibreOffice subprocess preferred, OpenPDF + Flying Saucer fallback)
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000022, 'PdfRenderTool', 'PDF Render', 'Render Markdown into a final-form .pdf and return a one-time download link. Two backends (LibreOffice subprocess preferred, OpenPDF + Flying Saucer fallback); supports YAML frontmatter for cover / page header / page footer.', 'builtin', 'pdfRenderTool', '📄', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- Example MCP Server: Filesystem (see MateClaw docs mcpServers.filesystem)
|
||||
INSERT INTO mate_mcp_server (id, name, description, transport, url, headers_json, command, args_json, env_json, cwd,
|
||||
enabled, connect_timeout_seconds, read_timeout_seconds, last_status, last_error,
|
||||
|
||||
@ -531,6 +531,21 @@ INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name
|
||||
VALUES (1000000019, 'DocxRenderTool', 'DOCX 渲染', '将 Markdown 直接渲染为 .docx 并返回一次性下载链接。进程内 Apache POI 实现,无需 Node.js 子进程;支持标题、加粗、列表、表格。新建文档场景的首选工具。', 'builtin', 'docxRenderTool', '📝', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- 内置工具:XLSX 渲染(进程内 Apache POI,从 Markdown 表格生成多 sheet 工作簿)
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000020, 'XlsxRenderTool', 'XLSX 渲染', '将 Markdown 直接渲染为 .xlsx 工作簿并返回一次性下载链接。进程内 Apache POI 实现;每个 # 一级标题生成一个 sheet,竖线表格成为行内容,数字单元格自动识别。', 'builtin', 'xlsxRenderTool', '📊', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- 内置工具:PPTX 渲染(进程内 Apache POI,Marp 风格 Markdown 生成 .pptx)
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000021, 'PptxRenderTool', 'PPTX 渲染', '将 Marp 风格的 Markdown 直接渲染为 .pptx 演示文稿并返回一次性下载链接。进程内 Apache POI 实现;--- 分页、# / ## 作幻灯片标题、- 作要点、<!-- ... --> 作演讲者备注。', 'builtin', 'pptxRenderTool', '🎞️', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- 内置工具:PDF 渲染(双 backend:LibreOffice 子进程优先,进程内 OpenPDF + Flying Saucer 兜底)
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000022, 'PdfRenderTool', 'PDF 渲染', '将 Markdown 渲染为最终交付形态的 .pdf 并返回一次性下载链接。双 backend 自动切换(优先 LibreOffice,不可用时回落到进程内 OpenPDF + Flying Saucer);通过 YAML frontmatter 控制封面、页眉、页脚。', 'builtin', 'pdfRenderTool', '📄', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), tool_type=VALUES(tool_type), bean_name=VALUES(bean_name), icon=VALUES(icon), enabled=VALUES(enabled), builtin=VALUES(builtin), update_time=VALUES(update_time), deleted=VALUES(deleted);
|
||||
|
||||
-- 示例 MCP Server:Filesystem(参考 MateClaw 文档中的 mcpServers.filesystem)
|
||||
INSERT INTO mate_mcp_server (
|
||||
id, name, description, transport, url, headers_json, command, args_json, env_json, cwd,
|
||||
|
||||
@ -485,6 +485,21 @@ MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name,
|
||||
KEY (id)
|
||||
VALUES (1000000019, 'DocxRenderTool', 'DOCX 渲染', '将 Markdown 直接渲染为 .docx 并返回一次性下载链接。进程内 Apache POI 实现,无需 Node.js 子进程;支持标题、加粗、列表、表格。新建文档场景的首选工具。', 'builtin', 'docxRenderTool', '📝', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- 内置工具:XLSX 渲染(进程内 Apache POI,从 Markdown 表格生成多 sheet 工作簿)
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000020, 'XlsxRenderTool', 'XLSX 渲染', '将 Markdown 直接渲染为 .xlsx 工作簿并返回一次性下载链接。进程内 Apache POI 实现;每个 # 一级标题生成一个 sheet,竖线表格成为行内容,数字单元格自动识别。', 'builtin', 'xlsxRenderTool', '📊', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- 内置工具:PPTX 渲染(进程内 Apache POI,Marp 风格 Markdown 生成 .pptx)
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000021, 'PptxRenderTool', 'PPTX 渲染', '将 Marp 风格的 Markdown 直接渲染为 .pptx 演示文稿并返回一次性下载链接。进程内 Apache POI 实现;--- 分页、# / ## 作幻灯片标题、- 作要点、<!-- ... --> 作演讲者备注。', 'builtin', 'pptxRenderTool', '🎞️', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- 内置工具:PDF 渲染(双 backend:LibreOffice 子进程优先,进程内 OpenPDF + Flying Saucer 兜底)
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000022, 'PdfRenderTool', 'PDF 渲染', '将 Markdown 渲染为最终交付形态的 .pdf 并返回一次性下载链接。双 backend 自动切换(优先 LibreOffice,不可用时回落到进程内 OpenPDF + Flying Saucer);通过 YAML frontmatter 控制封面、页眉、页脚。', 'builtin', 'pdfRenderTool', '📄', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
-- 示例 MCP Server:Filesystem(参考 MateClaw 文档中的 mcpServers.filesystem)
|
||||
MERGE INTO mate_mcp_server (
|
||||
id, name, description, transport, url, headers_json, command, args_json, env_json, cwd,
|
||||
|
||||
@ -0,0 +1,16 @@
|
||||
-- V94: Register XlsxRenderTool / PptxRenderTool / PdfRenderTool as built-in tools.
|
||||
-- These mirror DocxRenderTool (V31) so agents can bind them through the tool picker
|
||||
-- and so the AvailableToolService surfaces them in the UI.
|
||||
-- Idempotent: MERGE INTO updates existing rows when id matches.
|
||||
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000020, 'XlsxRenderTool', 'XLSX Render', 'Render Markdown directly into a .xlsx workbook and return a one-time download link. In-process Apache POI; each # heading becomes a sheet, pipe tables become rows, numeric cells auto-detected.', 'builtin', 'xlsxRenderTool', '📊', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000021, 'PptxRenderTool', 'PPTX Render', 'Render Marp-style Markdown directly into a .pptx deck and return a one-time download link. In-process Apache POI; --- separates slides, # / ## titles, - bullets, <!-- speaker notes -->.', 'builtin', 'pptxRenderTool', '🎞️', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
|
||||
MERGE INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
KEY (id)
|
||||
VALUES (1000000022, 'PdfRenderTool', 'PDF Render', 'Render Markdown into a final-form .pdf and return a one-time download link. Two backends (LibreOffice subprocess preferred, OpenPDF + Flying Saucer fallback); supports YAML frontmatter for cover / page header / page footer.', 'builtin', 'pdfRenderTool', '📄', TRUE, TRUE, NOW(), NOW(), 0);
|
||||
@ -0,0 +1,16 @@
|
||||
-- V94: Register XlsxRenderTool / PptxRenderTool / PdfRenderTool as built-in tools.
|
||||
-- These mirror DocxRenderTool (V31) so agents can bind them through the tool picker
|
||||
-- and so the AvailableToolService surfaces them in the UI.
|
||||
-- Idempotent: ON DUPLICATE KEY UPDATE keeps rows in sync if they already exist.
|
||||
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000020, 'XlsxRenderTool', 'XLSX Render', 'Render Markdown directly into a .xlsx workbook and return a one-time download link. In-process Apache POI; each # heading becomes a sheet, pipe tables become rows, numeric cells auto-detected.', 'builtin', 'xlsxRenderTool', '📊', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), bean_name=VALUES(bean_name), icon=VALUES(icon), update_time=VALUES(update_time);
|
||||
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000021, 'PptxRenderTool', 'PPTX Render', 'Render Marp-style Markdown directly into a .pptx deck and return a one-time download link. In-process Apache POI; --- separates slides, # / ## titles, - bullets, <!-- speaker notes -->.', 'builtin', 'pptxRenderTool', '🎞️', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), bean_name=VALUES(bean_name), icon=VALUES(icon), update_time=VALUES(update_time);
|
||||
|
||||
INSERT INTO mate_tool (id, name, display_name, description, tool_type, bean_name, icon, enabled, builtin, create_time, update_time, deleted)
|
||||
VALUES (1000000022, 'PdfRenderTool', 'PDF Render', 'Render Markdown into a final-form .pdf and return a one-time download link. Two backends (LibreOffice subprocess preferred, OpenPDF + Flying Saucer fallback); supports YAML frontmatter for cover / page header / page footer.', 'builtin', 'pdfRenderTool', '📄', TRUE, TRUE, NOW(), NOW(), 0)
|
||||
ON DUPLICATE KEY UPDATE name=VALUES(name), display_name=VALUES(display_name), description=VALUES(description), bean_name=VALUES(bean_name), icon=VALUES(icon), update_time=VALUES(update_time);
|
||||
Loading…
Reference in New Issue
Block a user