mirror of
https://gitee.com/mateos/mateclaw.git
synced 2026-09-13 11:13:43 +08:00
feat(wiki): expose method=tika short-circuit on extract_document_text
This commit is contained in:
parent
c752c1f2ae
commit
c13d9b4c88
@ -45,20 +45,26 @@ public class DocumentExtractTool {
|
||||
- Excel (.xlsx, .xls) - 提取为文本表格
|
||||
- PowerPoint (.pptx, .ppt)
|
||||
|
||||
提取策略(自动选择最优方式):
|
||||
提取策略(默认自动选择最优方式):
|
||||
1. 优先使用系统命令(pdftotext, textutil, pandoc 等)
|
||||
2. 系统命令不可用时使用纯 Java 实现
|
||||
3. 返回详细的提取过程和元数据
|
||||
3. PDF 扫描版进入 OCR
|
||||
4. 全部失败前用 Apache Tika 兜底(覆盖 SmartArt、共享字符串表等盲区)
|
||||
5. 返回详细的提取过程和元数据
|
||||
|
||||
参数 options 可包含:
|
||||
- pages: 指定页码范围(如 "1-5" 或 "1,3,5")
|
||||
- preserveLayout: 是否保留布局(默认 true)
|
||||
- method: 强制指定提取器,跳过自动 fallback 链。当前支持:
|
||||
* "auto"(默认)—— 走完整 fallback 链
|
||||
* "tika" —— 直接用 Apache Tika 抽取,适合 Windows 上没装
|
||||
Poppler/Python 的环境,或验证 Tika 单独是否能解开
|
||||
|
||||
如果提取失败,会返回详细的尝试过程和错误信息
|
||||
""")
|
||||
public String extract_document_text(
|
||||
@ToolParam(description = "文件的绝对路径或相对路径") String filePath,
|
||||
@ToolParam(description = "可选参数 JSON,如 {\"pages\": \"1-5\", \"preserveLayout\": true}", required = false) String options) {
|
||||
@ToolParam(description = "可选参数 JSON,如 {\"pages\": \"1-5\", \"method\": \"tika\"}", required = false) String options) {
|
||||
|
||||
JSONObject result = new JSONObject();
|
||||
result.set("filePath", filePath);
|
||||
@ -75,6 +81,38 @@ public class DocumentExtractTool {
|
||||
String mimeType = detectMimeType(path);
|
||||
result.set("mimeType", mimeType);
|
||||
|
||||
// RFC-051: method=tika 短路 —— 跳过整条 fallback 链,直接调 Tika。
|
||||
// 用于:1) 测试 Tika 集成是否健康;2) 用户明知系统命令不可用、想免去
|
||||
// 那一长串失败日志的场景。结果里仍然带 attempts 数组,告知"应用户要求跳过自动链"。
|
||||
String forcedMethod = extractOption(options, "method");
|
||||
if ("tika".equalsIgnoreCase(forcedMethod)) {
|
||||
long t = System.currentTimeMillis();
|
||||
String text = TikaExtractor.extract(path);
|
||||
attempts.add("user-forced method=tika: skipped automatic fallback chain");
|
||||
if (text == null || text.isBlank()) {
|
||||
attempts.add("tika: 失败或不可用 (" + (System.currentTimeMillis() - t) + "ms)");
|
||||
return errorResult(filePath, "Tika 抽取无文本(可能格式不支持或文件损坏)", attempts);
|
||||
}
|
||||
attempts.add("tika: 成功 (" + (System.currentTimeMillis() - t) + "ms)");
|
||||
|
||||
String capped = text;
|
||||
boolean trunc = false;
|
||||
if (capped.length() > MAX_OUTPUT_LENGTH) {
|
||||
capped = capped.substring(0, MAX_OUTPUT_LENGTH)
|
||||
+ "\n\n... [内容已截断,总长度: " + text.length() + " 字符]";
|
||||
trunc = true;
|
||||
}
|
||||
result.set("text", capped);
|
||||
result.set("method", "tika");
|
||||
result.set("pages", estimatePages(text));
|
||||
result.set("attempts", attempts);
|
||||
result.set("truncated", trunc);
|
||||
result.set("success", true);
|
||||
log.info("[DocumentExtract] {} 使用 method=tika 强制提取成功,{} 字符",
|
||||
filePath, text.length());
|
||||
return JSONUtil.toJsonPrettyStr(result);
|
||||
}
|
||||
|
||||
// 根据类型选择提取器
|
||||
ExtractedContent content;
|
||||
if (mimeType.contains("pdf")) {
|
||||
|
||||
Loading…
Reference in New Issue
Block a user