mirror of
https://gitee.com/mateos/mateclaw.git
synced 2026-09-15 20:08:18 +08:00
feat(wiki): Tika as last-resort document extractor
This commit is contained in:
parent
6474d0e6be
commit
c752c1f2ae
@ -296,8 +296,6 @@
|
|||||||
Used by WikiContentNormalizer to strip nav/footer/script/style/aside
|
Used by WikiContentNormalizer to strip nav/footer/script/style/aside
|
||||||
and ad-class nodes from URL/HTML uploads before chunking. Small
|
and ad-class nodes from URL/HTML uploads before chunking. Small
|
||||||
(~430KB), no transitive deps, JVM-only — safe for the desktop bundle.
|
(~430KB), no transitive deps, JVM-only — safe for the desktop bundle.
|
||||||
Tika is intentionally not pulled in: DocumentExtractTool already
|
|
||||||
covers PDF/Office via pdftotext, pdfplumber, and Java fallbacks.
|
|
||||||
-->
|
-->
|
||||||
<dependency>
|
<dependency>
|
||||||
<groupId>org.jsoup</groupId>
|
<groupId>org.jsoup</groupId>
|
||||||
@ -305,6 +303,35 @@
|
|||||||
<version>1.18.3</version>
|
<version>1.18.3</version>
|
||||||
</dependency>
|
</dependency>
|
||||||
|
|
||||||
|
<!-- ===== Apache Tika (RFC-051 PR-?: Java-side last-resort extractor) ===== -->
|
||||||
|
<!--
|
||||||
|
Wired as the FINAL fallback in DocumentExtractTool's PDF/DOCX/XLSX/PPTX
|
||||||
|
chains, after every system command + Python + POI-based path has failed.
|
||||||
|
Used in production primarily by Windows users without Python or poppler
|
||||||
|
installed; otherwise idle.
|
||||||
|
|
||||||
|
Pinned to the precise format modules called out in RFC-051 §5.2 — we
|
||||||
|
deliberately avoid `tika-parsers-standard-package`, which pulls in mail,
|
||||||
|
audio, archive, RTF / ODT, scientific, etc. (~80MB). Current footprint:
|
||||||
|
tika-core (~700KB) + tika-parser-pdf-module (PDFBox ~5MB) +
|
||||||
|
tika-parser-microsoft-module (POI-scratchpad ~10MB) ≈ 16MB.
|
||||||
|
-->
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.apache.tika</groupId>
|
||||||
|
<artifactId>tika-core</artifactId>
|
||||||
|
<version>3.0.0</version>
|
||||||
|
</dependency>
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.apache.tika</groupId>
|
||||||
|
<artifactId>tika-parser-pdf-module</artifactId>
|
||||||
|
<version>3.0.0</version>
|
||||||
|
</dependency>
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.apache.tika</groupId>
|
||||||
|
<artifactId>tika-parser-microsoft-module</artifactId>
|
||||||
|
<version>3.0.0</version>
|
||||||
|
</dependency>
|
||||||
|
|
||||||
<!-- ===== Database Migration (Flyway) ===== -->
|
<!-- ===== Database Migration (Flyway) ===== -->
|
||||||
<dependency>
|
<dependency>
|
||||||
<groupId>org.flywaydb</groupId>
|
<groupId>org.flywaydb</groupId>
|
||||||
|
|||||||
@ -222,15 +222,26 @@ public class DocumentExtractTool {
|
|||||||
}
|
}
|
||||||
// attempts 已由 tryOcrExtract 内部记录失败原因
|
// attempts 已由 tryOcrExtract 内部记录失败原因
|
||||||
|
|
||||||
|
// 5. Tika 兜底(RFC-051 §5.2):所有命令行 / Python / PDFBox / OCR 都失败时
|
||||||
|
// 用 Java 内置的 Tika 再试一次。主要服务于 Windows 没装 Poppler / Python 的桌面用户。
|
||||||
|
long t4 = System.currentTimeMillis();
|
||||||
|
content = TikaExtractor.extract(path);
|
||||||
|
if (content != null && !content.isBlank()) {
|
||||||
|
attempts.add("tika: 成功 (" + (System.currentTimeMillis() - t4) + "ms)");
|
||||||
|
int pages = realPageCount > 0 ? realPageCount : estimatePages(content);
|
||||||
|
return new ExtractedContent(content, "tika", pages);
|
||||||
|
}
|
||||||
|
attempts.add("tika: 失败或不可用");
|
||||||
|
|
||||||
// 返回之前级别的部分结果(如果有)
|
// 返回之前级别的部分结果(如果有)
|
||||||
if (bestContent != null) {
|
if (bestContent != null) {
|
||||||
log.warn("[DocumentExtract] OCR 不可用,返回部分文本结果: method={}, length={}",
|
log.warn("[DocumentExtract] OCR/Tika 不可用,返回部分文本结果: method={}, length={}",
|
||||||
bestMethod, bestContent.strip().length());
|
bestMethod, bestContent.strip().length());
|
||||||
int pages = realPageCount > 0 ? realPageCount : estimatePages(bestContent);
|
int pages = realPageCount > 0 ? realPageCount : estimatePages(bestContent);
|
||||||
return new ExtractedContent(bestContent, bestMethod + "_partial", pages);
|
return new ExtractedContent(bestContent, bestMethod + "_partial", pages);
|
||||||
}
|
}
|
||||||
|
|
||||||
throw new Exception("所有 PDF 提取方法都失败(包括 OCR)");
|
throw new Exception("所有 PDF 提取方法都失败(包括 OCR 与 Tika)");
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@ -556,7 +567,17 @@ public class DocumentExtractTool {
|
|||||||
}
|
}
|
||||||
attempts.add("java_zip_xml: 失败");
|
attempts.add("java_zip_xml: 失败");
|
||||||
|
|
||||||
throw new Exception("所有 DOCX 提取方法都失败");
|
// 5. Tika 兜底(RFC-051 §5.2)—— 当 textutil/pandoc/libreoffice/ZIP-XML 全失败时。
|
||||||
|
// Tika 的 Microsoft 模块覆盖到 .docx 内嵌 SmartArt、批注、复杂表格等场景,正好填补
|
||||||
|
// 我们手写的 ZIP XML 解析器的盲区。
|
||||||
|
content = TikaExtractor.extract(path);
|
||||||
|
if (content != null && !content.isBlank()) {
|
||||||
|
attempts.add("tika: 成功");
|
||||||
|
return new ExtractedContent(content, "tika", 0);
|
||||||
|
}
|
||||||
|
attempts.add("tika: 失败或不可用");
|
||||||
|
|
||||||
|
throw new Exception("所有 DOCX 提取方法都失败(包括 Tika)");
|
||||||
}
|
}
|
||||||
|
|
||||||
private String tryTextutil(Path path) {
|
private String tryTextutil(Path path) {
|
||||||
@ -687,6 +708,17 @@ public class DocumentExtractTool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Our ZIP-XML extractor only reads <v> tags and skips the shared-strings table,
|
||||||
|
// so cells full of text labels look "empty". When that happens, fall through to
|
||||||
|
// Tika which knows how to resolve the shared-strings indirection.
|
||||||
|
if (text.toString().replaceAll("---.*?---", "").strip().isEmpty()) {
|
||||||
|
String fallback = TikaExtractor.extract(path);
|
||||||
|
if (fallback != null && !fallback.isBlank()) {
|
||||||
|
attempts.add("tika: 成功(ZIP-XML 仅有数字 / 共享字符串未解析)");
|
||||||
|
return new ExtractedContent(fallback, "tika", 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
attempts.add("java_zip_xml: 成功");
|
attempts.add("java_zip_xml: 成功");
|
||||||
return new ExtractedContent(text.toString(), "java_zip_xml", 0);
|
return new ExtractedContent(text.toString(), "java_zip_xml", 0);
|
||||||
}
|
}
|
||||||
@ -721,6 +753,17 @@ public class DocumentExtractTool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Slide layouts with text inside SmartArt / charts / grouped shapes don't surface
|
||||||
|
// through the simple <a:t> grep — Tika walks the full DrawingML graph and pulls
|
||||||
|
// them out. Only invoke when our walker produced nothing useful.
|
||||||
|
if (text.toString().replaceAll("---.*?---", "").strip().isEmpty()) {
|
||||||
|
String fallback = TikaExtractor.extract(path);
|
||||||
|
if (fallback != null && !fallback.isBlank()) {
|
||||||
|
attempts.add("tika: 成功(ZIP-XML 未抓到正文,可能是 SmartArt / 图表)");
|
||||||
|
return new ExtractedContent(fallback, "tika", Math.max(0, slideNum - 1));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
attempts.add("java_zip_xml: 成功");
|
attempts.add("java_zip_xml: 成功");
|
||||||
return new ExtractedContent(text.toString(), "java_zip_xml", Math.max(0, slideNum - 1));
|
return new ExtractedContent(text.toString(), "java_zip_xml", Math.max(0, slideNum - 1));
|
||||||
}
|
}
|
||||||
|
|||||||
@ -0,0 +1,91 @@
|
|||||||
|
package vip.mate.tool.builtin;
|
||||||
|
|
||||||
|
import lombok.extern.slf4j.Slf4j;
|
||||||
|
import org.apache.tika.exception.WriteLimitReachedException;
|
||||||
|
import org.apache.tika.metadata.Metadata;
|
||||||
|
import org.apache.tika.parser.AutoDetectParser;
|
||||||
|
import org.apache.tika.parser.ParseContext;
|
||||||
|
import org.apache.tika.sax.BodyContentHandler;
|
||||||
|
|
||||||
|
import java.io.InputStream;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* RFC-051 §5.2: Apache Tika as the last-resort document extractor.
|
||||||
|
* <p>
|
||||||
|
* Used by {@link DocumentExtractTool} only after every other path
|
||||||
|
* (pdftotext / pdfplumber / pdfbox / OCR for PDFs, and the system-command
|
||||||
|
* + ZIP-XML chain for Office formats) has failed. Tika ships its own
|
||||||
|
* PDFBox + POI internals, so it works on Windows installs without Python
|
||||||
|
* or Poppler — which is the actual scenario the RFC §13.1 pointed to.
|
||||||
|
*
|
||||||
|
* <h2>Safety</h2>
|
||||||
|
* <ul>
|
||||||
|
* <li>{@link BodyContentHandler} caps output at {@code maxChars}; when the
|
||||||
|
* cap is hit Tika throws {@link WriteLimitReachedException}, which we
|
||||||
|
* treat as a successful (truncated) extract rather than a failure.</li>
|
||||||
|
* <li>Tika 3.x has built-in zip-bomb defenses on its zip readers (POI's
|
||||||
|
* {@code ZipSecureFile}); we don't disable them.</li>
|
||||||
|
* <li>Any other parse failure returns {@code null} so the caller can fall
|
||||||
|
* through to its existing structured-error path.</li>
|
||||||
|
* </ul>
|
||||||
|
*
|
||||||
|
* The extractor is deliberately stateless and synchronous: callers drive
|
||||||
|
* concurrency externally.
|
||||||
|
*/
|
||||||
|
@Slf4j
|
||||||
|
public final class TikaExtractor {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reasonable default for a single-document parse. 5MB of text is
|
||||||
|
* well above any source we'd actually feed into the wiki pipeline,
|
||||||
|
* and well below what would OOM a typical desktop install.
|
||||||
|
*/
|
||||||
|
public static final int DEFAULT_MAX_CHARS = 5_000_000;
|
||||||
|
|
||||||
|
private TikaExtractor() {}
|
||||||
|
|
||||||
|
/** Extract with the default cap. */
|
||||||
|
public static String extract(Path path) {
|
||||||
|
return extract(path, DEFAULT_MAX_CHARS);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Extract text from {@code path} using Tika's {@link AutoDetectParser},
|
||||||
|
* capping output at {@code maxChars}. Returns the extracted text on
|
||||||
|
* success (possibly truncated), or {@code null} on any failure.
|
||||||
|
*/
|
||||||
|
public static String extract(Path path, int maxChars) {
|
||||||
|
if (path == null) return null;
|
||||||
|
if (!Files.isRegularFile(path)) {
|
||||||
|
log.debug("[Tika] Path is not a regular file: {}", path);
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
int cap = maxChars <= 0 ? DEFAULT_MAX_CHARS : maxChars;
|
||||||
|
|
||||||
|
BodyContentHandler handler = new BodyContentHandler(cap);
|
||||||
|
AutoDetectParser parser = new AutoDetectParser();
|
||||||
|
Metadata metadata = new Metadata();
|
||||||
|
ParseContext context = new ParseContext();
|
||||||
|
|
||||||
|
try (InputStream is = Files.newInputStream(path)) {
|
||||||
|
parser.parse(is, handler, metadata, context);
|
||||||
|
return handler.toString();
|
||||||
|
} catch (WriteLimitReachedException truncated) {
|
||||||
|
// Cap hit — Tika filled the handler before parsing finished. The
|
||||||
|
// partial text is still useful, especially since callers will chunk
|
||||||
|
// anyway and only want the leading prose for routing/embedding.
|
||||||
|
String partial = handler.toString();
|
||||||
|
log.info("[Tika] Output cap reached at {} chars for {}; returning partial",
|
||||||
|
partial.length(), path.getFileName());
|
||||||
|
return partial.isBlank() ? null : partial;
|
||||||
|
} catch (Throwable t) {
|
||||||
|
// Catching Throwable on purpose: Tika can throw NoClassDefFoundError /
|
||||||
|
// LinkageError when an obscure transitive parser is missing on a
|
||||||
|
// minimal classpath, and that should not crash the extract chain.
|
||||||
|
log.warn("[Tika] Parse failed for {}: {}", path.getFileName(), t.getMessage());
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Loading…
Reference in New Issue
Block a user