diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/service/PdfImageExtractor.java b/mateclaw-server/src/main/java/vip/mate/wiki/service/PdfImageExtractor.java index 742d3228..443a8cef 100644 --- a/mateclaw-server/src/main/java/vip/mate/wiki/service/PdfImageExtractor.java +++ b/mateclaw-server/src/main/java/vip/mate/wiki/service/PdfImageExtractor.java @@ -63,54 +63,6 @@ public class PdfImageExtractor { // a mock or a no-op wrapper. } - /** - * Quick check: does the PDF contain at least one inline image that - * passes the size threshold? Stops scanning at the first qualifying - * image so it's cheap on large documents. - * - *
Used by the upload pipeline to decide whether the absence of - * captions is an "actually nothing to caption" outcome (safe to cache) - * vs a "vision was unavailable" outcome (defer caching until flag - * flips on so the next read retries). - */ - public boolean hasInlineImages(Path pdfPath) { - if (pdfPath == null) { - return false; - } - File pdfFile = pdfPath.toFile(); - if (!pdfFile.isFile()) { - return false; - } - try (PDDocument doc = Loader.loadPDF(pdfFile)) { - for (PDPage page : doc.getPages()) { - PDResources resources = page.getResources(); - if (resources == null) continue; - for (COSName name : resources.getXObjectNames()) { - PDXObject obj; - try { - obj = resources.getXObject(name); - } catch (IOException ignored) { - continue; - } - if (!(obj instanceof PDImageXObject pdImage)) continue; - BufferedImage bi; - try { - bi = pdImage.getImage(); - } catch (IOException | RuntimeException ignored) { - continue; - } - if (bi.getWidth() >= MIN_IMAGE_SIDE_PX - && bi.getHeight() >= MIN_IMAGE_SIDE_PX) { - return true; - } - } - } - } catch (IOException e) { - log.debug("[PdfImage] hasInlineImages probe failed for {}: {}", pdfPath, e.getMessage()); - } - return false; - } - /** * Walks every page of {@code pdfPath}, captions each qualifying inline * image, and returns the rendered marker lines in document order. diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java index 8085b6ad..f9446c52 100644 --- a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java +++ b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java @@ -360,26 +360,20 @@ public class WikiRawMaterialService { // Append inline-image captions for PDFs so chunk-level search // hits chart/diagram contents that the text extractor missed. // Failures are non-fatal: the body text is still returned. - EnrichmentOutcome enrichment = appendPdfImageCaptions(entity, text); + String enriched = appendPdfImageCaptions(entity, text); if (truncated) { - // 截断的结果不缓存,避免永久丢失后半内容。返回文本供分块处理使用。 + // Truncated results are not cached so we don't lose the tail + // permanently — we still return the text for chunking use. log.warn("[Wiki] Extracted text truncated at {} chars for: {} (full document may be larger)", text.length(), entity.getSourcePath()); - } else if (!enrichment.shouldCache()) { - // Vision was unavailable but the PDF has images we'd otherwise - // caption — leave extracted_text NULL so the next call retries - // once the operator enables wiki.ocr.enabled. - log.info("[Wiki] PDF id={} captioning deferred (vision disabled or unavailable); " - + "extracted_text not cached so next read retries", entity.getId()); } else { - // Full extraction: cache to avoid re-extracting on subsequent calls. - updateExtractedText(entity.getId(), enrichment.text()); + updateExtractedText(entity.getId(), enriched); } log.info("[Wiki] Extracted text from {}: {} chars (text) → {} chars (enriched), method={}, truncated={}, cached={}", - entity.getSourcePath(), text.length(), enrichment.text().length(), - json.getStr("method"), truncated, enrichment.shouldCache() && !truncated); - return enrichment.text(); + entity.getSourcePath(), text.length(), enriched.length(), + json.getStr("method"), truncated, !truncated); + return enriched; } } log.warn("[Wiki] Document extraction returned no text for: {}", entity.getSourcePath()); @@ -390,75 +384,40 @@ public class WikiRawMaterialService { return entity.getOriginalContent(); } - /** - * Enrichment outcome: text + a hint about whether the result is final. - * - *
{@code shouldCache=false} signals the caller to skip the - * extracted_text cache write so the next call re-runs PDF image - * captioning. This handles the common operator workflow of uploading - * a PDF before flipping {@code wiki.ocr.enabled} on — without the - * skip, the partial text-only result would block the eventual - * caption catch-up. - */ - private record EnrichmentOutcome(String text, boolean shouldCache) { } - /** * For PDF raw materials, walks the inline images and appends a section of * {@code [图 P{n}#{m}]:
Returns an {@link EnrichmentOutcome} that distinguishes three cases: - *
When {@link #VISION_FLAG_KEY} is off, returns the body unchanged
+ * without re-parsing the PDF — the operator's "off = ignore images"
+ * intent. Flipping the flag on later requires a manual reprocess for
+ * existing rows to pick up captions.
*/
- private EnrichmentOutcome appendPdfImageCaptions(WikiRawMaterialEntity entity, String body) {
+ private String appendPdfImageCaptions(WikiRawMaterialEntity entity, String body) {
if (!"pdf".equals(entity.getSourceType()) || pdfImageExtractor == null) {
- return new EnrichmentOutcome(body, true);
+ return body;
+ }
+ if (featureFlagService == null || !featureFlagService.isEnabled(VISION_FLAG_KEY)) {
+ return body;
}
java.nio.file.Path pdfPath = java.nio.file.Paths.get(entity.getSourcePath());
-
- // Vision-disabled path: only refuse the cache when the PDF actually has
- // qualifying images. Image-less PDFs can still be cached as text-only.
- boolean visionEnabled = featureFlagService != null
- && featureFlagService.isEnabled(VISION_FLAG_KEY);
- if (!visionEnabled) {
- boolean hasImages = pdfImageExtractor.hasInlineImages(pdfPath);
- if (hasImages) {
- log.info("[Wiki] PDF id={} has inline images but {} is disabled; "
- + "skipping captioning and deferring extracted_text cache",
- entity.getId(), VISION_FLAG_KEY);
- return new EnrichmentOutcome(body, false);
- }
- return new EnrichmentOutcome(body, true);
- }
-
try {
List