mirror of
https://gitee.com/mateos/mateclaw.git
synced 2026-09-15 11:58:34 +08:00
fix(wiki): skip vision pipeline entirely when wiki.ocr.enabled is off
This commit is contained in:
parent
f910d762a3
commit
44ea8c784b
@ -63,54 +63,6 @@ public class PdfImageExtractor {
|
|||||||
// a mock or a no-op wrapper.
|
// a mock or a no-op wrapper.
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Quick check: does the PDF contain at least one inline image that
|
|
||||||
* passes the size threshold? Stops scanning at the first qualifying
|
|
||||||
* image so it's cheap on large documents.
|
|
||||||
*
|
|
||||||
* <p>Used by the upload pipeline to decide whether the absence of
|
|
||||||
* captions is an "actually nothing to caption" outcome (safe to cache)
|
|
||||||
* vs a "vision was unavailable" outcome (defer caching until flag
|
|
||||||
* flips on so the next read retries).
|
|
||||||
*/
|
|
||||||
public boolean hasInlineImages(Path pdfPath) {
|
|
||||||
if (pdfPath == null) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
File pdfFile = pdfPath.toFile();
|
|
||||||
if (!pdfFile.isFile()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
try (PDDocument doc = Loader.loadPDF(pdfFile)) {
|
|
||||||
for (PDPage page : doc.getPages()) {
|
|
||||||
PDResources resources = page.getResources();
|
|
||||||
if (resources == null) continue;
|
|
||||||
for (COSName name : resources.getXObjectNames()) {
|
|
||||||
PDXObject obj;
|
|
||||||
try {
|
|
||||||
obj = resources.getXObject(name);
|
|
||||||
} catch (IOException ignored) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (!(obj instanceof PDImageXObject pdImage)) continue;
|
|
||||||
BufferedImage bi;
|
|
||||||
try {
|
|
||||||
bi = pdImage.getImage();
|
|
||||||
} catch (IOException | RuntimeException ignored) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (bi.getWidth() >= MIN_IMAGE_SIDE_PX
|
|
||||||
&& bi.getHeight() >= MIN_IMAGE_SIDE_PX) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} catch (IOException e) {
|
|
||||||
log.debug("[PdfImage] hasInlineImages probe failed for {}: {}", pdfPath, e.getMessage());
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Walks every page of {@code pdfPath}, captions each qualifying inline
|
* Walks every page of {@code pdfPath}, captions each qualifying inline
|
||||||
* image, and returns the rendered marker lines in document order.
|
* image, and returns the rendered marker lines in document order.
|
||||||
|
|||||||
@ -360,26 +360,20 @@ public class WikiRawMaterialService {
|
|||||||
// Append inline-image captions for PDFs so chunk-level search
|
// Append inline-image captions for PDFs so chunk-level search
|
||||||
// hits chart/diagram contents that the text extractor missed.
|
// hits chart/diagram contents that the text extractor missed.
|
||||||
// Failures are non-fatal: the body text is still returned.
|
// Failures are non-fatal: the body text is still returned.
|
||||||
EnrichmentOutcome enrichment = appendPdfImageCaptions(entity, text);
|
String enriched = appendPdfImageCaptions(entity, text);
|
||||||
|
|
||||||
if (truncated) {
|
if (truncated) {
|
||||||
// 截断的结果不缓存,避免永久丢失后半内容。返回文本供分块处理使用。
|
// Truncated results are not cached so we don't lose the tail
|
||||||
|
// permanently — we still return the text for chunking use.
|
||||||
log.warn("[Wiki] Extracted text truncated at {} chars for: {} (full document may be larger)",
|
log.warn("[Wiki] Extracted text truncated at {} chars for: {} (full document may be larger)",
|
||||||
text.length(), entity.getSourcePath());
|
text.length(), entity.getSourcePath());
|
||||||
} else if (!enrichment.shouldCache()) {
|
|
||||||
// Vision was unavailable but the PDF has images we'd otherwise
|
|
||||||
// caption — leave extracted_text NULL so the next call retries
|
|
||||||
// once the operator enables wiki.ocr.enabled.
|
|
||||||
log.info("[Wiki] PDF id={} captioning deferred (vision disabled or unavailable); "
|
|
||||||
+ "extracted_text not cached so next read retries", entity.getId());
|
|
||||||
} else {
|
} else {
|
||||||
// Full extraction: cache to avoid re-extracting on subsequent calls.
|
updateExtractedText(entity.getId(), enriched);
|
||||||
updateExtractedText(entity.getId(), enrichment.text());
|
|
||||||
}
|
}
|
||||||
log.info("[Wiki] Extracted text from {}: {} chars (text) → {} chars (enriched), method={}, truncated={}, cached={}",
|
log.info("[Wiki] Extracted text from {}: {} chars (text) → {} chars (enriched), method={}, truncated={}, cached={}",
|
||||||
entity.getSourcePath(), text.length(), enrichment.text().length(),
|
entity.getSourcePath(), text.length(), enriched.length(),
|
||||||
json.getStr("method"), truncated, enrichment.shouldCache() && !truncated);
|
json.getStr("method"), truncated, !truncated);
|
||||||
return enrichment.text();
|
return enriched;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
log.warn("[Wiki] Document extraction returned no text for: {}", entity.getSourcePath());
|
log.warn("[Wiki] Document extraction returned no text for: {}", entity.getSourcePath());
|
||||||
@ -390,75 +384,40 @@ public class WikiRawMaterialService {
|
|||||||
return entity.getOriginalContent();
|
return entity.getOriginalContent();
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Enrichment outcome: text + a hint about whether the result is final.
|
|
||||||
*
|
|
||||||
* <p>{@code shouldCache=false} signals the caller to skip the
|
|
||||||
* extracted_text cache write so the next call re-runs PDF image
|
|
||||||
* captioning. This handles the common operator workflow of uploading
|
|
||||||
* a PDF before flipping {@code wiki.ocr.enabled} on — without the
|
|
||||||
* skip, the partial text-only result would block the eventual
|
|
||||||
* caption catch-up.
|
|
||||||
*/
|
|
||||||
private record EnrichmentOutcome(String text, boolean shouldCache) { }
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* For PDF raw materials, walks the inline images and appends a section of
|
* For PDF raw materials, walks the inline images and appends a section of
|
||||||
* {@code [图 P{n}#{m}]: <caption>} markers so downstream chunking and
|
* {@code [图 P{n}#{m}]: <caption>} markers so downstream chunking and
|
||||||
* search can index image contents.
|
* search can index image contents.
|
||||||
*
|
*
|
||||||
* <p>Returns an {@link EnrichmentOutcome} that distinguishes three cases:
|
* <p>When {@link #VISION_FLAG_KEY} is off, returns the body unchanged
|
||||||
* <ul>
|
* without re-parsing the PDF — the operator's "off = ignore images"
|
||||||
* <li>Non-PDF / no extractor / no images present → text unchanged,
|
* intent. Flipping the flag on later requires a manual reprocess for
|
||||||
* {@code shouldCache=true}.</li>
|
* existing rows to pick up captions.
|
||||||
* <li>Vision disabled but the PDF has inline images → text unchanged,
|
|
||||||
* {@code shouldCache=false} so the next read retries after the
|
|
||||||
* operator flips {@code wiki.ocr.enabled} on.</li>
|
|
||||||
* <li>Vision enabled and at least one image captioned → enriched text,
|
|
||||||
* {@code shouldCache=true}.</li>
|
|
||||||
* <li>Vision enabled but every image's caption call failed → text
|
|
||||||
* unchanged, {@code shouldCache=true} (transient failures should
|
|
||||||
* not block the cache; if the model is wedged, retry happens via
|
|
||||||
* a manual reprocess).</li>
|
|
||||||
* </ul>
|
|
||||||
*/
|
*/
|
||||||
private EnrichmentOutcome appendPdfImageCaptions(WikiRawMaterialEntity entity, String body) {
|
private String appendPdfImageCaptions(WikiRawMaterialEntity entity, String body) {
|
||||||
if (!"pdf".equals(entity.getSourceType()) || pdfImageExtractor == null) {
|
if (!"pdf".equals(entity.getSourceType()) || pdfImageExtractor == null) {
|
||||||
return new EnrichmentOutcome(body, true);
|
return body;
|
||||||
|
}
|
||||||
|
if (featureFlagService == null || !featureFlagService.isEnabled(VISION_FLAG_KEY)) {
|
||||||
|
return body;
|
||||||
}
|
}
|
||||||
|
|
||||||
java.nio.file.Path pdfPath = java.nio.file.Paths.get(entity.getSourcePath());
|
java.nio.file.Path pdfPath = java.nio.file.Paths.get(entity.getSourcePath());
|
||||||
|
|
||||||
// Vision-disabled path: only refuse the cache when the PDF actually has
|
|
||||||
// qualifying images. Image-less PDFs can still be cached as text-only.
|
|
||||||
boolean visionEnabled = featureFlagService != null
|
|
||||||
&& featureFlagService.isEnabled(VISION_FLAG_KEY);
|
|
||||||
if (!visionEnabled) {
|
|
||||||
boolean hasImages = pdfImageExtractor.hasInlineImages(pdfPath);
|
|
||||||
if (hasImages) {
|
|
||||||
log.info("[Wiki] PDF id={} has inline images but {} is disabled; "
|
|
||||||
+ "skipping captioning and deferring extracted_text cache",
|
|
||||||
entity.getId(), VISION_FLAG_KEY);
|
|
||||||
return new EnrichmentOutcome(body, false);
|
|
||||||
}
|
|
||||||
return new EnrichmentOutcome(body, true);
|
|
||||||
}
|
|
||||||
|
|
||||||
try {
|
try {
|
||||||
List<String> snippets = pdfImageExtractor.captionInlineImages(pdfPath);
|
List<String> snippets = pdfImageExtractor.captionInlineImages(pdfPath);
|
||||||
if (snippets.isEmpty()) {
|
if (snippets.isEmpty()) {
|
||||||
return new EnrichmentOutcome(body, true);
|
return body;
|
||||||
}
|
}
|
||||||
StringBuilder sb = new StringBuilder(body);
|
StringBuilder sb = new StringBuilder(body);
|
||||||
sb.append("\n\n--- Inline images ---\n");
|
sb.append("\n\n--- Inline images ---\n");
|
||||||
for (String snippet : snippets) {
|
for (String snippet : snippets) {
|
||||||
sb.append(snippet).append('\n');
|
sb.append(snippet).append('\n');
|
||||||
}
|
}
|
||||||
return new EnrichmentOutcome(sb.toString(), true);
|
return sb.toString();
|
||||||
} catch (Exception e) {
|
} catch (Exception e) {
|
||||||
log.warn("[Wiki] PDF inline-image captioning failed for id={}: {}",
|
log.warn("[Wiki] PDF inline-image captioning failed for id={}: {}",
|
||||||
entity.getId(), e.getMessage());
|
entity.getId(), e.getMessage());
|
||||||
return new EnrichmentOutcome(body, true);
|
return body;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -479,6 +438,11 @@ public class WikiRawMaterialService {
|
|||||||
log.warn("[Wiki] Image raw material missing sourcePath: id={}", entity.getId());
|
log.warn("[Wiki] Image raw material missing sourcePath: id={}", entity.getId());
|
||||||
return "";
|
return "";
|
||||||
}
|
}
|
||||||
|
if (featureFlagService == null || !featureFlagService.isEnabled(VISION_FLAG_KEY)) {
|
||||||
|
log.info("[Wiki] Image raw id={} skipped: {} is disabled",
|
||||||
|
entity.getId(), VISION_FLAG_KEY);
|
||||||
|
return "";
|
||||||
|
}
|
||||||
byte[] imageBytes;
|
byte[] imageBytes;
|
||||||
try {
|
try {
|
||||||
imageBytes = java.nio.file.Files.readAllBytes(java.nio.file.Paths.get(entity.getSourcePath()));
|
imageBytes = java.nio.file.Files.readAllBytes(java.nio.file.Paths.get(entity.getSourcePath()));
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user