feat(wiki): caption inline PDF images via the vision-in pipeline

This commit is contained in:
matevip 2026-05-02 19:04:45 +08:00
parent 4dc280a3d1
commit 51995275bd
3 changed files with 238 additions and 4 deletions

View File

@ -404,6 +404,15 @@
<artifactId>jgrapht-core</artifactId>
<version>1.5.2</version>
</dependency>
<!-- PDF parsing for inline image extraction (wiki vision-in pipeline).
Used to walk PDPage resources and pull out PDImageXObject instances
for downstream captioning. -->
<dependency>
<groupId>org.apache.pdfbox</groupId>
<artifactId>pdfbox</artifactId>
<version>3.0.3</version>
</dependency>
</dependencies>
<build>

View File

@ -0,0 +1,187 @@
package vip.mate.wiki.service;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDResources;
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.stereotype.Component;
import vip.mate.tool.image.vision.ImageVisionService;
import vip.mate.tool.image.vision.VisionRequest;
import vip.mate.tool.image.vision.VisionResult;
import javax.imageio.ImageIO;
import java.awt.image.BufferedImage;
import java.io.ByteArrayOutputStream;
import java.io.File;
import java.io.IOException;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
/**
* Extracts inline images from a PDF and runs each through the vision-in
* pipeline to produce a {@code [ P{page}#{idx}]: <caption>} marker line.
*
* <p>Constants follow standard image-extractor conventions:
* <ul>
* <li>Min image side: 100 px (filters favicon-sized logos and dividers)</li>
* <li>Max images per PDF: 50 (vision API cost guard; tunable in a follow-up
* once production cost data is in)</li>
* </ul>
*
* <p>Output is appended to the compile pipeline's main text body so chunks
* become searchable by their image content (e.g. searching "营收" finds
* pages with revenue charts even if the figure caption is below the chart).
*
* @author MateClaw Team
*/
@Slf4j
@Component
@RequiredArgsConstructor
public class PdfImageExtractor {
private static final int MIN_IMAGE_SIDE_PX = 100;
private static final int MAX_IMAGES_PER_PDF = 50;
private final ImageVisionService imageVisionService;
/**
* Optional dependency: when {@link #captionInlineImages(Path)} is called
* but no vision service is wired (test harness, vision module excluded),
* the extractor returns an empty list rather than NPE-ing.
*/
@Autowired(required = false)
private void noopWhenVisionAbsent() {
// Marker method; the @RequiredArgsConstructor field is required at
// runtime in production. Tests construct the extractor directly with
// a mock or a no-op wrapper.
}
/**
* Walks every page of {@code pdfPath}, captions each qualifying inline
* image, and returns the rendered marker lines in document order.
*
* <p>Returns an empty list when the PDF can't be opened, contains no
* inline images, or all images fall below the size threshold. Per-image
* vision failures are logged and skipped; the rest of the document is
* still processed.
*/
public List<String> captionInlineImages(Path pdfPath) {
if (pdfPath == null) {
return List.of();
}
File pdfFile = pdfPath.toFile();
if (!pdfFile.isFile()) {
log.warn("[PdfImage] PDF path is not a regular file: {}", pdfPath);
return List.of();
}
List<String> snippets = new ArrayList<>();
int totalCaptioned = 0;
try (PDDocument doc = Loader.loadPDF(pdfFile)) {
int pageIdx = 0;
for (PDPage page : doc.getPages()) {
pageIdx++;
if (totalCaptioned >= MAX_IMAGES_PER_PDF) {
log.info("[PdfImage] hit max images cap ({}) for {}; stopping early",
MAX_IMAGES_PER_PDF, pdfPath);
break;
}
PDResources resources = page.getResources();
if (resources == null) {
continue;
}
int imgIdx = 0;
for (COSName xobjectName : resources.getXObjectNames()) {
if (totalCaptioned >= MAX_IMAGES_PER_PDF) {
break;
}
PDXObject xobject;
try {
xobject = resources.getXObject(xobjectName);
} catch (IOException ioe) {
log.debug("[PdfImage] xobject load failed at page={} name={}: {}",
pageIdx, xobjectName.getName(), ioe.getMessage());
continue;
}
if (!(xobject instanceof PDImageXObject pdImage)) {
continue;
}
BufferedImage bufferedImage;
try {
bufferedImage = pdImage.getImage();
} catch (IOException | RuntimeException e) {
log.debug("[PdfImage] decode failed at page={} idx={}: {}",
pageIdx, imgIdx + 1, e.getMessage());
continue;
}
if (bufferedImage.getWidth() < MIN_IMAGE_SIDE_PX
|| bufferedImage.getHeight() < MIN_IMAGE_SIDE_PX) {
continue;
}
imgIdx++;
totalCaptioned++;
byte[] pngBytes;
try {
pngBytes = encodeAsPng(bufferedImage);
} catch (IOException ioe) {
log.debug("[PdfImage] PNG encode failed at page={} idx={}: {}",
pageIdx, imgIdx, ioe.getMessage());
continue;
}
String caption = captionOrNull(pngBytes);
if (caption == null || caption.isBlank()) {
log.debug("[PdfImage] vision returned no caption page={} idx={}",
pageIdx, imgIdx);
continue;
}
snippets.add(String.format("[图 P%d#%d]: %s", pageIdx, imgIdx, caption));
}
}
} catch (IOException e) {
log.warn("[PdfImage] PDF load failed for {}: {}", pdfPath, e.getMessage());
}
log.info("[PdfImage] captioned {} image(s) from {}", snippets.size(), pdfPath);
return snippets;
}
private String captionOrNull(byte[] pngBytes) {
if (imageVisionService == null) {
return null;
}
try {
VisionResult result = imageVisionService.caption(VisionRequest.builder()
.imageBytes(pngBytes)
.mimeType("image/png")
.build());
return result == null ? null : result.getCaption();
} catch (Exception e) {
// Non-fatal: per-image vision failures should not abort the rest of the doc.
log.debug("[PdfImage] vision call failed: {}", e.getMessage());
return null;
}
}
private static byte[] encodeAsPng(BufferedImage image) throws IOException {
try (ByteArrayOutputStream baos = new ByteArrayOutputStream()) {
ImageIO.write(image, "png", baos);
return baos.toByteArray();
}
}
}

View File

@ -42,6 +42,7 @@ public class WikiRawMaterialService {
/** RFC-013删除时级联清理 chunk */
private final WikiChunkService chunkService;
private final ImageVisionService imageVisionService;
private final PdfImageExtractor pdfImageExtractor;
/**
* RFC-012 follow-up #3 partial 状态触发的 reprocess 会在此 set 中打标
@ -352,17 +353,22 @@ public class WikiRawMaterialService {
String text = json.getStr("text");
if (text != null && !text.isBlank()) {
boolean truncated = json.getBool("truncated", false);
// Append inline-image captions for PDFs so chunk-level search
// hits chart/diagram contents that the text extractor missed.
// Failures are non-fatal: the body text is still returned.
String enriched = appendPdfImageCaptions(entity, text);
if (truncated) {
// 截断的结果不缓存避免永久丢失后半内容返回文本供分块处理使用
log.warn("[Wiki] Extracted text truncated at {} chars for: {} (full document may be larger)",
text.length(), entity.getSourcePath());
} else {
// Full extraction: cache to avoid re-extracting on subsequent calls.
updateExtractedText(entity.getId(), text);
updateExtractedText(entity.getId(), enriched);
}
log.info("[Wiki] Extracted text from {}: {} chars, method={}, truncated={}",
entity.getSourcePath(), text.length(), json.getStr("method"), truncated);
return text;
log.info("[Wiki] Extracted text from {}: {} chars (text) → {} chars (enriched), method={}, truncated={}",
entity.getSourcePath(), text.length(), enriched.length(), json.getStr("method"), truncated);
return enriched;
}
}
log.warn("[Wiki] Document extraction returned no text for: {}", entity.getSourcePath());
@ -373,6 +379,38 @@ public class WikiRawMaterialService {
return entity.getOriginalContent();
}
/**
* For PDF raw materials, walks the inline images and appends a section of
* {@code [ P{n}#{m}]: <caption>} markers so downstream chunking and
* search can index image contents. No-op for non-PDF source types and
* when the vision pipeline is unavailable.
*/
private String appendPdfImageCaptions(WikiRawMaterialEntity entity, String body) {
if (!"pdf".equals(entity.getSourceType())) {
return body;
}
if (pdfImageExtractor == null) {
return body;
}
try {
List<String> snippets = pdfImageExtractor.captionInlineImages(
java.nio.file.Paths.get(entity.getSourcePath()));
if (snippets.isEmpty()) {
return body;
}
StringBuilder sb = new StringBuilder(body);
sb.append("\n\n--- Inline images ---\n");
for (String snippet : snippets) {
sb.append(snippet).append('\n');
}
return sb.toString();
} catch (Exception e) {
log.warn("[Wiki] PDF inline-image captioning failed for id={}: {}",
entity.getId(), e.getMessage());
return body;
}
}
/**
* Routes an image-typed raw material through the vision-in pipeline,
* caches the resulting caption into {@code extracted_text} so the next