} marker line.
+ *
+ * Constants follow standard image-extractor conventions:
+ *
+ * - Min image side: 100 px (filters favicon-sized logos and dividers)
+ * - Max images per PDF: 50 (vision API cost guard; tunable in a follow-up
+ * once production cost data is in)
+ *
+ *
+ * Output is appended to the compile pipeline's main text body so chunks
+ * become searchable by their image content (e.g. searching "营收" finds
+ * pages with revenue charts even if the figure caption is below the chart).
+ *
+ * @author MateClaw Team
+ */
+@Slf4j
+@Component
+@RequiredArgsConstructor
+public class PdfImageExtractor {
+
+ private static final int MIN_IMAGE_SIDE_PX = 100;
+ private static final int MAX_IMAGES_PER_PDF = 50;
+
+ private final ImageVisionService imageVisionService;
+
+ /**
+ * Optional dependency: when {@link #captionInlineImages(Path)} is called
+ * but no vision service is wired (test harness, vision module excluded),
+ * the extractor returns an empty list rather than NPE-ing.
+ */
+ @Autowired(required = false)
+ private void noopWhenVisionAbsent() {
+ // Marker method; the @RequiredArgsConstructor field is required at
+ // runtime in production. Tests construct the extractor directly with
+ // a mock or a no-op wrapper.
+ }
+
+ /**
+ * Walks every page of {@code pdfPath}, captions each qualifying inline
+ * image, and returns the rendered marker lines in document order.
+ *
+ *
Returns an empty list when the PDF can't be opened, contains no
+ * inline images, or all images fall below the size threshold. Per-image
+ * vision failures are logged and skipped; the rest of the document is
+ * still processed.
+ */
+ public List captionInlineImages(Path pdfPath) {
+ if (pdfPath == null) {
+ return List.of();
+ }
+ File pdfFile = pdfPath.toFile();
+ if (!pdfFile.isFile()) {
+ log.warn("[PdfImage] PDF path is not a regular file: {}", pdfPath);
+ return List.of();
+ }
+
+ List snippets = new ArrayList<>();
+ int totalCaptioned = 0;
+
+ try (PDDocument doc = Loader.loadPDF(pdfFile)) {
+ int pageIdx = 0;
+ for (PDPage page : doc.getPages()) {
+ pageIdx++;
+ if (totalCaptioned >= MAX_IMAGES_PER_PDF) {
+ log.info("[PdfImage] hit max images cap ({}) for {}; stopping early",
+ MAX_IMAGES_PER_PDF, pdfPath);
+ break;
+ }
+
+ PDResources resources = page.getResources();
+ if (resources == null) {
+ continue;
+ }
+
+ int imgIdx = 0;
+ for (COSName xobjectName : resources.getXObjectNames()) {
+ if (totalCaptioned >= MAX_IMAGES_PER_PDF) {
+ break;
+ }
+
+ PDXObject xobject;
+ try {
+ xobject = resources.getXObject(xobjectName);
+ } catch (IOException ioe) {
+ log.debug("[PdfImage] xobject load failed at page={} name={}: {}",
+ pageIdx, xobjectName.getName(), ioe.getMessage());
+ continue;
+ }
+ if (!(xobject instanceof PDImageXObject pdImage)) {
+ continue;
+ }
+
+ BufferedImage bufferedImage;
+ try {
+ bufferedImage = pdImage.getImage();
+ } catch (IOException | RuntimeException e) {
+ log.debug("[PdfImage] decode failed at page={} idx={}: {}",
+ pageIdx, imgIdx + 1, e.getMessage());
+ continue;
+ }
+ if (bufferedImage.getWidth() < MIN_IMAGE_SIDE_PX
+ || bufferedImage.getHeight() < MIN_IMAGE_SIDE_PX) {
+ continue;
+ }
+
+ imgIdx++;
+ totalCaptioned++;
+
+ byte[] pngBytes;
+ try {
+ pngBytes = encodeAsPng(bufferedImage);
+ } catch (IOException ioe) {
+ log.debug("[PdfImage] PNG encode failed at page={} idx={}: {}",
+ pageIdx, imgIdx, ioe.getMessage());
+ continue;
+ }
+
+ String caption = captionOrNull(pngBytes);
+ if (caption == null || caption.isBlank()) {
+ log.debug("[PdfImage] vision returned no caption page={} idx={}",
+ pageIdx, imgIdx);
+ continue;
+ }
+
+ snippets.add(String.format("[图 P%d#%d]: %s", pageIdx, imgIdx, caption));
+ }
+ }
+ } catch (IOException e) {
+ log.warn("[PdfImage] PDF load failed for {}: {}", pdfPath, e.getMessage());
+ }
+
+ log.info("[PdfImage] captioned {} image(s) from {}", snippets.size(), pdfPath);
+ return snippets;
+ }
+
+ private String captionOrNull(byte[] pngBytes) {
+ if (imageVisionService == null) {
+ return null;
+ }
+ try {
+ VisionResult result = imageVisionService.caption(VisionRequest.builder()
+ .imageBytes(pngBytes)
+ .mimeType("image/png")
+ .build());
+ return result == null ? null : result.getCaption();
+ } catch (Exception e) {
+ // Non-fatal: per-image vision failures should not abort the rest of the doc.
+ log.debug("[PdfImage] vision call failed: {}", e.getMessage());
+ return null;
+ }
+ }
+
+ private static byte[] encodeAsPng(BufferedImage image) throws IOException {
+ try (ByteArrayOutputStream baos = new ByteArrayOutputStream()) {
+ ImageIO.write(image, "png", baos);
+ return baos.toByteArray();
+ }
+ }
+}
diff --git a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java
index 4baaf2d5..d4b1a746 100644
--- a/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java
+++ b/mateclaw-server/src/main/java/vip/mate/wiki/service/WikiRawMaterialService.java
@@ -42,6 +42,7 @@ public class WikiRawMaterialService {
/** RFC-013:删除时级联清理 chunk */
private final WikiChunkService chunkService;
private final ImageVisionService imageVisionService;
+ private final PdfImageExtractor pdfImageExtractor;
/**
* RFC-012 follow-up #3:从 partial 状态触发的 reprocess 会在此 set 中打标,
@@ -352,17 +353,22 @@ public class WikiRawMaterialService {
String text = json.getStr("text");
if (text != null && !text.isBlank()) {
boolean truncated = json.getBool("truncated", false);
+ // Append inline-image captions for PDFs so chunk-level search
+ // hits chart/diagram contents that the text extractor missed.
+ // Failures are non-fatal: the body text is still returned.
+ String enriched = appendPdfImageCaptions(entity, text);
+
if (truncated) {
// 截断的结果不缓存,避免永久丢失后半内容。返回文本供分块处理使用。
log.warn("[Wiki] Extracted text truncated at {} chars for: {} (full document may be larger)",
text.length(), entity.getSourcePath());
} else {
// Full extraction: cache to avoid re-extracting on subsequent calls.
- updateExtractedText(entity.getId(), text);
+ updateExtractedText(entity.getId(), enriched);
}
- log.info("[Wiki] Extracted text from {}: {} chars, method={}, truncated={}",
- entity.getSourcePath(), text.length(), json.getStr("method"), truncated);
- return text;
+ log.info("[Wiki] Extracted text from {}: {} chars (text) → {} chars (enriched), method={}, truncated={}",
+ entity.getSourcePath(), text.length(), enriched.length(), json.getStr("method"), truncated);
+ return enriched;
}
}
log.warn("[Wiki] Document extraction returned no text for: {}", entity.getSourcePath());
@@ -373,6 +379,38 @@ public class WikiRawMaterialService {
return entity.getOriginalContent();
}
+ /**
+ * For PDF raw materials, walks the inline images and appends a section of
+ * {@code [图 P{n}#{m}]: } markers so downstream chunking and
+ * search can index image contents. No-op for non-PDF source types and
+ * when the vision pipeline is unavailable.
+ */
+ private String appendPdfImageCaptions(WikiRawMaterialEntity entity, String body) {
+ if (!"pdf".equals(entity.getSourceType())) {
+ return body;
+ }
+ if (pdfImageExtractor == null) {
+ return body;
+ }
+ try {
+ List snippets = pdfImageExtractor.captionInlineImages(
+ java.nio.file.Paths.get(entity.getSourcePath()));
+ if (snippets.isEmpty()) {
+ return body;
+ }
+ StringBuilder sb = new StringBuilder(body);
+ sb.append("\n\n--- Inline images ---\n");
+ for (String snippet : snippets) {
+ sb.append(snippet).append('\n');
+ }
+ return sb.toString();
+ } catch (Exception e) {
+ log.warn("[Wiki] PDF inline-image captioning failed for id={}: {}",
+ entity.getId(), e.getMessage());
+ return body;
+ }
+ }
+
/**
* Routes an image-typed raw material through the vision-in pipeline,
* caches the resulting caption into {@code extracted_text} so the next