fix(extract): trigger OCR when text extraction returns unreadable bytes

This commit is contained in:
matevip 2026-05-12 08:44:20 +08:00
parent 071ccff9bf
commit dac3f5be20
4 changed files with 105 additions and 27 deletions

View File

@ -213,13 +213,12 @@ public class DocumentExtractTool {
long t0 = System.currentTimeMillis();
String content = tryPdftotext(path, options);
if (content != null && !content.isBlank()) {
if (!needsOcr(content, realPageCount)) {
ExtractionQuality q = classifyExtraction(content, realPageCount);
if (!q.needsOcr()) {
attempts.add("pdftotext: 成功 (" + (System.currentTimeMillis() - t0) + "ms)");
return new ExtractedContent(content, "pdftotext", realPageCount > 0 ? realPageCount : estimatePages(content));
}
double perPage = realPageCount > 0 ? (double) content.strip().length() / realPageCount : 0;
attempts.add("pdftotext: 文本过少 (总 " + content.strip().length() + " 字符, "
+ realPageCount + " 页, 每页 " + String.format("%.0f", perPage) + " 字符),可能是扫描版");
attempts.add("pdftotext: 触发 OCR (" + describeTrigger(q, content.strip().length(), realPageCount) + ")");
bestContent = content;
bestMethod = "pdftotext";
} else {
@ -230,11 +229,12 @@ public class DocumentExtractTool {
long t1 = System.currentTimeMillis();
content = tryPythonPdfExtractor(path, options);
if (content != null && !content.isBlank()) {
if (!needsOcr(content, realPageCount)) {
ExtractionQuality q = classifyExtraction(content, realPageCount);
if (!q.needsOcr()) {
attempts.add("python_pdf: 成功 (" + (System.currentTimeMillis() - t1) + "ms)");
return new ExtractedContent(content, "python_pdfplumber", realPageCount > 0 ? realPageCount : estimatePages(content));
}
attempts.add("python_pdf: 文本过少");
attempts.add("python_pdf: 触发 OCR (" + describeTrigger(q, content.strip().length(), realPageCount) + ")");
if (bestContent == null || content.strip().length() > bestContent.strip().length()) {
bestContent = content;
bestMethod = "python_pdfplumber";
@ -247,11 +247,12 @@ public class DocumentExtractTool {
long t2 = System.currentTimeMillis();
content = extractPdfWithJava(path);
if (content != null && !content.isBlank()) {
if (!needsOcr(content, realPageCount)) {
ExtractionQuality q = classifyExtraction(content, realPageCount);
if (!q.needsOcr()) {
attempts.add("java_pdf: 成功 (" + (System.currentTimeMillis() - t2) + "ms)");
return new ExtractedContent(content, "java_pdfbox", realPageCount > 0 ? realPageCount : estimatePages(content));
}
attempts.add("java_pdf: 文本过少");
attempts.add("java_pdf: 触发 OCR (" + describeTrigger(q, content.strip().length(), realPageCount) + ")");
if (bestContent == null || content.strip().length() > bestContent.strip().length()) {
bestContent = content;
bestMethod = "java_pdfbox";
@ -325,22 +326,99 @@ public class DocumentExtractTool {
return 0; // 未知页数
}
/** Fraction below which extracted text is judged unreadable and an OCR pass is forced. */
static final double READABLE_RATIO_THRESHOLD = 0.5;
/** Outcome of {@link #classifyExtraction}; {@link #trigger()} is {@code null} when usable. */
record ExtractionQuality(String trigger, double readableRatio, double charsPerPage) {
boolean needsOcr() { return trigger != null; }
}
/**
* 判断提取到的文本是否太少需要尝试 OCR
* 使用真实页数来自 getPdfPageCount计算字符密度不再依赖 estimatePages 反推
* 页数未知0只看总字符数
* Classify the quality of a text extraction pass.
* <p>
* Three failure modes can fire an OCR retry:
* <ul>
* <li>{@code empty} / {@code too_short}: nothing extracted, typical of image-only PDFs.</li>
* <li>{@code low_readable_ratio}: extractor returned plenty of characters but most of
* them are control bytes / high-Latin junk typical of CID-encoded fonts without
* a {@code ToUnicode} CMap, where the engine dumps glyph indices as bytes.</li>
* <li>{@code low_char_density}: per-page char count is far below what a real text PDF
* would yield, typical of scanned PDFs with a thin OCR layer applied upstream.</li>
* </ul>
*/
private boolean needsOcr(String text, int realPageCount) {
if (text == null || text.isBlank()) return true;
String stripped = text.strip();
if (stripped.length() < 20) return true;
if (realPageCount <= 0) {
// 页数未知时回退到总字符数判定保守阈值
return stripped.length() < 100;
static ExtractionQuality classifyExtraction(String text, int realPageCount) {
if (text == null || text.isBlank()) {
return new ExtractionQuality("empty", 0.0, 0.0);
}
double perPage = (double) stripped.length() / realPageCount;
// 正常文本 PDF 每页至少数百字符每页不到 30 字符大概率是扫描版
return perPage < 30;
String stripped = text.strip();
if (stripped.length() < 20) {
return new ExtractionQuality("too_short", 0.0, 0.0);
}
double ratio = readableRatio(stripped);
double perPage = realPageCount > 0
? (double) stripped.length() / realPageCount
: stripped.length();
if (ratio < READABLE_RATIO_THRESHOLD) {
return new ExtractionQuality("low_readable_ratio", ratio, perPage);
}
if (realPageCount <= 0) {
// Page count unknown fall back to a conservative total-length cutoff.
if (stripped.length() < 100) {
return new ExtractionQuality("too_short", ratio, perPage);
}
} else if (perPage < 30) {
return new ExtractionQuality("low_char_density", ratio, perPage);
}
return new ExtractionQuality(null, ratio, perPage);
}
/**
* Fraction of code points that are obviously readable: ASCII printable, tab/newline,
* CJK Unified Ideographs (+ ext A), CJK punctuation, halfwidth/fullwidth forms,
* hiragana/katakana, hangul syllables. Returns 0 for empty input.
* <p>
* The threshold {@link #READABLE_RATIO_THRESHOLD} separates real-world noisy
* extraction (well above 0.7 even with OCR errors) from font-encoding garbage,
* which typically lands below 0.1 because the bytes fall outside every script range.
*/
static double readableRatio(String text) {
if (text == null || text.isEmpty()) return 0.0;
int total = 0, good = 0;
for (int i = 0; i < text.length(); ) {
int cp = text.codePointAt(i);
i += Character.charCount(cp);
total++;
if (isReadable(cp)) good++;
}
return total == 0 ? 0.0 : (double) good / total;
}
/** Compact one-line summary of why an extraction was rejected, for the attempts log. */
private static String describeTrigger(ExtractionQuality q, int totalChars, int realPageCount) {
return switch (q.trigger()) {
case "low_readable_ratio" -> String.format(
"readable=%.2f<%.2f, %d 字符多为非可读字节,可能是字体编码异常",
q.readableRatio(), READABLE_RATIO_THRESHOLD, totalChars);
case "low_char_density" -> String.format(
"每页 %.0f 字符(总 %d, %d 页),可能是扫描版",
q.charsPerPage(), totalChars, realPageCount);
case "too_short" -> "" + totalChars + " 字符,文本过少";
case "empty" -> "提取结果为空";
default -> "trigger=" + q.trigger();
};
}
private static boolean isReadable(int cp) {
if (cp == 9 || cp == 10 || cp == 13) return true;
if (cp >= 0x20 && cp <= 0x7E) return true; // ASCII printable
if (cp >= 0x3000 && cp <= 0x303F) return true; // CJK punctuation
if (cp >= 0x3040 && cp <= 0x30FF) return true; // hiragana / katakana
if (cp >= 0x3400 && cp <= 0x4DBF) return true; // CJK ext A
if (cp >= 0x4E00 && cp <= 0x9FFF) return true; // CJK unified
if (cp >= 0xAC00 && cp <= 0xD7AF) return true; // hangul syllables
if (cp >= 0xFF00 && cp <= 0xFFEF) return true; // halfwidth / fullwidth
return false;
}
/** OCR 结果(含成功/失败页数统计) */

View File

@ -147,7 +147,7 @@ err.auth.wrong_password=\u539f\u5bc6\u7801\u9519\u8bef
err.agent.not_found=Agent\u4e0d\u5b58\u5728
err.agent.disabled=Agent \u5df2\u7981\u7528
err.agent.name_required=Agent \u540d\u79f0\u4e0d\u80fd\u4e3a\u7a7a
err.agent.duplicate_name=\u5f53\u524d\u5de5\u4f5c\u533a\u5df2\u5b58\u5728\u540c\u540d Agent
err.agent.duplicate_name=\u5f53\u524d\u5de5\u4f5c\u533a\u5df2\u5b58\u5728\u540c\u540d\u5458\u5de5\uff0c\u8bf7\u6362\u4e2a\u540d\u5b57\u518d\u8bd5
err.workspace.not_found=\u5de5\u4f5c\u533a\u4e0d\u5b58\u5728
err.workspace.slug_exists=\u5de5\u4f5c\u533a\u6807\u8bc6\u5df2\u5b58\u5728
err.workspace.cannot_modify_default=\u4e0d\u80fd\u4fee\u6539\u9ed8\u8ba4\u5de5\u4f5c\u533a\u7684\u6807\u8bc6

View File

@ -153,7 +153,7 @@ err.auth.wrong_password=Incorrect current password
err.agent.not_found=Agent not found
err.agent.disabled=Agent is disabled
err.agent.name_required=Agent name is required
err.agent.duplicate_name=An Agent with this name already exists in the current workspace
err.agent.duplicate_name=An employee with this name already exists in this workspace — try a different name
# workspace
err.workspace.not_found=Workspace not found
err.workspace.slug_exists=Workspace slug already exists

View File

@ -792,8 +792,8 @@ async function applyTemplate(id: string) {
ElMessage.success(t('agents.templates.applied'))
showTemplateSelector.value = false
await loadAgents()
} catch {
ElMessage.error(t('agents.messages.saveFailed'))
} catch (e: any) {
ElMessage.error(e?.message || t('agents.messages.saveFailed'))
} finally {
applyingTemplate.value = false
}
@ -887,8 +887,8 @@ async function saveAgent() {
ElMessage.success(t('agents.messages.saveSuccess'))
closeModal()
await loadAgents()
} catch {
ElMessage.error(t('agents.messages.saveFailed'))
} catch (e: any) {
ElMessage.error(e?.message || t('agents.messages.saveFailed'))
}
}