fix(detect): classify tiled scans with garbage OCR as Scanned
Scanned PDFs with JBIG2/tiled image strips were misclassified as TextBased because no individual image tile exceeded the template threshold. Now checks aggregate image area per page (≥2M pixels). Also adds is_garbage_text() check: if a Mixed/template PDF's extracted text is predominantly non-alphanumeric (<50%), upgrade to Scanned so callers use proper OCR instead of the garbage text layer. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
442a169ddf
commit
3ff40987d6
+19
-4
@@ -197,10 +197,11 @@ pub(crate) fn detect_from_document(
|
|||||||
let analysis = analyze_page_content(doc, page_id);
|
let analysis = analyze_page_content(doc, page_id);
|
||||||
pages_actually_sampled += 1;
|
pages_actually_sampled += 1;
|
||||||
log::debug!(
|
log::debug!(
|
||||||
"page {}: text_ops={} images={} image_count={} template={} unique_chars={} path_ops={} vector_text={} image_area={}",
|
"page {}: text_ops={} images={} image_count={} template={} unique_chars={} alphanum={} path_ops={} vector_text={} image_area={}",
|
||||||
page_num, analysis.text_operator_count, analysis.has_images,
|
page_num, analysis.text_operator_count, analysis.has_images,
|
||||||
analysis.image_count, analysis.has_template_image,
|
analysis.image_count, analysis.has_template_image,
|
||||||
analysis.unique_text_chars, analysis.path_op_count, analysis.has_vector_text,
|
analysis.unique_text_chars, analysis.unique_alphanum_chars,
|
||||||
|
analysis.path_op_count, analysis.has_vector_text,
|
||||||
analysis.total_image_area
|
analysis.total_image_area
|
||||||
);
|
);
|
||||||
let is_image_dominated = analysis.image_count > 10
|
let is_image_dominated = analysis.image_count > 10
|
||||||
@@ -265,9 +266,8 @@ pub(crate) fn detect_from_document(
|
|||||||
|
|
||||||
// Classification logic
|
// Classification logic
|
||||||
let (pdf_type, confidence) = if has_template_images && pages_with_text > 0 {
|
let (pdf_type, confidence) = if has_template_images && pages_with_text > 0 {
|
||||||
// Template-based PDF: has text but images provide essential context
|
|
||||||
// Classify as Mixed with lower confidence
|
|
||||||
ocr_recommended = true;
|
ocr_recommended = true;
|
||||||
|
// Template-based PDF: has text but images provide essential context
|
||||||
(PdfType::Mixed, 0.5 + (0.3 * (1.0 - template_ratio)))
|
(PdfType::Mixed, 0.5 + (0.3 * (1.0 - template_ratio)))
|
||||||
} else if text_ratio >= config.text_page_ratio_threshold {
|
} else if text_ratio >= config.text_page_ratio_threshold {
|
||||||
ocr_recommended = false;
|
ocr_recommended = false;
|
||||||
@@ -383,6 +383,8 @@ struct PageAnalysis {
|
|||||||
image_count: u32,
|
image_count: u32,
|
||||||
/// Number of unique non-whitespace text characters found in string operands
|
/// Number of unique non-whitespace text characters found in string operands
|
||||||
unique_text_chars: u32,
|
unique_text_chars: u32,
|
||||||
|
/// Number of unique ASCII alphanumeric bytes (letters + digits) in string operands
|
||||||
|
unique_alphanum_chars: u32,
|
||||||
/// Number of path construction/painting ops (m, l, c, h, f, re, etc.)
|
/// Number of path construction/painting ops (m, l, c, h, f, re, etc.)
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code)]
|
||||||
path_op_count: u32,
|
path_op_count: u32,
|
||||||
@@ -454,6 +456,11 @@ fn analyze_page_content(doc: &Document, page_id: ObjectId) -> PageAnalysis {
|
|||||||
// outlined text produces thousands of path ops.
|
// outlined text produces thousands of path ops.
|
||||||
let has_vector_text = path_ops >= 1000 && path_ops > text_ops.saturating_mul(200);
|
let has_vector_text = path_ops >= 1000 && path_ops > text_ops.saturating_mul(200);
|
||||||
|
|
||||||
|
let unique_alphanum_chars = all_unique_chars
|
||||||
|
.iter()
|
||||||
|
.filter(|b| b.is_ascii_alphanumeric())
|
||||||
|
.count() as u32;
|
||||||
|
|
||||||
PageAnalysis {
|
PageAnalysis {
|
||||||
text_operator_count: text_ops,
|
text_operator_count: text_ops,
|
||||||
has_images,
|
has_images,
|
||||||
@@ -461,6 +468,7 @@ fn analyze_page_content(doc: &Document, page_id: ObjectId) -> PageAnalysis {
|
|||||||
total_image_area,
|
total_image_area,
|
||||||
image_count,
|
image_count,
|
||||||
unique_text_chars: all_unique_chars.len() as u32,
|
unique_text_chars: all_unique_chars.len() as u32,
|
||||||
|
unique_alphanum_chars,
|
||||||
path_op_count: path_ops,
|
path_op_count: path_ops,
|
||||||
has_vector_text,
|
has_vector_text,
|
||||||
}
|
}
|
||||||
@@ -801,6 +809,13 @@ fn analyze_page_images(doc: &Document, page_id: ObjectId) -> (bool, u64, bool) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Tiled scans: many small image tiles (e.g., JBIG2 strips) that together
|
||||||
|
// cover the full page. No individual tile triggers the template threshold,
|
||||||
|
// but the aggregate area clearly indicates a scanned/image-backed page.
|
||||||
|
if !has_template_image && total_area >= TEMPLATE_IMAGE_THRESHOLD * 4 {
|
||||||
|
has_template_image = true;
|
||||||
|
}
|
||||||
|
|
||||||
(has_images, total_area, has_template_image)
|
(has_images, total_area, has_template_image)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+36
@@ -391,6 +391,16 @@ fn process_document(
|
|||||||
None => (None, LayoutComplexity::default(), false),
|
None => (None, LayoutComplexity::default(), false),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// If the extracted text is predominantly garbage (non-alphanumeric) and
|
||||||
|
// the PDF is image-backed (Mixed/template), upgrade to Scanned — the text
|
||||||
|
// layer comes from a bad OCR pass, and callers should use proper OCR.
|
||||||
|
let (pdf_type, markdown, confidence) =
|
||||||
|
if pdf_type == PdfType::Mixed && markdown.as_ref().is_some_and(|m| is_garbage_text(m)) {
|
||||||
|
(PdfType::Scanned, None, 0.95)
|
||||||
|
} else {
|
||||||
|
(pdf_type, markdown, confidence)
|
||||||
|
};
|
||||||
|
|
||||||
Ok(PdfProcessResult {
|
Ok(PdfProcessResult {
|
||||||
pdf_type,
|
pdf_type,
|
||||||
markdown,
|
markdown,
|
||||||
@@ -444,6 +454,32 @@ fn detect_encoding_issues(markdown: &str) -> bool {
|
|||||||
false
|
false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Check if extracted text is predominantly garbage (non-alphanumeric).
|
||||||
|
///
|
||||||
|
/// Broken font encodings produce text like "----1-.-.-.___ --.-. .._ I_---."
|
||||||
|
/// where most characters are punctuation/symbols. Real text in any language
|
||||||
|
/// has >50% alphanumeric characters.
|
||||||
|
fn is_garbage_text(markdown: &str) -> bool {
|
||||||
|
let mut alphanum = 0usize;
|
||||||
|
let mut non_alphanum = 0usize;
|
||||||
|
for ch in markdown.chars() {
|
||||||
|
if ch.is_whitespace() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// Skip markdown syntax chars that we add (not from the PDF)
|
||||||
|
if matches!(ch, '#' | '*' | '|' | '-' | '\n') {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if ch.is_alphanumeric() {
|
||||||
|
alphanum += 1;
|
||||||
|
} else {
|
||||||
|
non_alphanum += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let total = alphanum + non_alphanum;
|
||||||
|
total >= 50 && alphanum * 2 < total
|
||||||
|
}
|
||||||
|
|
||||||
/// Analyse extracted items and rects for layout complexity.
|
/// Analyse extracted items and rects for layout complexity.
|
||||||
fn compute_layout_complexity(
|
fn compute_layout_complexity(
|
||||||
items: &[types::TextItem],
|
items: &[types::TextItem],
|
||||||
|
|||||||
Reference in New Issue
Block a user