From 100cbe54532935bc6bfb4e79b302a39bbb430e6b Mon Sep 17 00:00:00 2001 From: Abimael Martell Date: Tue, 14 Apr 2026 21:11:41 -0700 Subject: [PATCH] fix: remove template image influence from page classification Template images (large background/figure images) no longer affect pages_needing_ocr. In the region-based pipeline, text regions are extracted independently from image regions, and per-region needs_ocr quality checks handle scanned-with-OCR garbage text. Also makes the invisible text retry (for OCR text layers) trigger on text quality rather than PDF type, so it works regardless of classification. Co-Authored-By: Claude Opus 4.6 (1M context) --- src/detector.rs | 52 ++++++++++++++----------------------------------- src/lib.rs | 32 +++++++++++++----------------- 2 files changed, 29 insertions(+), 55 deletions(-) diff --git a/src/detector.rs b/src/detector.rs index e3151a4..b14bb7f 100644 --- a/src/detector.rs +++ b/src/detector.rs @@ -185,7 +185,6 @@ pub(crate) fn detect_from_document( let mut pages_with_text = 0u32; let mut pages_with_images = 0u32; - let mut pages_with_template_images = 0u32; let mut pages_with_vector_text = 0u32; let mut total_text_ops = 0u32; // Cache Phase 1 results to avoid re-analyzing sampled pages in Phase 2 @@ -223,17 +222,13 @@ pub(crate) fn detect_from_document( if analysis.has_images { pages_with_images += 1; } - // Only count as a template-image page if it looks like a scan - // (single full-page image) rather than a text page with figures. - // Scanned-with-OCR PDFs have 1 large image per page + OCR text overlay; - // text PDFs with figures have multiple smaller images alongside real text. - if analysis.has_template_image - && (analysis.image_count <= 1 - || analysis.text_operator_count < 50 - || analysis.unique_alphanum_chars < 10) - { - pages_with_template_images += 1; - } + // Template images (large background/figure images) no longer + // influence page classification. In the region-based pipeline, + // text regions are extracted independently from image regions, + // and per-region `needs_ocr` quality checks handle garbage text + // from scanned-with-OCR pages. Counting template images here + // caused false OCR recommendations for text PDFs with figures + // (e.g. academic papers, reports with charts). if analysis.has_vector_text { pages_with_vector_text += 1; } @@ -260,26 +255,13 @@ pub(crate) fn detect_from_document( 0.0 }; - // Check if this is a template-based PDF (images provide essential context) - // Template PDFs have text AND large background images on most pages - let has_template_images = pages_with_template_images > 0; - let template_ratio = if pages_sampled > 0 { - pages_with_template_images as f32 / pages_sampled as f32 - } else { - 0.0 - }; - - // OCR is recommended when: - // 1. Template images are present (text alone is insufficient), OR - // 2. PDF is scanned/image-based + // Classification logic. + // Template images no longer influence classification — in the region-based + // pipeline, text regions are extracted independently and per-region + // `needs_ocr` quality checks handle scanned-with-OCR garbage text. let ocr_recommended: bool; - // Classification logic - let (pdf_type, confidence) = if has_template_images && pages_with_text > 0 { - ocr_recommended = true; - // Template-based PDF: has text but images provide essential context - (PdfType::Mixed, 0.5 + (0.3 * (1.0 - template_ratio))) - } else if text_ratio >= config.text_page_ratio_threshold { + let (pdf_type, confidence) = if text_ratio >= config.text_page_ratio_threshold { ocr_recommended = false; (PdfType::TextBased, text_ratio) } else if pages_with_text == 0 && (pages_with_images > 0 || pages_with_vector_text > 0) { @@ -358,13 +340,9 @@ pub(crate) fn detect_from_document( } else { continue; }; - // Template images only need OCR when it looks like a scan - // (single full-page image) rather than figures alongside text. - let looks_like_scan = analysis.image_count <= 1 - || analysis.text_operator_count < 50 - || analysis.unique_alphanum_chars < 10; - if (analysis.has_template_image && looks_like_scan) - || analysis.has_vector_text + // Template images no longer trigger OCR — per-region + // quality checks handle scanned-with-OCR garbage text. + if analysis.has_vector_text || (analysis.text_operator_count < config.min_text_ops_per_page && analysis.has_images) { diff --git a/src/lib.rs b/src/lib.rs index afc7161..6615f65 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1023,31 +1023,27 @@ fn process_document( options.page_filter.as_ref(), ); - // For Mixed/template PDFs: if normal extraction produces garbage text - // (mostly non-alphanumeric), retry with invisible (Tr=3) text included. - // This unlocks OCR text layers behind scanned images. - if pdf_type == PdfType::Mixed { - if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) { - let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect(); - if is_garbage_text(&sample) || sample.trim().is_empty() { - extractor::extract_positioned_text_include_invisible( - &doc, - &font_cmaps, - options.page_filter.as_ref(), - ) - } else { - result - } - } else { - // Normal extraction failed — try invisible as fallback + // If normal extraction produces garbage text or is empty, retry with + // invisible (Tr=3) text included. This unlocks OCR text layers behind + // scanned images regardless of PDF type classification. + if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) { + let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect(); + if is_garbage_text(&sample) || sample.trim().is_empty() { extractor::extract_positioned_text_include_invisible( &doc, &font_cmaps, options.page_filter.as_ref(), ) + } else { + result } } else { - result + // Normal extraction failed — try invisible as fallback + extractor::extract_positioned_text_include_invisible( + &doc, + &font_cmaps, + options.page_filter.as_ref(), + ) } };