diff --git a/src/bin/detect_pdf.rs b/src/bin/detect_pdf.rs index 59792c4..1294970 100644 --- a/src/bin/detect_pdf.rs +++ b/src/bin/detect_pdf.rs @@ -24,8 +24,13 @@ fn main() { let elapsed = start.elapsed(); if json_output { + let ocr_pages: Vec = result + .pages_needing_ocr + .iter() + .map(|p| p.to_string()) + .collect(); println!( - r#"{{"pdf_type":"{}","page_count":{},"pages_sampled":{},"pages_with_text":{},"confidence":{:.2},"title":{},"ocr_recommended":{},"detection_time_ms":{}}}"#, + r#"{{"pdf_type":"{}","page_count":{},"pages_sampled":{},"pages_with_text":{},"confidence":{:.2},"title":{},"ocr_recommended":{},"pages_needing_ocr":[{}],"detection_time_ms":{}}}"#, match result.pdf_type { PdfType::TextBased => "text_based", PdfType::Scanned => "scanned", @@ -42,6 +47,7 @@ fn main() { .map(|t| format!("\"{}\"", t.replace('"', "\\\""))) .unwrap_or_else(|| "null".to_string()), result.ocr_recommended, + ocr_pages.join(","), elapsed.as_millis() ); } else { @@ -67,6 +73,16 @@ fn main() { "OCR recommended: {}", if result.ocr_recommended { "YES" } else { "NO" } ); + if !result.pages_needing_ocr.is_empty() { + if result.pages_needing_ocr.len() == result.page_count as usize { + println!("Pages needing OCR: all (of {})", result.page_count); + } else { + println!( + "Pages needing OCR: {:?} (of {})", + result.pages_needing_ocr, result.page_count + ); + } + } if let Some(title) = &result.title { println!("Title: {}", title); } diff --git a/src/bin/pdf2md.rs b/src/bin/pdf2md.rs index 075a364..20caa34 100644 --- a/src/bin/pdf2md.rs +++ b/src/bin/pdf2md.rs @@ -43,8 +43,13 @@ fn main() { }) .unwrap_or_default(); + let ocr_pages: Vec = result + .pages_needing_ocr + .iter() + .map(|p| p.to_string()) + .collect(); println!( - r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"markdown":"{}"}}"#, + r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"markdown":"{}"}}"#, match result.pdf_type { PdfType::TextBased => "text_based", PdfType::Scanned => "scanned", @@ -55,6 +60,7 @@ fn main() { result.text.is_some(), result.processing_time_ms, result.markdown.as_ref().map(|m| m.len()).unwrap_or(0), + ocr_pages.join(","), md_escaped ); } else if raw_output { @@ -120,7 +126,11 @@ fn main() { if let Some(markdown) = &result.markdown { eprintln!(); - eprintln!("Note: Some pages may contain images that require OCR."); + if result.pages_needing_ocr.is_empty() { + eprintln!("Note: Some pages may contain images that require OCR."); + } else { + eprintln!("Pages needing OCR: {:?}", result.pages_needing_ocr); + } eprintln!(); if let Some(output) = output_file { diff --git a/src/detector.rs b/src/detector.rs index e897bd4..daf503a 100644 --- a/src/detector.rs +++ b/src/detector.rs @@ -6,6 +6,7 @@ use crate::PdfError; use lopdf::{Document, Object, ObjectId}; +use std::collections::HashMap; use std::path::Path; /// PDF type classification @@ -39,6 +40,9 @@ pub struct PdfTypeResult { /// Whether OCR is recommended for better extraction /// True when images provide essential context (e.g., template-based PDFs) pub ocr_recommended: bool, + /// 1-indexed page numbers that need OCR (image-only or insufficient text). + /// Empty for TextBased. All pages for Scanned/ImageBased. Specific pages for Mixed. + pub pages_needing_ocr: Vec, } /// Configuration for PDF type detection @@ -150,6 +154,9 @@ fn detect_from_document( let mut pages_with_template_images = 0u32; let mut total_text_ops = 0u32; + // Cache Phase 1 results to avoid re-analyzing sampled pages in Phase 2 + let mut analysis_cache: HashMap = HashMap::new(); + for page_num in &sample_indices { if let Some(&page_id) = pages.get(page_num) { let analysis = analyze_page_content(doc, page_id); @@ -163,6 +170,7 @@ fn detect_from_document( pages_with_template_images += 1; } total_text_ops += analysis.text_operator_count; + analysis_cache.insert(*page_num, analysis); } } @@ -214,6 +222,30 @@ fn detect_from_document( (PdfType::TextBased, text_ratio.max(0.5)) }; + // Phase 2: Build per-page OCR list + let pages_needing_ocr = match pdf_type { + PdfType::TextBased => Vec::new(), + PdfType::Scanned | PdfType::ImageBased => (1..=total_pages).collect(), + PdfType::Mixed => { + let mut ocr_pages = Vec::new(); + for page_num in 1..=total_pages { + let analysis = if let Some(cached) = analysis_cache.get(&page_num) { + cached.clone() + } else if let Some(&page_id) = pages.get(&page_num) { + analyze_page_content(doc, page_id) + } else { + continue; + }; + if analysis.text_operator_count < config.min_text_ops_per_page + && (analysis.has_images || analysis.has_template_image) + { + ocr_pages.push(page_num); + } + } + ocr_pages + } + }; + // Try to get title from metadata let title = get_document_title(doc); @@ -225,10 +257,12 @@ fn detect_from_document( confidence, title, ocr_recommended, + pages_needing_ocr, }) } /// Page content analysis result +#[derive(Clone)] struct PageAnalysis { text_operator_count: u32, has_images: bool, diff --git a/src/lib.rs b/src/lib.rs index d4df0f3..6eca678 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -31,6 +31,8 @@ pub struct PdfProcessResult { pub page_count: u32, /// Processing time in milliseconds pub processing_time_ms: u64, + /// 1-indexed page numbers that need OCR. + pub pages_needing_ocr: Vec, } /// Process a PDF file with smart detection and extraction @@ -46,29 +48,34 @@ pub fn process_pdf>(path: P) -> Result { // Step 2: Full extraction with position-aware reading order let items = extract_text_with_positions(&path)?; let markdown = to_markdown_from_items(items, MarkdownOptions::default()); PdfProcessResult { - pdf_type: PdfType::TextBased, + pdf_type, text: None, // We now produce markdown directly markdown: Some(markdown), - page_count: detection.page_count, + page_count, processing_time_ms: start.elapsed().as_millis() as u64, + pages_needing_ocr, } } PdfType::Scanned | PdfType::ImageBased => { // Return early - OCR needed PdfProcessResult { - pdf_type: detection.pdf_type, + pdf_type, text: None, markdown: None, - page_count: detection.page_count, + page_count, processing_time_ms: start.elapsed().as_millis() as u64, + pages_needing_ocr, } } PdfType::Mixed => { @@ -77,11 +84,12 @@ pub fn process_pdf>(path: P) -> Result Result { // Step 1: Smart detection (fast, no full load) let detection = detector::detect_pdf_type_mem(buffer)?; + let page_count = detection.page_count; + let pdf_type = detection.pdf_type; + let pages_needing_ocr = detection.pages_needing_ocr; - let result = match detection.pdf_type { + let result = match pdf_type { PdfType::TextBased => { // Step 2: Full extraction with position-aware reading order let items = extractor::extract_text_with_positions_mem(buffer)?; let markdown = to_markdown_from_items(items, MarkdownOptions::default()); PdfProcessResult { - pdf_type: PdfType::TextBased, + pdf_type, text: None, markdown: Some(markdown), - page_count: detection.page_count, + page_count, processing_time_ms: start.elapsed().as_millis() as u64, + pages_needing_ocr, } } PdfType::Scanned | PdfType::ImageBased => PdfProcessResult { - pdf_type: detection.pdf_type, + pdf_type, text: None, markdown: None, - page_count: detection.page_count, + page_count, processing_time_ms: start.elapsed().as_millis() as u64, + pages_needing_ocr, }, PdfType::Mixed => { let items = extractor::extract_text_with_positions_mem(buffer).ok(); let markdown = items.map(|i| to_markdown_from_items(i, MarkdownOptions::default())); PdfProcessResult { - pdf_type: PdfType::Mixed, + pdf_type, text: None, markdown, - page_count: detection.page_count, + page_count, processing_time_ms: start.elapsed().as_millis() as u64, + pages_needing_ocr, } } }; diff --git a/tests/integration_tests.rs b/tests/integration_tests.rs index 2a3a3c3..6b94bc9 100644 --- a/tests/integration_tests.rs +++ b/tests/integration_tests.rs @@ -836,3 +836,69 @@ fn test_not_a_pdf_extract_text_mem() { let result = pdf_inspector::extractor::extract_text_mem(xml); assert_not_a_pdf(result, "XML"); } + +// ============================================================================ +// Pages Needing OCR Tests +// ============================================================================ + +#[test] +fn test_pages_needing_ocr_field_accessible() { + // Compile-time check: verify the field exists on both structs + let detection_result = pdf_inspector::detector::PdfTypeResult { + pdf_type: PdfType::TextBased, + page_count: 1, + pages_sampled: 1, + pages_with_text: 1, + confidence: 1.0, + title: None, + ocr_recommended: false, + pages_needing_ocr: Vec::new(), + }; + assert!(detection_result.pages_needing_ocr.is_empty()); + + let process_result = pdf_inspector::PdfProcessResult { + pdf_type: PdfType::TextBased, + text: None, + markdown: None, + page_count: 1, + processing_time_ms: 0, + pages_needing_ocr: vec![1, 3], + }; + assert_eq!(process_result.pages_needing_ocr, vec![1, 3]); +} + +#[test] +fn test_text_pdf_process_result_empty_ocr_pages() { + // A minimal valid PDF that is text-based should have empty pages_needing_ocr. + // We use a minimal PDF buffer with a text content stream. + let pdf_bytes = b"%PDF-1.0 +1 0 obj<>endobj +2 0 obj<>endobj +3 0 obj<>endobj +4 0 obj<> +stream +BT /F1 12 Tf 100 700 Td (Hello World) Tj ET +endstream +endobj +xref +0 5 +0000000000 65535 f +0000000009 00000 n +0000000058 00000 n +0000000115 00000 n +0000000206 00000 n +trailer<> +startxref +300 +%%EOF"; + let result = pdf_inspector::process_pdf_mem(pdf_bytes); + // The minimal PDF may fail to parse fully, but if it succeeds, + // a text-based PDF should have empty pages_needing_ocr. + if let Ok(result) = result { + assert!( + result.pages_needing_ocr.is_empty(), + "Text-based PDF should have empty pages_needing_ocr, got: {:?}", + result.pages_needing_ocr + ); + } +}