Adds a new function that takes a PDF buffer and page+bbox regions (same interface as extractTextInRegions), runs heuristic table detection on items within each region, and returns markdown pipe-tables. Falls back to needs_ocr=true when no table structure is found or text quality is suspect. Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
1439 lines
47 KiB
Rust
1439 lines
47 KiB
Rust
//! Integration tests for pdf-to-markdown library
|
|
|
|
use pdf_inspector::detector::{DetectionConfig, ScanStrategy};
|
|
use pdf_inspector::extractor::group_into_lines;
|
|
use pdf_inspector::types::TextLine;
|
|
use pdf_inspector::{
|
|
detect_pdf_type, extract_tables_in_regions_mem, extract_text, extract_text_in_regions_mem,
|
|
extract_text_with_positions, process_pdf_mem, process_pdf_with_options, to_markdown,
|
|
MarkdownOptions, PdfError, PdfOptions, PdfType, TextItem,
|
|
};
|
|
use std::collections::HashSet;
|
|
|
|
// Helper to create test TextItems
|
|
fn make_text_item(text: &str, x: f32, y: f32, font_size: f32, page: u32) -> TextItem {
|
|
use pdf_inspector::types::ItemType;
|
|
TextItem {
|
|
text: text.to_string(),
|
|
x,
|
|
y,
|
|
width: text.len() as f32 * font_size * 0.5,
|
|
height: font_size,
|
|
font: "Helvetica".to_string(),
|
|
font_size,
|
|
page,
|
|
is_bold: false,
|
|
is_italic: false,
|
|
item_type: ItemType::Text,
|
|
mcid: None,
|
|
}
|
|
}
|
|
|
|
fn make_text_item_with_font(
|
|
text: &str,
|
|
x: f32,
|
|
y: f32,
|
|
font_size: f32,
|
|
font: &str,
|
|
page: u32,
|
|
) -> TextItem {
|
|
use pdf_inspector::extractor::{is_bold_font, is_italic_font, ItemType};
|
|
TextItem {
|
|
text: text.to_string(),
|
|
x,
|
|
y,
|
|
width: text.len() as f32 * font_size * 0.5,
|
|
height: font_size,
|
|
font: font.to_string(),
|
|
font_size,
|
|
page,
|
|
is_bold: is_bold_font(font),
|
|
is_italic: is_italic_font(font),
|
|
item_type: ItemType::Text,
|
|
mcid: None,
|
|
}
|
|
}
|
|
|
|
// ============================================================================
|
|
// Detection Config Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_detection_config_default() {
|
|
let config = DetectionConfig::default();
|
|
assert!(matches!(config.strategy, ScanStrategy::Sample(8)));
|
|
assert_eq!(config.min_text_ops_per_page, 3);
|
|
assert!((config.text_page_ratio_threshold - 0.6).abs() < 0.001);
|
|
}
|
|
|
|
#[test]
|
|
fn test_detection_config_custom() {
|
|
let config = DetectionConfig {
|
|
strategy: ScanStrategy::Sample(10),
|
|
min_text_ops_per_page: 5,
|
|
text_page_ratio_threshold: 0.8,
|
|
};
|
|
assert!(matches!(config.strategy, ScanStrategy::Sample(10)));
|
|
assert_eq!(config.min_text_ops_per_page, 5);
|
|
assert!((config.text_page_ratio_threshold - 0.8).abs() < 0.001);
|
|
}
|
|
|
|
// ============================================================================
|
|
// PdfType Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_pdf_type_equality() {
|
|
assert_eq!(PdfType::TextBased, PdfType::TextBased);
|
|
assert_eq!(PdfType::Scanned, PdfType::Scanned);
|
|
assert_eq!(PdfType::ImageBased, PdfType::ImageBased);
|
|
assert_eq!(PdfType::Mixed, PdfType::Mixed);
|
|
assert_ne!(PdfType::TextBased, PdfType::Scanned);
|
|
}
|
|
|
|
#[test]
|
|
fn test_pdf_type_clone() {
|
|
let original = PdfType::TextBased;
|
|
let cloned = original.clone();
|
|
assert_eq!(original, cloned);
|
|
}
|
|
|
|
#[test]
|
|
fn test_pdf_type_debug() {
|
|
let pdf_type = PdfType::TextBased;
|
|
let debug_str = format!("{:?}", pdf_type);
|
|
assert_eq!(debug_str, "TextBased");
|
|
}
|
|
|
|
// ============================================================================
|
|
// TextItem Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_text_item_creation() {
|
|
let item = make_text_item("Hello", 100.0, 700.0, 12.0, 1);
|
|
assert_eq!(item.text, "Hello");
|
|
assert_eq!(item.x, 100.0);
|
|
assert_eq!(item.y, 700.0);
|
|
assert_eq!(item.font_size, 12.0);
|
|
assert_eq!(item.page, 1);
|
|
}
|
|
|
|
#[test]
|
|
fn test_text_item_clone() {
|
|
let item = make_text_item("Test", 50.0, 600.0, 14.0, 2);
|
|
let cloned = item.clone();
|
|
assert_eq!(item.text, cloned.text);
|
|
assert_eq!(item.x, cloned.x);
|
|
assert_eq!(item.y, cloned.y);
|
|
}
|
|
|
|
// ============================================================================
|
|
// TextLine Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_text_line_text_method() {
|
|
let items = vec![
|
|
make_text_item("Hello", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("World", 160.0, 700.0, 12.0, 1),
|
|
];
|
|
let line = TextLine {
|
|
items,
|
|
y: 700.0,
|
|
page: 1,
|
|
adaptive_threshold: 0.10,
|
|
};
|
|
assert_eq!(line.text(), "Hello World");
|
|
}
|
|
|
|
#[test]
|
|
fn test_text_line_single_item() {
|
|
let items = vec![make_text_item("Single", 100.0, 700.0, 12.0, 1)];
|
|
let line = TextLine {
|
|
items,
|
|
y: 700.0,
|
|
page: 1,
|
|
adaptive_threshold: 0.10,
|
|
};
|
|
assert_eq!(line.text(), "Single");
|
|
}
|
|
|
|
#[test]
|
|
fn test_text_line_empty() {
|
|
let line = TextLine {
|
|
items: vec![],
|
|
y: 700.0,
|
|
page: 1,
|
|
adaptive_threshold: 0.10,
|
|
};
|
|
assert_eq!(line.text(), "");
|
|
}
|
|
|
|
// ============================================================================
|
|
// Group Into Lines Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_group_into_lines_empty() {
|
|
let items: Vec<TextItem> = vec![];
|
|
let lines = group_into_lines(items);
|
|
assert!(lines.is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn test_group_into_lines_same_line() {
|
|
let items = vec![
|
|
make_text_item("A", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("B", 120.0, 700.0, 12.0, 1),
|
|
make_text_item("C", 140.0, 700.0, 12.0, 1),
|
|
];
|
|
let lines = group_into_lines(items);
|
|
assert_eq!(lines.len(), 1);
|
|
assert_eq!(lines[0].items.len(), 3);
|
|
assert_eq!(lines[0].text(), "A B C");
|
|
}
|
|
|
|
#[test]
|
|
fn test_group_into_lines_different_lines() {
|
|
let items = vec![
|
|
make_text_item("Line1", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("Line2", 100.0, 680.0, 12.0, 1),
|
|
make_text_item("Line3", 100.0, 660.0, 12.0, 1),
|
|
];
|
|
let lines = group_into_lines(items);
|
|
assert_eq!(lines.len(), 3);
|
|
assert_eq!(lines[0].text(), "Line1");
|
|
assert_eq!(lines[1].text(), "Line2");
|
|
assert_eq!(lines[2].text(), "Line3");
|
|
}
|
|
|
|
#[test]
|
|
fn test_group_into_lines_y_tolerance() {
|
|
// Items within 3.0 Y tolerance should be grouped
|
|
// Note: items are sorted by Y descending, then X ascending
|
|
let items = vec![
|
|
make_text_item("A", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("B", 150.0, 700.0, 12.0, 1), // Same Y
|
|
];
|
|
let lines = group_into_lines(items);
|
|
assert_eq!(lines.len(), 1);
|
|
assert_eq!(lines[0].text(), "A B");
|
|
}
|
|
|
|
#[test]
|
|
fn test_group_into_lines_multiple_pages() {
|
|
let items = vec![
|
|
make_text_item("Page1Text", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("Page2Text", 100.0, 700.0, 12.0, 2),
|
|
];
|
|
let lines = group_into_lines(items);
|
|
assert_eq!(lines.len(), 2);
|
|
assert_eq!(lines[0].page, 1);
|
|
assert_eq!(lines[1].page, 2);
|
|
}
|
|
|
|
#[test]
|
|
fn test_group_into_lines_sorting_by_x() {
|
|
// Items on same line should be sorted by X position
|
|
let items = vec![
|
|
make_text_item("Third", 200.0, 700.0, 12.0, 1),
|
|
make_text_item("First", 50.0, 700.0, 12.0, 1),
|
|
make_text_item("Second", 100.0, 700.0, 12.0, 1),
|
|
];
|
|
let lines = group_into_lines(items);
|
|
assert_eq!(lines.len(), 1);
|
|
assert_eq!(lines[0].text(), "First Second Third");
|
|
}
|
|
|
|
// ============================================================================
|
|
// MarkdownOptions Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_markdown_options_default() {
|
|
let opts = MarkdownOptions::default();
|
|
assert!(opts.detect_headers);
|
|
assert!(opts.detect_lists);
|
|
assert!(opts.detect_code);
|
|
assert!(opts.base_font_size.is_none());
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_options_custom() {
|
|
let opts = MarkdownOptions {
|
|
detect_headers: false,
|
|
detect_lists: true,
|
|
detect_code: false,
|
|
base_font_size: Some(14.0),
|
|
remove_page_numbers: false,
|
|
format_urls: false,
|
|
fix_hyphenation: false,
|
|
detect_bold: false,
|
|
detect_italic: false,
|
|
include_images: false,
|
|
include_links: false,
|
|
include_page_numbers: false,
|
|
..Default::default()
|
|
};
|
|
assert!(!opts.detect_headers);
|
|
assert!(opts.detect_lists);
|
|
assert!(!opts.detect_code);
|
|
assert_eq!(opts.base_font_size, Some(14.0));
|
|
assert!(!opts.remove_page_numbers);
|
|
assert!(!opts.format_urls);
|
|
assert!(!opts.fix_hyphenation);
|
|
assert!(!opts.detect_bold);
|
|
assert!(!opts.detect_italic);
|
|
assert!(!opts.include_images);
|
|
assert!(!opts.include_links);
|
|
}
|
|
|
|
// ============================================================================
|
|
// Markdown Conversion Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_to_markdown_basic() {
|
|
let text = "Hello World";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("Hello World"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_multiple_lines() {
|
|
let text = "Line one\nLine two\nLine three";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("Line one"));
|
|
assert!(md.contains("Line two"));
|
|
assert!(md.contains("Line three"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_bullet_list() {
|
|
let text = "• First\n• Second\n• Third";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("- First"));
|
|
assert!(md.contains("- Second"));
|
|
assert!(md.contains("- Third"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_dash_list() {
|
|
let text = "- One\n- Two\n- Three";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("- One"));
|
|
assert!(md.contains("- Two"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_numbered_list() {
|
|
let text = "1. First\n2. Second\n3. Third";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("1. First"));
|
|
assert!(md.contains("2. Second"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_code_detection() {
|
|
let text = "const x = 5;\nlet y = 10;";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("```"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_no_code_detection() {
|
|
let text = "const x = 5;";
|
|
let opts = MarkdownOptions {
|
|
detect_code: false,
|
|
..Default::default()
|
|
};
|
|
let md = to_markdown(text, opts);
|
|
assert!(!md.contains("```"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_no_list_detection() {
|
|
let text = "• Item";
|
|
let opts = MarkdownOptions {
|
|
detect_lists: false,
|
|
..Default::default()
|
|
};
|
|
let md = to_markdown(text, opts);
|
|
// Should keep original bullet character
|
|
assert!(md.contains("•"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_empty_lines() {
|
|
let text = "Para one\n\nPara two";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("Para one"));
|
|
assert!(md.contains("Para two"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_to_markdown_whitespace_only_lines() {
|
|
let text = "Content\n \nMore content";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.contains("Content"));
|
|
assert!(md.contains("More content"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// Markdown From Items Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_markdown_from_items_empty() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
let items: Vec<TextItem> = vec![];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_from_items_single() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
let items = vec![make_text_item("Hello", 100.0, 700.0, 12.0, 1)];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("Hello"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_from_items_header_detection() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Need multiple body items to establish base font size
|
|
let items = vec![
|
|
make_text_item("Title", 100.0, 750.0, 24.0, 1), // Large font = H1
|
|
make_text_item("Body text one", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("Body text two", 100.0, 680.0, 12.0, 1),
|
|
make_text_item("Body text three", 100.0, 660.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("# Title"));
|
|
assert!(md.contains("Body text"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_from_items_h2_detection() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Two heading tiers: 24.0 → H1, 18.0 → H2
|
|
let items = vec![
|
|
make_text_item("Title", 100.0, 800.0, 24.0, 1),
|
|
make_text_item("Subtitle", 100.0, 750.0, 18.0, 1),
|
|
make_text_item("Body text one", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("Body text two", 100.0, 680.0, 12.0, 1),
|
|
make_text_item("Body text three", 100.0, 660.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("## Subtitle"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_from_items_monospace_code() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
let items = vec![make_text_item_with_font(
|
|
"let x = 5",
|
|
100.0,
|
|
700.0,
|
|
12.0,
|
|
"Courier",
|
|
1,
|
|
)];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("```"));
|
|
assert!(md.contains("let x = 5"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_from_items_page_breaks() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
let items = vec![
|
|
make_text_item("Content on first page", 100.0, 700.0, 12.0, 1),
|
|
make_text_item("Content on second page", 100.0, 700.0, 12.0, 2),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
// Pages should be separated by blank lines (no --- markers)
|
|
assert!(!md.contains("---"));
|
|
assert!(md.contains("Content on first page"));
|
|
assert!(md.contains("Content on second page"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// Markdown From Lines Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_markdown_from_lines_empty() {
|
|
use pdf_inspector::markdown::to_markdown_from_lines;
|
|
let lines: Vec<TextLine> = vec![];
|
|
let md = to_markdown_from_lines(lines, MarkdownOptions::default());
|
|
assert!(md.is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn test_markdown_from_lines_basic() {
|
|
use pdf_inspector::markdown::to_markdown_from_lines;
|
|
let lines = vec![
|
|
TextLine {
|
|
items: vec![make_text_item("First", 100.0, 700.0, 12.0, 1)],
|
|
y: 700.0,
|
|
page: 1,
|
|
adaptive_threshold: 0.10,
|
|
},
|
|
TextLine {
|
|
items: vec![make_text_item("Second", 100.0, 680.0, 12.0, 1)],
|
|
y: 680.0,
|
|
page: 1,
|
|
adaptive_threshold: 0.10,
|
|
},
|
|
];
|
|
let md = to_markdown_from_lines(lines, MarkdownOptions::default());
|
|
assert!(md.contains("First"));
|
|
assert!(md.contains("Second"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// Error Handling Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_extract_text_nonexistent_file() {
|
|
let result = extract_text("/nonexistent/file.pdf");
|
|
assert!(result.is_err());
|
|
}
|
|
|
|
#[test]
|
|
fn test_detect_pdf_type_nonexistent_file() {
|
|
let result = detect_pdf_type("/nonexistent/file.pdf");
|
|
assert!(result.is_err());
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_text_with_positions_nonexistent_file() {
|
|
let result = extract_text_with_positions("/nonexistent/file.pdf");
|
|
assert!(result.is_err());
|
|
}
|
|
|
|
// ============================================================================
|
|
// List Pattern Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_bullet_variations() {
|
|
// Unicode bullets get converted to markdown dash
|
|
let unicode_bullets = ["• Item", "○ Item", "● Item", "◦ Item"];
|
|
for bullet in &unicode_bullets {
|
|
let md = to_markdown(bullet, MarkdownOptions::default());
|
|
assert!(md.contains("- Item"), "Failed for: {}", bullet);
|
|
}
|
|
|
|
// Markdown-compatible bullets stay as-is
|
|
let md_bullets = ["- Item", "* Item"];
|
|
for bullet in &md_bullets {
|
|
let md = to_markdown(bullet, MarkdownOptions::default());
|
|
assert!(md.contains(bullet), "Failed for: {}", bullet);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_numbered_list_variations() {
|
|
let lists = ["1. First", "2) Second", "10. Tenth"];
|
|
for item in &lists {
|
|
let md = to_markdown(item, MarkdownOptions::default());
|
|
assert!(md.trim().len() > 0, "Failed for: {}", item);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_letter_list_items() {
|
|
let md = to_markdown("a. Letter item", MarkdownOptions::default());
|
|
assert!(md.contains("a. Letter item"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// Code Detection Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_code_keywords() {
|
|
let keywords = [
|
|
"import foo",
|
|
"export default",
|
|
"const x = 5;",
|
|
"let y = 10;",
|
|
"function test() {",
|
|
"class MyClass {",
|
|
"def func():",
|
|
"pub fn main() {",
|
|
"async fn process() {",
|
|
"impl Trait {",
|
|
];
|
|
for code in &keywords {
|
|
let md = to_markdown(code, MarkdownOptions::default());
|
|
assert!(md.contains("```"), "Code not detected for: {}", code);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_code_syntax_patterns() {
|
|
// Patterns that start with code keywords/syntax
|
|
let patterns = [
|
|
"=> value", // Starts with =>
|
|
"-> Result", // Starts with ->
|
|
":: io::Result", // Starts with ::
|
|
];
|
|
for code in &patterns {
|
|
let md = to_markdown(code, MarkdownOptions::default());
|
|
assert!(md.contains("```"), "Code not detected for: {}", code);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_code_special_chars() {
|
|
let code = "if (x > 0) { return y; }";
|
|
let md = to_markdown(code, MarkdownOptions::default());
|
|
assert!(md.contains("```"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_non_code_text() {
|
|
let text = "This is regular text about programming.";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(!md.contains("```"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// Monospace Font Detection Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_monospace_font_names() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Font names that contain the patterns in is_monospace_font
|
|
let monospace_fonts = [
|
|
"Courier",
|
|
"Consolas",
|
|
"Monaco",
|
|
"Menlo",
|
|
"Fira Code",
|
|
"JetBrains Mono",
|
|
"Inconsolata",
|
|
"DejaVu Sans Mono",
|
|
"Liberation Mono",
|
|
"Fixed",
|
|
"Terminal",
|
|
];
|
|
|
|
for font in &monospace_fonts {
|
|
let items = vec![make_text_item_with_font(
|
|
"code", 100.0, 700.0, 12.0, font, 1,
|
|
)];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(
|
|
md.contains("```"),
|
|
"Font not detected as monospace: {}",
|
|
font
|
|
);
|
|
}
|
|
}
|
|
|
|
// ============================================================================
|
|
// Header Level Detection Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_header_level_h1() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// 24.0 / 12.0 = 2.0x = H1
|
|
// Need multiple body items to establish base font size
|
|
let items = vec![
|
|
make_text_item("H1 Title", 100.0, 700.0, 24.0, 1),
|
|
make_text_item("body text one", 100.0, 650.0, 12.0, 1),
|
|
make_text_item("body text two", 100.0, 630.0, 12.0, 1),
|
|
make_text_item("body text three", 100.0, 610.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("# H1 Title"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_single_heading_tier_becomes_h1() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Single heading tier: 18.0pt on 12.0pt base → H1 (not H2)
|
|
let items = vec![
|
|
make_text_item("Section Title", 100.0, 700.0, 18.0, 1),
|
|
make_text_item("body text one", 100.0, 650.0, 12.0, 1),
|
|
make_text_item("body text two", 100.0, 630.0, 12.0, 1),
|
|
make_text_item("body text three", 100.0, 610.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("# Section Title"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_header_level_h2() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Two heading tiers: 24.0 → H1, 18.0 → H2
|
|
let items = vec![
|
|
make_text_item("H1 Title", 100.0, 750.0, 24.0, 1),
|
|
make_text_item("H2 Title", 100.0, 700.0, 18.0, 1),
|
|
make_text_item("body text one", 100.0, 650.0, 12.0, 1),
|
|
make_text_item("body text two", 100.0, 630.0, 12.0, 1),
|
|
make_text_item("body text three", 100.0, 610.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("# H1 Title"));
|
|
assert!(md.contains("## H2 Title"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_header_level_h3() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Three heading tiers: 24.0 → H1, 18.0 → H2, 15.0 → H3
|
|
let items = vec![
|
|
make_text_item("H1 Title", 100.0, 800.0, 24.0, 1),
|
|
make_text_item("H2 Title", 100.0, 750.0, 18.0, 1),
|
|
make_text_item("H3 Title", 100.0, 700.0, 15.0, 1),
|
|
make_text_item("body text one", 100.0, 650.0, 12.0, 1),
|
|
make_text_item("body text two", 100.0, 630.0, 12.0, 1),
|
|
make_text_item("body text three", 100.0, 610.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("### H3 Title"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_header_level_h4() {
|
|
use pdf_inspector::markdown::to_markdown_from_items;
|
|
// Four heading tiers: 24.0 → H1, 18.0 → H2, 15.0 → H3, 14.5 → H4
|
|
let items = vec![
|
|
make_text_item("H1 Title", 100.0, 850.0, 24.0, 1),
|
|
make_text_item("H2 Title", 100.0, 800.0, 18.0, 1),
|
|
make_text_item("H3 Title", 100.0, 750.0, 15.0, 1),
|
|
make_text_item("H4 Title", 100.0, 700.0, 14.5, 1),
|
|
make_text_item("body text one", 100.0, 650.0, 12.0, 1),
|
|
make_text_item("body text two", 100.0, 630.0, 12.0, 1),
|
|
make_text_item("body text three", 100.0, 610.0, 12.0, 1),
|
|
];
|
|
let md = to_markdown_from_items(items, MarkdownOptions::default());
|
|
assert!(md.contains("#### H4 Title"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// Clean Markdown Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_excessive_newlines_preserved_in_plain_text() {
|
|
// Plain text to_markdown preserves structure from input
|
|
let text = "Para one\n\n\n\n\nPara two";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
// The function processes line by line, empty lines become single newlines
|
|
assert!(md.contains("Para one"));
|
|
assert!(md.contains("Para two"));
|
|
}
|
|
|
|
#[test]
|
|
fn test_trailing_newline() {
|
|
let text = "Content";
|
|
let md = to_markdown(text, MarkdownOptions::default());
|
|
assert!(md.ends_with('\n'));
|
|
assert!(!md.ends_with("\n\n"));
|
|
}
|
|
|
|
// ============================================================================
|
|
// NotAPdf Detection Tests
|
|
// ============================================================================
|
|
|
|
/// Helper: assert that an error is NotAPdf and its message contains the given substring.
|
|
fn assert_not_a_pdf(result: Result<impl std::fmt::Debug, PdfError>, expected_hint: &str) {
|
|
match result {
|
|
Err(PdfError::NotAPdf(msg)) => {
|
|
assert!(
|
|
msg.to_lowercase().contains(&expected_hint.to_lowercase()),
|
|
"Expected hint '{}' in NotAPdf message, got: '{}'",
|
|
expected_hint,
|
|
msg,
|
|
);
|
|
}
|
|
other => panic!(
|
|
"Expected Err(NotAPdf) containing '{}', got: {:?}",
|
|
expected_hint, other,
|
|
),
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_html_input() {
|
|
let html = b"<!DOCTYPE html><html><body>Hello</body></html>";
|
|
let result = pdf_inspector::process_pdf_mem(html);
|
|
assert_not_a_pdf(result, "HTML");
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_xml_input() {
|
|
let xml = b"<?xml version=\"1.0\"?><root><item>data</item></root>";
|
|
let result = pdf_inspector::process_pdf_mem(xml);
|
|
assert_not_a_pdf(result, "XML");
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_json_input() {
|
|
let json = b"{\"error\": \"download failed\"}";
|
|
let result = pdf_inspector::process_pdf_mem(json);
|
|
assert_not_a_pdf(result, "JSON");
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_plain_text_input() {
|
|
let text = b"This is a plain text file that is not a PDF at all.";
|
|
let result = pdf_inspector::process_pdf_mem(text);
|
|
assert_not_a_pdf(result, "plain text");
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_empty_buffer() {
|
|
let result = pdf_inspector::process_pdf_mem(b"");
|
|
assert_not_a_pdf(result, "empty");
|
|
}
|
|
|
|
#[test]
|
|
fn test_valid_pdf_header_not_rejected() {
|
|
// A truncated but valid PDF header should NOT produce NotAPdf —
|
|
// it should fail with Parse or InvalidStructure instead.
|
|
let truncated_pdf = b"%PDF-1.4\ntruncated content";
|
|
let result = pdf_inspector::process_pdf_mem(truncated_pdf);
|
|
match result {
|
|
Err(PdfError::NotAPdf(_)) => panic!("Valid PDF header should not be rejected as NotAPdf"),
|
|
_ => {} // Parse or InvalidStructure is fine
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_bom_prefixed_pdf_header_not_rejected() {
|
|
// UTF-8 BOM + %PDF- should still be recognized as a PDF
|
|
let mut bom_pdf = vec![0xEF, 0xBB, 0xBF];
|
|
bom_pdf.extend_from_slice(b"%PDF-1.7\ntruncated");
|
|
let result = pdf_inspector::process_pdf_mem(&bom_pdf);
|
|
match result {
|
|
Err(PdfError::NotAPdf(_)) => {
|
|
panic!("BOM-prefixed PDF header should not be rejected as NotAPdf")
|
|
}
|
|
_ => {} // Parse or InvalidStructure is fine
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_detect_pdf_type_mem() {
|
|
// Verify detect_pdf_type_mem is also guarded
|
|
let html = b"<html><head><title>Not a PDF</title></head></html>";
|
|
let result = pdf_inspector::detector::detect_pdf_type_mem(html);
|
|
assert_not_a_pdf(result, "HTML");
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_extract_text_with_positions_mem() {
|
|
// Verify extract_text_with_positions_mem is also guarded
|
|
let html = b"<!DOCTYPE html><html><body>content</body></html>";
|
|
let result = pdf_inspector::extractor::extract_text_with_positions_mem(html);
|
|
assert_not_a_pdf(result, "HTML");
|
|
}
|
|
|
|
#[test]
|
|
fn test_not_a_pdf_extract_text_mem() {
|
|
// Verify extract_text_mem is also guarded
|
|
let xml = b"<?xml version=\"1.0\"?><data/>";
|
|
let result = pdf_inspector::extractor::extract_text_mem(xml);
|
|
assert_not_a_pdf(result, "XML");
|
|
}
|
|
|
|
// ============================================================================
|
|
// Snapshot Regression Tests (PDF fixtures)
|
|
// ============================================================================
|
|
|
|
/// Process a PDF fixture and compare output against the golden snapshot.
|
|
///
|
|
/// This catches regressions where code changes silently alter extraction
|
|
/// or markdown output. If a change is intentional, update the snapshot:
|
|
/// cargo run --release --bin pdf2md -- tests/fixtures/<name>.pdf > tests/snapshots/<name>.md
|
|
fn assert_snapshot(fixture: &str) {
|
|
let fixture_path = format!("tests/fixtures/{}.pdf", fixture);
|
|
let snapshot_path = format!("tests/snapshots/{}.md", fixture);
|
|
|
|
let result = pdf_inspector::process_pdf(&fixture_path)
|
|
.unwrap_or_else(|e| panic!("Failed to process {}: {}", fixture_path, e));
|
|
let actual = result.markdown.unwrap_or_default();
|
|
let actual = actual.trim_end();
|
|
|
|
let expected = std::fs::read_to_string(&snapshot_path)
|
|
.unwrap_or_else(|e| panic!("Failed to read snapshot {}: {}", snapshot_path, e));
|
|
let expected = expected.trim_end();
|
|
|
|
if actual != expected {
|
|
// Show a helpful diff summary
|
|
let actual_lines: Vec<&str> = actual.lines().collect();
|
|
let expected_lines: Vec<&str> = expected.lines().collect();
|
|
|
|
let mut diffs = Vec::new();
|
|
let max_lines = actual_lines.len().max(expected_lines.len());
|
|
for i in 0..max_lines {
|
|
let a = actual_lines.get(i).unwrap_or(&"<missing>");
|
|
let e = expected_lines.get(i).unwrap_or(&"<missing>");
|
|
if a != e {
|
|
diffs.push(format!(
|
|
" line {}: expected {:?}, got {:?}",
|
|
i + 1,
|
|
&e[..e.len().min(80)],
|
|
&a[..a.len().min(80)]
|
|
));
|
|
if diffs.len() >= 5 {
|
|
diffs.push(" ... (more diffs truncated)".to_string());
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
panic!(
|
|
"Snapshot mismatch for {}:\n{}\n\nTo update: cargo run --release --bin pdf2md -- {} > {}",
|
|
fixture,
|
|
diffs.join("\n"),
|
|
fixture_path,
|
|
snapshot_path,
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_snapshot_nexo_price_en() {
|
|
assert_snapshot("nexo-price-en");
|
|
}
|
|
|
|
#[test]
|
|
fn test_snapshot_thermo_freon12() {
|
|
assert_snapshot("thermo-freon12");
|
|
}
|
|
|
|
#[test]
|
|
fn test_snapshot_td9264() {
|
|
assert_snapshot("td9264");
|
|
}
|
|
|
|
#[test]
|
|
fn test_snapshot_p1244() {
|
|
assert_snapshot("p1244-1996");
|
|
}
|
|
|
|
#[test]
|
|
fn test_snapshot_real_estate_pricing() {
|
|
assert_snapshot("real-estate-pricing");
|
|
}
|
|
|
|
#[test]
|
|
fn test_snapshot_2013_app2() {
|
|
assert_snapshot("2013-app2");
|
|
}
|
|
|
|
// ============================================================================
|
|
// Pages Needing OCR Tests
|
|
// ============================================================================
|
|
|
|
#[test]
|
|
fn test_pages_needing_ocr_field_accessible() {
|
|
// Compile-time check: verify the field exists on both structs
|
|
let detection_result = pdf_inspector::detector::PdfTypeResult {
|
|
pdf_type: PdfType::TextBased,
|
|
page_count: 1,
|
|
pages_sampled: 1,
|
|
pages_with_text: 1,
|
|
confidence: 1.0,
|
|
title: None,
|
|
ocr_recommended: false,
|
|
pages_needing_ocr: Vec::new(),
|
|
};
|
|
assert!(detection_result.pages_needing_ocr.is_empty());
|
|
|
|
let process_result = pdf_inspector::PdfProcessResult {
|
|
pdf_type: PdfType::TextBased,
|
|
markdown: None,
|
|
page_count: 1,
|
|
processing_time_ms: 0,
|
|
pages_needing_ocr: vec![1, 3],
|
|
title: None,
|
|
confidence: 1.0,
|
|
layout: pdf_inspector::LayoutComplexity::default(),
|
|
has_encoding_issues: false,
|
|
};
|
|
assert_eq!(process_result.pages_needing_ocr, vec![1, 3]);
|
|
}
|
|
|
|
#[test]
|
|
fn test_text_pdf_process_result_empty_ocr_pages() {
|
|
// A minimal valid PDF that is text-based should have empty pages_needing_ocr.
|
|
// We use a minimal PDF buffer with a text content stream.
|
|
let pdf_bytes = b"%PDF-1.0
|
|
1 0 obj<</Type/Catalog/Pages 2 0 R>>endobj
|
|
2 0 obj<</Type/Pages/Kids[3 0 R]/Count 1>>endobj
|
|
3 0 obj<</Type/Page/MediaBox[0 0 612 792]/Parent 2 0 R/Contents 4 0 R>>endobj
|
|
4 0 obj<</Length 44>>
|
|
stream
|
|
BT /F1 12 Tf 100 700 Td (Hello World) Tj ET
|
|
endstream
|
|
endobj
|
|
xref
|
|
0 5
|
|
0000000000 65535 f
|
|
0000000009 00000 n
|
|
0000000058 00000 n
|
|
0000000115 00000 n
|
|
0000000206 00000 n
|
|
trailer<</Size 5/Root 1 0 R>>
|
|
startxref
|
|
300
|
|
%%EOF";
|
|
let result = pdf_inspector::process_pdf_mem(pdf_bytes);
|
|
// The minimal PDF may fail to parse fully, but if it succeeds,
|
|
// a text-based PDF should have empty pages_needing_ocr.
|
|
if let Ok(result) = result {
|
|
assert!(
|
|
result.pages_needing_ocr.is_empty(),
|
|
"Text-based PDF should have empty pages_needing_ocr, got: {:?}",
|
|
result.pages_needing_ocr
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_firecrawl_tagged_pdf_struct_tree() {
|
|
use lopdf::Document;
|
|
use pdf_inspector::structure_tree::{StructRole, StructTree};
|
|
|
|
let doc = Document::load("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
|
|
let tree = StructTree::from_doc(&doc).expect("Should have a structure tree");
|
|
|
|
// Verify structure tree contains expected roles
|
|
let page_ids = doc.get_pages();
|
|
let roles = tree.mcid_to_roles(&page_ids);
|
|
assert!(!roles.is_empty(), "Should have MCID roles across pages");
|
|
|
|
let flat = tree.flatten();
|
|
let has_code = flat.iter().any(|e| matches!(e.role, StructRole::Code));
|
|
let has_h1 = flat.iter().any(|e| matches!(e.role, StructRole::H1));
|
|
let has_li = flat.iter().any(|e| matches!(e.role, StructRole::LI));
|
|
let has_caption = flat.iter().any(|e| matches!(e.role, StructRole::Caption));
|
|
assert!(has_code, "Should have Code elements");
|
|
assert!(has_h1, "Should have H1 elements");
|
|
assert!(has_li, "Should have LI elements");
|
|
assert!(has_caption, "Should have Caption elements");
|
|
|
|
// Full conversion: code fences should be generated from Code struct elements
|
|
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
|
|
let result = pdf_inspector::process_pdf_mem(&buf).unwrap();
|
|
let md = result.markdown.unwrap();
|
|
let fence_count = md.matches("```").count();
|
|
assert!(
|
|
fence_count > 0,
|
|
"Should produce code fences from tagged Code elements"
|
|
);
|
|
// Fences come in open/close pairs
|
|
assert_eq!(fence_count % 2, 0, "Code fences should be balanced");
|
|
}
|
|
|
|
#[test]
|
|
fn test_identity_h_no_tounicode_suppresses_garbage() {
|
|
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
|
|
// ToUnicode CMap. The raw CID values look like random Latin characters.
|
|
// We should suppress the garbage and flag the page for OCR.
|
|
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
|
let result = pdf_inspector::process_pdf_mem(&buf).unwrap();
|
|
|
|
// Page 1 should be flagged for OCR
|
|
assert!(
|
|
result.pages_needing_ocr.contains(&1),
|
|
"Page with Identity-H font without ToUnicode should be flagged for OCR"
|
|
);
|
|
|
|
// Markdown should be empty (garbage suppressed)
|
|
let md = result.markdown.unwrap_or_default();
|
|
assert!(
|
|
md.trim().is_empty(),
|
|
"Garbage CID text should be suppressed, got {} chars: {:?}",
|
|
md.len(),
|
|
&md[..md.len().min(100)]
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_rotated_table_layout_correction() {
|
|
// tnagriculture_06_12.pdf has landscape content in a portrait page via
|
|
// a 90° CCW text matrix [0, b, -b, 0, tx, ty]. Without rotation
|
|
// correction, the table is read sideways (jumbled numbers).
|
|
let result =
|
|
process_pdf_with_options("tests/fixtures/tnagriculture_06_12.pdf", PdfOptions::new())
|
|
.unwrap();
|
|
let md = result.markdown.unwrap_or_default();
|
|
|
|
// Title should appear near the top
|
|
assert!(
|
|
md.contains("DISTRICT WISE PRODUCTION OF SPICES AND CONDIMENTS"),
|
|
"Should extract the table title"
|
|
);
|
|
|
|
// District names should be readable (not jumbled with numbers)
|
|
assert!(
|
|
md.contains("Ariyalur"),
|
|
"Should extract district name Ariyalur"
|
|
);
|
|
assert!(
|
|
md.contains("Coimbatore"),
|
|
"Should extract district name Coimbatore"
|
|
);
|
|
|
|
// Spice column headers should appear
|
|
assert!(
|
|
md.contains("CARDAMOM"),
|
|
"Should extract spice header CARDAMOM"
|
|
);
|
|
assert!(
|
|
md.contains("RED CHILLIES"),
|
|
"Should extract spice header RED CHILLIES"
|
|
);
|
|
|
|
// Table should be formatted as markdown table (has pipe delimiters)
|
|
let has_table_row = md
|
|
.lines()
|
|
.any(|l: &str| l.contains('|') && l.contains("Ariyalur"));
|
|
assert!(
|
|
has_table_row,
|
|
"District data should be in a markdown table row"
|
|
);
|
|
}
|
|
|
|
// =========================================================================
|
|
// extract_text_in_regions_mem tests
|
|
// =========================================================================
|
|
|
|
/// Build full-page region args for `page_count` pages.
|
|
/// Uses a generously large bbox (1200x1200) to capture any page size.
|
|
fn full_page_regions(page_count: u32) -> Vec<(u32, Vec<[f32; 4]>)> {
|
|
(0..page_count)
|
|
.map(|p| (p, vec![[0.0, 0.0, 1200.0, 1200.0]]))
|
|
.collect()
|
|
}
|
|
|
|
/// Normalize text for comparison: lowercase, strip non-alphanumeric, split into words.
|
|
fn normalize_words(text: &str) -> HashSet<String> {
|
|
text.split(|c: char| !c.is_alphanumeric())
|
|
.map(|w| w.to_lowercase())
|
|
.filter(|w| w.len() > 3)
|
|
.collect()
|
|
}
|
|
|
|
/// Fraction of normalized words in `a` that also appear in `b`.
|
|
fn word_overlap_ratio(a: &str, b: &str) -> f64 {
|
|
let words_a = normalize_words(a);
|
|
if words_a.is_empty() {
|
|
return if normalize_words(b).is_empty() {
|
|
1.0
|
|
} else {
|
|
0.0
|
|
};
|
|
}
|
|
let words_b = normalize_words(b);
|
|
let overlap = words_a.intersection(&words_b).count();
|
|
overlap as f64 / words_a.len() as f64
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_basic_text_pdf() {
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let result = process_pdf_mem(&buf).unwrap();
|
|
let page_count = result.page_count;
|
|
|
|
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(page_count)).unwrap();
|
|
assert_eq!(regions.len(), page_count as usize);
|
|
|
|
// Each result should have exactly 1 region (we passed one per page)
|
|
for r in ®ions {
|
|
assert_eq!(r.regions.len(), 1);
|
|
}
|
|
|
|
// First page should have non-empty text
|
|
let first = ®ions[0].regions[0];
|
|
assert!(!first.text.trim().is_empty(), "First page should have text");
|
|
assert_eq!(regions[0].page, 0);
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_identity_h_needs_ocr() {
|
|
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
|
let regions =
|
|
extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
|
assert_eq!(regions.len(), 1);
|
|
assert!(
|
|
regions[0].regions[0].needs_ocr,
|
|
"Identity-H font without ToUnicode should trigger needs_ocr"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_multiple_regions_per_page() {
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let regions = extract_text_in_regions_mem(
|
|
&buf,
|
|
&[(
|
|
0,
|
|
vec![
|
|
[0.0, 0.0, 300.0, 100.0], // small top-left
|
|
[0.0, 0.0, 1200.0, 1200.0], // full page
|
|
],
|
|
)],
|
|
)
|
|
.unwrap();
|
|
|
|
assert_eq!(regions.len(), 1);
|
|
assert_eq!(regions[0].regions.len(), 2);
|
|
|
|
let small_len = regions[0].regions[0].text.len();
|
|
let full_len = regions[0].regions[1].text.len();
|
|
assert!(
|
|
full_len >= small_len,
|
|
"Full-page region ({full_len}) should have at least as much text as small region ({small_len})"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_nonexistent_page() {
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let regions =
|
|
extract_text_in_regions_mem(&buf, &[(9999, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
|
assert_eq!(regions.len(), 1);
|
|
assert!(
|
|
regions[0].regions[0].needs_ocr,
|
|
"Nonexistent page should trigger needs_ocr"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_empty_region() {
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let regions = extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 0.0, 0.0]])]).unwrap();
|
|
assert_eq!(regions.len(), 1);
|
|
assert!(
|
|
regions[0].regions[0].needs_ocr,
|
|
"Zero-area region should trigger needs_ocr"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_not_a_pdf() {
|
|
let result = extract_text_in_regions_mem(b"not a pdf", &[(0, vec![[0.0, 0.0, 100.0, 100.0]])]);
|
|
assert!(result.is_err(), "Non-PDF input should return an error");
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_regions_mem_rotated_page_not_false_empty() {
|
|
let buf = std::fs::read("tests/fixtures/tnagriculture_06_12.pdf").unwrap();
|
|
let regions =
|
|
extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
|
assert_eq!(regions.len(), 1);
|
|
assert_eq!(regions[0].regions.len(), 1);
|
|
let region = ®ions[0].regions[0];
|
|
assert!(
|
|
!region.text.trim().is_empty(),
|
|
"Rotated page full-region extraction should not be empty"
|
|
);
|
|
assert!(
|
|
!region.needs_ocr,
|
|
"Rotated page with native text should not be flagged for OCR fallback"
|
|
);
|
|
assert!(
|
|
region
|
|
.text
|
|
.contains("DISTRICT WISE PRODUCTION OF SPICES AND CONDIMENTS"),
|
|
"Expected known title from rotated fixture in extracted region text"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_collect_text_in_region_keeps_partial_overlap_items() {
|
|
let item = make_text_item("EdgeWord", 100.0, 700.0, 12.0, 1);
|
|
// Region intersects only the left edge of the item. Center x=124 falls
|
|
// outside x=[95,120], so center-only containment would drop it.
|
|
let text = pdf_inspector::collect_text_in_region(&[item], 95.0, 80.0, 120.0, 110.0, 800.0);
|
|
assert!(
|
|
text.contains("EdgeWord"),
|
|
"Partially overlapping items should be retained in region extraction"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_collect_text_in_region_uses_rtl_sorting() {
|
|
let items = vec![
|
|
make_text_item("بكم", 240.0, 700.0, 12.0, 1),
|
|
make_text_item("مرحبا", 300.0, 700.0, 12.0, 1),
|
|
];
|
|
let text = pdf_inspector::collect_text_in_region(&items, 0.0, 0.0, 600.0, 800.0, 800.0);
|
|
assert_eq!(
|
|
text, "مرحبا بكم",
|
|
"Region path should reuse RTL-aware line sorting"
|
|
);
|
|
}
|
|
|
|
// =========================================================================
|
|
// Fast vs normal extraction comparison
|
|
// =========================================================================
|
|
|
|
/// For each text-based fixture PDF, compare `extract_text_in_regions_mem` (fast path)
|
|
/// against `process_pdf_mem` (normal path). If the fast path claims needs_ocr=false
|
|
/// for a page, verify the extracted text has meaningful overlap with the normal
|
|
/// markdown output — catching silent quality regressions.
|
|
#[test]
|
|
fn test_extract_regions_fast_vs_normal_comparison() {
|
|
let fixtures = [
|
|
"tests/fixtures/nexo-price-en.pdf",
|
|
"tests/fixtures/td9264.pdf",
|
|
"tests/fixtures/p1244-1996.pdf",
|
|
"tests/fixtures/real-estate-pricing.pdf",
|
|
"tests/fixtures/2013-app2.pdf",
|
|
"tests/fixtures/firecrawl_docs_tagged.pdf",
|
|
"tests/fixtures/thermo-freon12.pdf",
|
|
];
|
|
|
|
for fixture in &fixtures {
|
|
let buf = std::fs::read(fixture).unwrap();
|
|
let normal = process_pdf_mem(&buf).unwrap();
|
|
let normal_md = normal.markdown.as_deref().unwrap_or("");
|
|
let page_count = normal.page_count;
|
|
let ocr_pages: HashSet<u32> = normal.pages_needing_ocr.iter().copied().collect();
|
|
|
|
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(page_count)).unwrap();
|
|
|
|
assert_eq!(
|
|
regions.len(),
|
|
page_count as usize,
|
|
"{fixture}: result count should match page count"
|
|
);
|
|
|
|
for pr in ®ions {
|
|
let region = &pr.regions[0];
|
|
if !region.needs_ocr && !region.text.trim().is_empty() {
|
|
// Fast path claims this text is trustworthy.
|
|
// Check that its words appear in the normal markdown output.
|
|
let overlap = word_overlap_ratio(®ion.text, normal_md);
|
|
assert!(
|
|
overlap >= 0.3,
|
|
"{fixture} page {}: fast path says needs_ocr=false but only {:.0}% word \
|
|
overlap with normal extraction (threshold 30%). \
|
|
Fast text sample: {:?}",
|
|
pr.page,
|
|
overlap * 100.0,
|
|
®ion.text[..region.text.len().min(200)],
|
|
);
|
|
}
|
|
|
|
// If fast path flags needs_ocr but normal path didn't, that's overly
|
|
// conservative but not a bug — just worth knowing.
|
|
if region.needs_ocr && !ocr_pages.contains(&(pr.page + 1)) {
|
|
eprintln!(
|
|
"INFO: {fixture} page {}: fast path says needs_ocr=true but normal path extracted fine (conservative, not a bug)",
|
|
pr.page,
|
|
);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// =========================================================================
|
|
// extract_tables_in_regions_mem tests
|
|
// =========================================================================
|
|
|
|
#[test]
|
|
fn test_extract_tables_in_regions_table_pdf() {
|
|
// tnagriculture has a clear table with district names and spice columns
|
|
let buf = std::fs::read("tests/fixtures/tnagriculture_06_12.pdf").unwrap();
|
|
let results =
|
|
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
|
|
|
assert_eq!(results.len(), 1);
|
|
assert_eq!(results[0].regions.len(), 1);
|
|
|
|
let region = &results[0].regions[0];
|
|
// Should detect a table with pipe-delimited markdown
|
|
if !region.needs_ocr {
|
|
assert!(
|
|
region.text.contains('|'),
|
|
"Table output should contain pipe delimiters"
|
|
);
|
|
// Should have separator row
|
|
assert!(
|
|
region.text.lines().any(|l| l.contains("---")),
|
|
"Table output should contain separator row"
|
|
);
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_tables_in_regions_non_table_region() {
|
|
// Use a small region that likely won't contain enough items for a table
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let results =
|
|
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 50.0, 50.0]])]).unwrap();
|
|
|
|
assert_eq!(results.len(), 1);
|
|
assert_eq!(results[0].regions.len(), 1);
|
|
|
|
let region = &results[0].regions[0];
|
|
// Small region with few items should fall back to needs_ocr
|
|
assert!(
|
|
region.needs_ocr,
|
|
"Non-table region should set needs_ocr = true"
|
|
);
|
|
assert!(
|
|
region.text.is_empty(),
|
|
"Non-table region should have empty text"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_tables_in_regions_empty_region() {
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let results = extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 0.0, 0.0]])]).unwrap();
|
|
|
|
assert_eq!(results.len(), 1);
|
|
let region = &results[0].regions[0];
|
|
assert!(region.needs_ocr);
|
|
assert!(region.text.is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_tables_in_regions_identity_h_needs_ocr() {
|
|
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
|
let results =
|
|
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
|
|
|
assert_eq!(results.len(), 1);
|
|
let region = &results[0].regions[0];
|
|
assert!(region.needs_ocr, "Identity-H font should trigger needs_ocr");
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_tables_in_regions_not_a_pdf() {
|
|
let result =
|
|
extract_tables_in_regions_mem(b"not a pdf", &[(0, vec![[0.0, 0.0, 100.0, 100.0]])]);
|
|
assert!(result.is_err());
|
|
}
|
|
|
|
#[test]
|
|
fn test_extract_tables_in_regions_nonexistent_page() {
|
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
|
let results =
|
|
extract_tables_in_regions_mem(&buf, &[(9999, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
|
|
|
assert_eq!(results.len(), 1);
|
|
let region = &results[0].regions[0];
|
|
assert!(region.needs_ocr);
|
|
assert!(region.text.is_empty());
|
|
}
|