Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bc1b9ad553 |
+91
-1
@@ -1,6 +1,9 @@
|
|||||||
//! CLI tool for PDF to Markdown conversion
|
//! CLI tool for PDF to Markdown conversion
|
||||||
|
|
||||||
use pdf_inspector::{process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode};
|
use pdf_inspector::{
|
||||||
|
detect_layout, process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode,
|
||||||
|
RegionType,
|
||||||
|
};
|
||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
use std::env;
|
use std::env;
|
||||||
use std::fmt::Write;
|
use std::fmt::Write;
|
||||||
@@ -100,6 +103,7 @@ fn main() {
|
|||||||
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
||||||
eprintln!(" --detect-only Only detect PDF type (no extraction)");
|
eprintln!(" --detect-only Only detect PDF type (no extraction)");
|
||||||
eprintln!(" --analyze Detect + extract + layout analysis (no markdown)");
|
eprintln!(" --analyze Detect + extract + layout analysis (no markdown)");
|
||||||
|
eprintln!(" --layout Detect layout regions (tables, formulas, images, text)");
|
||||||
process::exit(1);
|
process::exit(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -109,6 +113,7 @@ fn main() {
|
|||||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||||
let analyze = args.iter().any(|a| a == "--analyze");
|
let analyze = args.iter().any(|a| a == "--analyze");
|
||||||
|
let layout_mode = args.iter().any(|a| a == "--layout");
|
||||||
|
|
||||||
// Parse --select-pages value
|
// Parse --select-pages value
|
||||||
let page_filter = args
|
let page_filter = args
|
||||||
@@ -148,6 +153,28 @@ fn main() {
|
|||||||
options.page_filter = Some(pages);
|
options.page_filter = Some(pages);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Layout region detection mode
|
||||||
|
if layout_mode {
|
||||||
|
match detect_layout(pdf_path) {
|
||||||
|
Ok(pages) => {
|
||||||
|
if json_output {
|
||||||
|
print_layout_json(&pages);
|
||||||
|
} else {
|
||||||
|
print_layout_text(&pages);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
if json_output {
|
||||||
|
println!(r#"{{"error":"{}"}}"#, e);
|
||||||
|
} else {
|
||||||
|
eprintln!("Error: {}", e);
|
||||||
|
}
|
||||||
|
process::exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
match process_pdf_with_options(pdf_path, options) {
|
match process_pdf_with_options(pdf_path, options) {
|
||||||
Ok(result) => {
|
Ok(result) => {
|
||||||
if detect_only || analyze {
|
if detect_only || analyze {
|
||||||
@@ -341,3 +368,66 @@ fn main() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn region_type_str(rt: &RegionType) -> &'static str {
|
||||||
|
match rt {
|
||||||
|
RegionType::Table => "table",
|
||||||
|
RegionType::Formula => "formula",
|
||||||
|
RegionType::Image => "image",
|
||||||
|
RegionType::Paragraph => "paragraph",
|
||||||
|
RegionType::Heading { .. } => "heading",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn print_layout_json(pages: &[pdf_inspector::PageLayout]) {
|
||||||
|
let mut out = String::from(r#"{"pages":["#);
|
||||||
|
for (pi, page) in pages.iter().enumerate() {
|
||||||
|
if pi > 0 {
|
||||||
|
out.push(',');
|
||||||
|
}
|
||||||
|
let _ = write!(out, r#"{{"page":{},"regions":["#, page.page);
|
||||||
|
for (ri, r) in page.regions.iter().enumerate() {
|
||||||
|
if ri > 0 {
|
||||||
|
out.push(',');
|
||||||
|
}
|
||||||
|
let type_str = region_type_str(&r.region_type);
|
||||||
|
let _ = write!(out, r#"{{"type":"{}""#, type_str,);
|
||||||
|
if let RegionType::Heading { level } = r.region_type {
|
||||||
|
let _ = write!(out, r#","level":{}"#, level);
|
||||||
|
}
|
||||||
|
let _ = write!(
|
||||||
|
out,
|
||||||
|
r#","bbox":{{"x_min":{:.1},"y_min":{:.1},"x_max":{:.1},"y_max":{:.1}}},"confidence":{:.2},"needs_ocr":{}}}"#,
|
||||||
|
r.bbox.x_min, r.bbox.y_min, r.bbox.x_max, r.bbox.y_max, r.confidence, r.needs_ocr,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
out.push_str("]}");
|
||||||
|
}
|
||||||
|
out.push_str("]}");
|
||||||
|
println!("{}", out);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn print_layout_text(pages: &[pdf_inspector::PageLayout]) {
|
||||||
|
for page in pages {
|
||||||
|
eprintln!("Page {}:", page.page);
|
||||||
|
for r in &page.regions {
|
||||||
|
let type_str = region_type_str(&r.region_type);
|
||||||
|
let level_suffix = if let RegionType::Heading { level } = r.region_type {
|
||||||
|
format!(" (H{})", level)
|
||||||
|
} else {
|
||||||
|
String::new()
|
||||||
|
};
|
||||||
|
eprintln!(
|
||||||
|
" {}{} [{:.0},{:.0} - {:.0},{:.0}] conf={:.2} ocr={}",
|
||||||
|
type_str,
|
||||||
|
level_suffix,
|
||||||
|
r.bbox.x_min,
|
||||||
|
r.bbox.y_min,
|
||||||
|
r.bbox.x_max,
|
||||||
|
r.bbox.y_max,
|
||||||
|
r.confidence,
|
||||||
|
r.needs_ocr,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,282 @@
|
|||||||
|
//! Formula/math region detection via font names, Unicode ranges, and sub/super density.
|
||||||
|
|
||||||
|
use crate::types::{BBox, ItemType, LayoutRegion, RegionType, TextItem};
|
||||||
|
|
||||||
|
/// Known math font prefixes (case-insensitive matching).
|
||||||
|
const MATH_FONT_PREFIXES: &[&str] = &[
|
||||||
|
"cmmi",
|
||||||
|
"cmsy",
|
||||||
|
"cmr",
|
||||||
|
"cmex",
|
||||||
|
"cmbx",
|
||||||
|
"cmti", // Computer Modern
|
||||||
|
"stix",
|
||||||
|
"stixmath", // STIX
|
||||||
|
"mathjax", // MathJax
|
||||||
|
"cambria math",
|
||||||
|
"cambriamath", // Cambria Math
|
||||||
|
"asana-math",
|
||||||
|
"asanamath", // Asana Math
|
||||||
|
"latin modern math", // Latin Modern
|
||||||
|
"xits math",
|
||||||
|
"xitsmath", // XITS
|
||||||
|
"mt extra", // MT Extra
|
||||||
|
];
|
||||||
|
|
||||||
|
/// Check if a font name indicates a math font.
|
||||||
|
fn is_math_font(font_name: &str) -> bool {
|
||||||
|
let lower = font_name.to_ascii_lowercase();
|
||||||
|
// Check known prefixes
|
||||||
|
if MATH_FONT_PREFIXES
|
||||||
|
.iter()
|
||||||
|
.any(|prefix| lower.starts_with(prefix))
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
// Generic: font name contains "math" (but not just "Mathilda" etc.)
|
||||||
|
if lower.contains("math") {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check if a character is in a math-related Unicode range.
|
||||||
|
fn is_math_char(c: char) -> bool {
|
||||||
|
matches!(c,
|
||||||
|
// Mathematical Operators
|
||||||
|
'\u{2200}'..='\u{22FF}' |
|
||||||
|
// Supplemental Math Operators
|
||||||
|
'\u{2A00}'..='\u{2AFF}' |
|
||||||
|
// Miscellaneous Mathematical Symbols A & B
|
||||||
|
'\u{27C0}'..='\u{27EF}' |
|
||||||
|
'\u{2980}'..='\u{29FF}' |
|
||||||
|
// Mathematical Alphanumeric Symbols
|
||||||
|
'\u{1D400}'..='\u{1D7FF}' |
|
||||||
|
// Arrows (often in math)
|
||||||
|
'\u{2190}'..='\u{21FF}' |
|
||||||
|
// Superscript/subscript block
|
||||||
|
'\u{2070}'..='\u{209F}' |
|
||||||
|
// Standalone math symbols not covered by ranges above
|
||||||
|
'\u{00B1}' | // ±
|
||||||
|
'\u{00D7}' | // ×
|
||||||
|
'\u{00F7}' | // ÷
|
||||||
|
// Greek letters (often used as math variables)
|
||||||
|
'\u{0391}'..='\u{03C9}'
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns true if the item has math signals (font or content).
|
||||||
|
fn has_math_signal(item: &TextItem) -> bool {
|
||||||
|
if is_math_font(&item.font) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
// Check if a significant portion of characters are math
|
||||||
|
let total = item.text.chars().count();
|
||||||
|
if total == 0 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let math_count = item.text.chars().filter(|&c| is_math_char(c)).count();
|
||||||
|
// For short items, even 1 math char counts. For longer, require 30%+.
|
||||||
|
if total <= 3 {
|
||||||
|
math_count >= 1
|
||||||
|
} else {
|
||||||
|
math_count as f32 / total as f32 >= 0.3
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect formula regions on a single page.
|
||||||
|
///
|
||||||
|
/// Groups adjacent items with math signals (font, Unicode, sub/super) into
|
||||||
|
/// formula regions. Requires a cluster of 3+ items to avoid false positives
|
||||||
|
/// from isolated math symbols in prose.
|
||||||
|
pub(crate) fn detect_formula_regions(page_items: &[TextItem], page: u32) -> Vec<LayoutRegion> {
|
||||||
|
// Score each item for math signals
|
||||||
|
let mut math_items: Vec<(usize, &TextItem)> = Vec::new();
|
||||||
|
for (i, item) in page_items.iter().enumerate() {
|
||||||
|
if !matches!(item.item_type, ItemType::Text) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if has_math_signal(item) {
|
||||||
|
math_items.push((i, item));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if math_items.is_empty() {
|
||||||
|
return Vec::new();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Detect sub/super density: items near math items with smaller font + y offset
|
||||||
|
let sub_super_indices: Vec<usize> = page_items
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.filter_map(|(i, item)| {
|
||||||
|
if !matches!(item.item_type, ItemType::Text) {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
// Check if this item is sub/super relative to any neighbor
|
||||||
|
for (_, math_item) in &math_items {
|
||||||
|
let font_ratio = item.font_size / math_item.font_size;
|
||||||
|
let y_diff = (item.y - math_item.y).abs();
|
||||||
|
let x_near = (item.x - (math_item.x + math_item.width)).abs()
|
||||||
|
< math_item.font_size * 2.0
|
||||||
|
|| (math_item.x - (item.x + item.width)).abs() < math_item.font_size * 2.0;
|
||||||
|
if font_ratio < 0.85 && y_diff > 1.0 && y_diff < math_item.font_size && x_near {
|
||||||
|
return Some(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
None
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// Merge math item indices + sub/super indices
|
||||||
|
let mut all_math_indices: Vec<usize> = math_items.iter().map(|(i, _)| *i).collect();
|
||||||
|
all_math_indices.extend(sub_super_indices);
|
||||||
|
all_math_indices.sort_unstable();
|
||||||
|
all_math_indices.dedup();
|
||||||
|
|
||||||
|
// Cluster into contiguous regions by y-proximity
|
||||||
|
let mut clusters: Vec<Vec<usize>> = Vec::new();
|
||||||
|
let mut current_cluster: Vec<usize> = Vec::new();
|
||||||
|
|
||||||
|
for &idx in &all_math_indices {
|
||||||
|
let item = &page_items[idx];
|
||||||
|
if current_cluster.is_empty() {
|
||||||
|
current_cluster.push(idx);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let last_item = &page_items[*current_cluster.last().unwrap()];
|
||||||
|
let line_height = last_item.font_size.max(item.font_size) * 2.0;
|
||||||
|
let y_gap = (item.y - last_item.y).abs();
|
||||||
|
let x_gap = (item.x - (last_item.x + last_item.width)).abs();
|
||||||
|
|
||||||
|
// Same cluster if vertically close (within 2x line height)
|
||||||
|
// and not too far horizontally (within page width / 3)
|
||||||
|
if y_gap <= line_height && x_gap < 300.0 {
|
||||||
|
current_cluster.push(idx);
|
||||||
|
} else {
|
||||||
|
if current_cluster.len() >= 3 {
|
||||||
|
clusters.push(current_cluster.clone());
|
||||||
|
}
|
||||||
|
current_cluster = vec![idx];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if current_cluster.len() >= 3 {
|
||||||
|
clusters.push(current_cluster);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Convert clusters to regions
|
||||||
|
clusters
|
||||||
|
.iter()
|
||||||
|
.filter_map(|cluster| {
|
||||||
|
let items: Vec<&TextItem> = cluster.iter().map(|&i| &page_items[i]).collect();
|
||||||
|
let x_min = items.iter().map(|i| i.x).fold(f32::MAX, f32::min);
|
||||||
|
let x_max = items.iter().map(|i| i.x + i.width).fold(f32::MIN, f32::max);
|
||||||
|
let y_min = items.iter().map(|i| i.y).fold(f32::MAX, f32::min);
|
||||||
|
let y_max = items
|
||||||
|
.iter()
|
||||||
|
.map(|i| i.y + i.height)
|
||||||
|
.fold(f32::MIN, f32::max);
|
||||||
|
|
||||||
|
if x_max <= x_min || y_max <= y_min {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Count how many signals: font-based vs content-based
|
||||||
|
let font_signal_count = items.iter().filter(|i| is_math_font(&i.font)).count();
|
||||||
|
let confidence = if font_signal_count > items.len() / 2 {
|
||||||
|
0.85
|
||||||
|
} else {
|
||||||
|
0.7
|
||||||
|
};
|
||||||
|
|
||||||
|
Some(LayoutRegion {
|
||||||
|
bbox: BBox {
|
||||||
|
x_min,
|
||||||
|
y_min,
|
||||||
|
x_max,
|
||||||
|
y_max,
|
||||||
|
},
|
||||||
|
region_type: RegionType::Formula,
|
||||||
|
page,
|
||||||
|
confidence,
|
||||||
|
needs_ocr: true,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::types::ItemType;
|
||||||
|
|
||||||
|
fn make_item(text: &str, x: f32, y: f32, fs: f32, font: &str) -> TextItem {
|
||||||
|
TextItem {
|
||||||
|
text: text.into(),
|
||||||
|
x,
|
||||||
|
y,
|
||||||
|
width: text.len() as f32 * fs * 0.5,
|
||||||
|
height: fs,
|
||||||
|
font: font.into(),
|
||||||
|
font_size: fs,
|
||||||
|
page: 1,
|
||||||
|
is_bold: false,
|
||||||
|
is_italic: false,
|
||||||
|
item_type: ItemType::Text,
|
||||||
|
mcid: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn math_font_detection() {
|
||||||
|
assert!(is_math_font("CMMI10"));
|
||||||
|
assert!(is_math_font("CMSY8"));
|
||||||
|
assert!(is_math_font("CMR12"));
|
||||||
|
assert!(is_math_font("STIXMath-Regular"));
|
||||||
|
assert!(is_math_font("Cambria Math"));
|
||||||
|
assert!(is_math_font("MathJax_Main"));
|
||||||
|
assert!(!is_math_font("Helvetica"));
|
||||||
|
assert!(!is_math_font("TimesNewRoman"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn math_unicode_detection() {
|
||||||
|
assert!(is_math_char('∑'));
|
||||||
|
assert!(is_math_char('∫'));
|
||||||
|
assert!(is_math_char('α'));
|
||||||
|
assert!(is_math_char('√'));
|
||||||
|
assert!(is_math_char('≤'));
|
||||||
|
assert!(!is_math_char('A'));
|
||||||
|
assert!(!is_math_char('1'));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn formula_cluster_detection() {
|
||||||
|
// Simulate a simple equation: y = αx + β
|
||||||
|
let items = vec![
|
||||||
|
make_item("y", 100.0, 500.0, 12.0, "CMMI10"),
|
||||||
|
make_item("=", 115.0, 500.0, 12.0, "CMR10"),
|
||||||
|
make_item("α", 130.0, 500.0, 12.0, "CMMI10"),
|
||||||
|
make_item("x", 145.0, 500.0, 12.0, "CMMI10"),
|
||||||
|
make_item("+", 160.0, 500.0, 12.0, "CMR10"),
|
||||||
|
make_item("β", 175.0, 500.0, 12.0, "CMMI10"),
|
||||||
|
];
|
||||||
|
let regions = detect_formula_regions(&items, 1);
|
||||||
|
assert_eq!(regions.len(), 1);
|
||||||
|
assert_eq!(regions[0].region_type, RegionType::Formula);
|
||||||
|
assert!(regions[0].needs_ocr);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn isolated_math_symbol_not_detected() {
|
||||||
|
// A single "+" in running text should not create a formula region
|
||||||
|
let items = vec![
|
||||||
|
make_item("The", 100.0, 500.0, 12.0, "Helvetica"),
|
||||||
|
make_item("+", 140.0, 500.0, 12.0, "Helvetica"),
|
||||||
|
make_item("operator", 155.0, 500.0, 12.0, "Helvetica"),
|
||||||
|
];
|
||||||
|
let regions = detect_formula_regions(&items, 1);
|
||||||
|
assert!(regions.is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,223 @@
|
|||||||
|
//! Heuristic layout region detection for PDF pages.
|
||||||
|
//!
|
||||||
|
//! Detects tables, formulas, images, paragraphs, and headings from
|
||||||
|
//! extracted text items, rectangles, and line segments — no ML model needed.
|
||||||
|
//!
|
||||||
|
//! Primary entry point: [`detect_layout_regions`].
|
||||||
|
|
||||||
|
mod formula;
|
||||||
|
mod tables;
|
||||||
|
mod text_blocks;
|
||||||
|
|
||||||
|
use std::collections::HashSet;
|
||||||
|
|
||||||
|
use crate::markdown::analysis::calculate_font_stats_from_items;
|
||||||
|
use crate::types::{
|
||||||
|
BBox, ItemType, LayoutRegion, PageLayout, PdfLine, PdfRect, RegionType, TextItem,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Detect layout regions for all pages from pre-extracted PDF data.
|
||||||
|
///
|
||||||
|
/// Runs table, formula, image, and text-block detection per page and returns
|
||||||
|
/// a sorted list of [`PageLayout`] results. Orchestrators can inspect each
|
||||||
|
/// region's [`LayoutRegion::needs_ocr`] flag to decide routing.
|
||||||
|
pub fn detect_layout_regions(
|
||||||
|
items: &[TextItem],
|
||||||
|
rects: &[PdfRect],
|
||||||
|
lines: &[PdfLine],
|
||||||
|
) -> Vec<PageLayout> {
|
||||||
|
let font_stats = calculate_font_stats_from_items(items);
|
||||||
|
let base_size = font_stats.most_common_size;
|
||||||
|
|
||||||
|
// Collect unique pages
|
||||||
|
let mut pages: Vec<u32> = items.iter().map(|i| i.page).collect();
|
||||||
|
pages.sort_unstable();
|
||||||
|
pages.dedup();
|
||||||
|
|
||||||
|
pages
|
||||||
|
.into_iter()
|
||||||
|
.map(|page| {
|
||||||
|
let page_items: Vec<TextItem> =
|
||||||
|
items.iter().filter(|i| i.page == page).cloned().collect();
|
||||||
|
|
||||||
|
let regions = detect_page_regions(&page_items, rects, lines, page, base_size);
|
||||||
|
PageLayout { page, regions }
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect all region types for a single page.
|
||||||
|
fn detect_page_regions(
|
||||||
|
page_items: &[TextItem],
|
||||||
|
rects: &[PdfRect],
|
||||||
|
lines: &[PdfLine],
|
||||||
|
page: u32,
|
||||||
|
base_size: f32,
|
||||||
|
) -> Vec<LayoutRegion> {
|
||||||
|
let mut regions = Vec::new();
|
||||||
|
let mut claimed_indices: HashSet<usize> = HashSet::new();
|
||||||
|
|
||||||
|
// 1. Tables (highest priority — claim items first)
|
||||||
|
let table_regions = tables::detect_table_regions(page_items, rects, lines, page, base_size);
|
||||||
|
for tr in &table_regions {
|
||||||
|
// Mark items inside the table bbox as claimed
|
||||||
|
for (i, item) in page_items.iter().enumerate() {
|
||||||
|
if bbox_contains(&tr.bbox, item.x, item.y) {
|
||||||
|
claimed_indices.insert(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
regions.extend(table_regions);
|
||||||
|
|
||||||
|
// 2. Images
|
||||||
|
for (i, item) in page_items.iter().enumerate() {
|
||||||
|
if matches!(item.item_type, ItemType::Image) {
|
||||||
|
claimed_indices.insert(i);
|
||||||
|
regions.push(LayoutRegion {
|
||||||
|
bbox: BBox {
|
||||||
|
x_min: item.x,
|
||||||
|
y_min: item.y,
|
||||||
|
x_max: item.x + item.width,
|
||||||
|
y_max: item.y + item.height,
|
||||||
|
},
|
||||||
|
region_type: RegionType::Image,
|
||||||
|
page,
|
||||||
|
confidence: 1.0,
|
||||||
|
needs_ocr: false,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. Formulas (from unclaimed text items)
|
||||||
|
let unclaimed_items: Vec<TextItem> = page_items
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.filter(|(i, _)| !claimed_indices.contains(i))
|
||||||
|
.map(|(_, item)| item.clone())
|
||||||
|
.collect();
|
||||||
|
let formula_regions = formula::detect_formula_regions(&unclaimed_items, page);
|
||||||
|
for fr in &formula_regions {
|
||||||
|
// Mark items inside formula bbox as claimed
|
||||||
|
for (i, item) in page_items.iter().enumerate() {
|
||||||
|
if !claimed_indices.contains(&i) && bbox_contains(&fr.bbox, item.x, item.y) {
|
||||||
|
claimed_indices.insert(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
regions.extend(formula_regions);
|
||||||
|
|
||||||
|
// 4. Paragraphs and headings (from remaining unclaimed items)
|
||||||
|
let text_regions = text_blocks::detect_text_block_regions(page_items, page, &claimed_indices);
|
||||||
|
regions.extend(text_regions);
|
||||||
|
|
||||||
|
// Sort: top-to-bottom (descending y in PDF coords), then left-to-right
|
||||||
|
regions.sort_by(|a, b| {
|
||||||
|
b.bbox
|
||||||
|
.y_max
|
||||||
|
.total_cmp(&a.bbox.y_max)
|
||||||
|
.then(a.bbox.x_min.total_cmp(&b.bbox.x_min))
|
||||||
|
});
|
||||||
|
|
||||||
|
regions
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check if a point falls within a bounding box.
|
||||||
|
fn bbox_contains(bbox: &BBox, x: f32, y: f32) -> bool {
|
||||||
|
x >= bbox.x_min && x <= bbox.x_max && y >= bbox.y_min && y <= bbox.y_max
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::types::ItemType;
|
||||||
|
|
||||||
|
fn make_text(text: &str, x: f32, y: f32, fs: f32) -> TextItem {
|
||||||
|
TextItem {
|
||||||
|
text: text.into(),
|
||||||
|
x,
|
||||||
|
y,
|
||||||
|
width: text.len() as f32 * fs * 0.5,
|
||||||
|
height: fs,
|
||||||
|
font: "Helvetica".into(),
|
||||||
|
font_size: fs,
|
||||||
|
page: 1,
|
||||||
|
is_bold: false,
|
||||||
|
is_italic: false,
|
||||||
|
item_type: ItemType::Text,
|
||||||
|
mcid: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn make_image(x: f32, y: f32, w: f32, h: f32) -> TextItem {
|
||||||
|
TextItem {
|
||||||
|
text: String::new(),
|
||||||
|
x,
|
||||||
|
y,
|
||||||
|
width: w,
|
||||||
|
height: h,
|
||||||
|
font: String::new(),
|
||||||
|
font_size: 0.0,
|
||||||
|
page: 1,
|
||||||
|
is_bold: false,
|
||||||
|
is_italic: false,
|
||||||
|
item_type: ItemType::Image,
|
||||||
|
mcid: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn mixed_page_detects_multiple_region_types() {
|
||||||
|
let items = vec![
|
||||||
|
// Heading
|
||||||
|
make_text("Title", 72.0, 750.0, 24.0),
|
||||||
|
// Paragraph
|
||||||
|
make_text("Some body text here.", 72.0, 720.0, 12.0),
|
||||||
|
make_text("More body text follows.", 72.0, 706.0, 12.0),
|
||||||
|
// Image
|
||||||
|
make_image(72.0, 400.0, 200.0, 150.0),
|
||||||
|
];
|
||||||
|
|
||||||
|
let pages = detect_layout_regions(&items, &[], &[]);
|
||||||
|
assert_eq!(pages.len(), 1);
|
||||||
|
let regions = &pages[0].regions;
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
regions.iter().any(|r| r.region_type == RegionType::Image),
|
||||||
|
"expected image region"
|
||||||
|
);
|
||||||
|
// Should have heading + paragraph (or at least some text regions)
|
||||||
|
let text_region_count = regions
|
||||||
|
.iter()
|
||||||
|
.filter(|r| {
|
||||||
|
matches!(
|
||||||
|
r.region_type,
|
||||||
|
RegionType::Paragraph | RegionType::Heading { .. }
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.count();
|
||||||
|
assert!(
|
||||||
|
text_region_count >= 1,
|
||||||
|
"expected text regions, got {regions:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn empty_page_produces_no_regions() {
|
||||||
|
let pages = detect_layout_regions(&[], &[], &[]);
|
||||||
|
assert!(pages.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bbox_contains_works() {
|
||||||
|
let bbox = BBox {
|
||||||
|
x_min: 10.0,
|
||||||
|
y_min: 10.0,
|
||||||
|
x_max: 100.0,
|
||||||
|
y_max: 100.0,
|
||||||
|
};
|
||||||
|
assert!(bbox_contains(&bbox, 50.0, 50.0));
|
||||||
|
assert!(bbox_contains(&bbox, 10.0, 10.0)); // edge
|
||||||
|
assert!(!bbox_contains(&bbox, 5.0, 50.0)); // outside left
|
||||||
|
assert!(!bbox_contains(&bbox, 50.0, 105.0)); // outside top
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,221 @@
|
|||||||
|
//! Table region detection — wraps existing 3-strategy cascade and emits bounding boxes.
|
||||||
|
|
||||||
|
use crate::markdown;
|
||||||
|
use crate::tables;
|
||||||
|
use crate::types::{BBox, LayoutRegion, PdfLine, PdfRect, RegionType, TextItem};
|
||||||
|
|
||||||
|
/// Detection strategy used, determines confidence score.
|
||||||
|
enum Strategy {
|
||||||
|
Rect,
|
||||||
|
Line,
|
||||||
|
Heuristic,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect table regions on a single page.
|
||||||
|
///
|
||||||
|
/// Runs the rect → line → heuristic cascade, including side-by-side band
|
||||||
|
/// splitting (same logic as `compute_layout_complexity`).
|
||||||
|
pub(crate) fn detect_table_regions(
|
||||||
|
page_items: &[TextItem],
|
||||||
|
rects: &[PdfRect],
|
||||||
|
lines: &[PdfLine],
|
||||||
|
page: u32,
|
||||||
|
base_font_size: f32,
|
||||||
|
) -> Vec<LayoutRegion> {
|
||||||
|
let bands = markdown::split_side_by_side(page_items);
|
||||||
|
|
||||||
|
let band_ranges: Vec<(f32, f32)> = if bands.is_empty() {
|
||||||
|
vec![(f32::MIN, f32::MAX)]
|
||||||
|
} else {
|
||||||
|
bands
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut regions = Vec::new();
|
||||||
|
|
||||||
|
for &(x_lo, x_hi) in &band_ranges {
|
||||||
|
let is_full = x_lo == f32::MIN;
|
||||||
|
let margin = 2.0;
|
||||||
|
|
||||||
|
let band_items: Vec<TextItem> = page_items
|
||||||
|
.iter()
|
||||||
|
.filter(|item| is_full || (item.x >= x_lo - margin && item.x < x_hi + margin))
|
||||||
|
.cloned()
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let band_rects: Vec<PdfRect> = if is_full {
|
||||||
|
rects.iter().filter(|r| r.page == page).cloned().collect()
|
||||||
|
} else {
|
||||||
|
markdown::filter_rects_to_band(rects, page, x_lo, x_hi)
|
||||||
|
};
|
||||||
|
|
||||||
|
let band_lines: Vec<PdfLine> = if is_full {
|
||||||
|
lines.iter().filter(|l| l.page == page).cloned().collect()
|
||||||
|
} else {
|
||||||
|
markdown::filter_lines_to_band(lines, page, x_lo, x_hi)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Rect-based
|
||||||
|
let (rect_tables, _) = tables::detect_tables_from_rects(&band_items, &band_rects, page);
|
||||||
|
if !rect_tables.is_empty() {
|
||||||
|
for t in &rect_tables {
|
||||||
|
if let Some(r) = table_to_region(t, page, Strategy::Rect) {
|
||||||
|
regions.push(r);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Line-based
|
||||||
|
let line_tables = tables::detect_tables_from_lines(&band_items, &band_lines, page);
|
||||||
|
if !line_tables.is_empty() {
|
||||||
|
for t in &line_tables {
|
||||||
|
if let Some(r) = table_to_region(t, page, Strategy::Line) {
|
||||||
|
regions.push(r);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Heuristic
|
||||||
|
let heuristic_tables = tables::detect_tables(&band_items, base_font_size, false);
|
||||||
|
for t in &heuristic_tables {
|
||||||
|
if let Some(r) = table_to_region(t, page, Strategy::Heuristic) {
|
||||||
|
regions.push(r);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
regions
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Convert a `Table` to a `LayoutRegion` by deriving its bounding box.
|
||||||
|
fn table_to_region(table: &tables::Table, page: u32, strategy: Strategy) -> Option<LayoutRegion> {
|
||||||
|
let x_min = table.columns.first().copied()?;
|
||||||
|
let x_max = table.columns.last().copied()?;
|
||||||
|
// rows are in descending y order
|
||||||
|
let y_max = table.rows.first().copied()?;
|
||||||
|
let y_min = table.rows.last().copied()?;
|
||||||
|
|
||||||
|
let confidence = match strategy {
|
||||||
|
Strategy::Rect => 0.9,
|
||||||
|
Strategy::Line => 0.85,
|
||||||
|
Strategy::Heuristic => 0.7,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Small padding to encompass text slightly outside grid boundaries
|
||||||
|
let pad = 2.0;
|
||||||
|
Some(LayoutRegion {
|
||||||
|
bbox: BBox {
|
||||||
|
x_min: x_min - pad,
|
||||||
|
y_min: y_min - pad,
|
||||||
|
x_max: x_max + pad,
|
||||||
|
y_max: y_max + pad,
|
||||||
|
},
|
||||||
|
region_type: RegionType::Table,
|
||||||
|
page,
|
||||||
|
confidence,
|
||||||
|
needs_ocr: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::types::ItemType;
|
||||||
|
|
||||||
|
fn make_item(text: &str, x: f32, y: f32, fs: f32) -> TextItem {
|
||||||
|
TextItem {
|
||||||
|
text: text.into(),
|
||||||
|
x,
|
||||||
|
y,
|
||||||
|
width: 40.0,
|
||||||
|
height: fs,
|
||||||
|
font: "F1".into(),
|
||||||
|
font_size: fs,
|
||||||
|
page: 1,
|
||||||
|
is_bold: false,
|
||||||
|
is_italic: false,
|
||||||
|
item_type: ItemType::Text,
|
||||||
|
mcid: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn table_to_region_derives_bbox() {
|
||||||
|
let table = tables::Table {
|
||||||
|
columns: vec![100.0, 200.0, 300.0],
|
||||||
|
rows: vec![500.0, 480.0, 460.0],
|
||||||
|
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||||
|
item_indices: vec![0, 1, 2, 3],
|
||||||
|
};
|
||||||
|
let region = table_to_region(&table, 1, Strategy::Rect).unwrap();
|
||||||
|
assert_eq!(region.region_type, RegionType::Table);
|
||||||
|
assert!(region.needs_ocr);
|
||||||
|
assert!((region.confidence - 0.9).abs() < 0.01);
|
||||||
|
assert!(region.bbox.x_min < 100.0); // padded
|
||||||
|
assert!(region.bbox.x_max > 300.0);
|
||||||
|
assert!(region.bbox.y_min < 460.0);
|
||||||
|
assert!(region.bbox.y_max > 500.0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn detect_table_regions_with_rect_borders() {
|
||||||
|
let items = vec![
|
||||||
|
make_item("A", 100.0, 500.0, 10.0),
|
||||||
|
make_item("B", 200.0, 500.0, 10.0),
|
||||||
|
make_item("C", 100.0, 480.0, 10.0),
|
||||||
|
make_item("D", 200.0, 480.0, 10.0),
|
||||||
|
];
|
||||||
|
let rects = vec![
|
||||||
|
PdfRect {
|
||||||
|
x: 95.0,
|
||||||
|
y: 475.0,
|
||||||
|
width: 55.0,
|
||||||
|
height: 30.0,
|
||||||
|
page: 1,
|
||||||
|
},
|
||||||
|
PdfRect {
|
||||||
|
x: 150.0,
|
||||||
|
y: 475.0,
|
||||||
|
width: 55.0,
|
||||||
|
height: 30.0,
|
||||||
|
page: 1,
|
||||||
|
},
|
||||||
|
PdfRect {
|
||||||
|
x: 195.0,
|
||||||
|
y: 475.0,
|
||||||
|
width: 55.0,
|
||||||
|
height: 30.0,
|
||||||
|
page: 1,
|
||||||
|
},
|
||||||
|
PdfRect {
|
||||||
|
x: 95.0,
|
||||||
|
y: 495.0,
|
||||||
|
width: 55.0,
|
||||||
|
height: 30.0,
|
||||||
|
page: 1,
|
||||||
|
},
|
||||||
|
PdfRect {
|
||||||
|
x: 150.0,
|
||||||
|
y: 495.0,
|
||||||
|
width: 55.0,
|
||||||
|
height: 30.0,
|
||||||
|
page: 1,
|
||||||
|
},
|
||||||
|
PdfRect {
|
||||||
|
x: 195.0,
|
||||||
|
y: 495.0,
|
||||||
|
width: 55.0,
|
||||||
|
height: 30.0,
|
||||||
|
page: 1,
|
||||||
|
},
|
||||||
|
];
|
||||||
|
let regions = detect_table_regions(&items, &rects, &[], 1, 10.0);
|
||||||
|
// Should find at least one table region (rect-based)
|
||||||
|
assert!(
|
||||||
|
regions.iter().any(|r| r.region_type == RegionType::Table),
|
||||||
|
"expected at least one table region, got {:?}",
|
||||||
|
regions
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,202 @@
|
|||||||
|
//! Paragraph and heading region detection from remaining text items.
|
||||||
|
|
||||||
|
use std::collections::HashSet;
|
||||||
|
|
||||||
|
use crate::extractor::group_into_lines;
|
||||||
|
use crate::markdown::analysis::{
|
||||||
|
calculate_font_stats_from_items, compute_heading_tiers, detect_header_level,
|
||||||
|
};
|
||||||
|
use crate::types::{BBox, ItemType, LayoutRegion, RegionType, TextItem};
|
||||||
|
|
||||||
|
/// Detect paragraph and heading regions from text items, excluding items
|
||||||
|
/// already claimed by table/formula/image regions.
|
||||||
|
///
|
||||||
|
/// `excluded_indices` are indices into `page_items` that belong to other regions.
|
||||||
|
pub(crate) fn detect_text_block_regions(
|
||||||
|
page_items: &[TextItem],
|
||||||
|
page: u32,
|
||||||
|
excluded_indices: &HashSet<usize>,
|
||||||
|
) -> Vec<LayoutRegion> {
|
||||||
|
// Filter to unclaimed text items
|
||||||
|
let remaining: Vec<TextItem> = page_items
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.filter(|(i, item)| {
|
||||||
|
!excluded_indices.contains(i) && matches!(item.item_type, ItemType::Text)
|
||||||
|
})
|
||||||
|
.map(|(_, item)| item.clone())
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
if remaining.is_empty() {
|
||||||
|
return Vec::new();
|
||||||
|
}
|
||||||
|
|
||||||
|
// Compute font stats and heading tiers
|
||||||
|
let font_stats = calculate_font_stats_from_items(&remaining);
|
||||||
|
let base_size = font_stats.most_common_size;
|
||||||
|
let lines = group_into_lines(remaining);
|
||||||
|
let heading_tiers = compute_heading_tiers(&lines, base_size);
|
||||||
|
|
||||||
|
// Classify each line as heading or paragraph, then group into blocks
|
||||||
|
let mut regions = Vec::new();
|
||||||
|
let mut current_type: Option<RegionType> = None;
|
||||||
|
let mut block_x_min = f32::MAX;
|
||||||
|
let mut block_x_max = f32::MIN;
|
||||||
|
let mut block_y_min = f32::MAX;
|
||||||
|
let mut block_y_max = f32::MIN;
|
||||||
|
let mut prev_y: Option<f32> = None;
|
||||||
|
|
||||||
|
for line in &lines {
|
||||||
|
let first = match line.items.first() {
|
||||||
|
Some(f) => f,
|
||||||
|
None => continue,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Classify this line
|
||||||
|
let heading_level = detect_header_level(first.font_size, base_size, &heading_tiers);
|
||||||
|
let line_type = if let Some(level) = heading_level {
|
||||||
|
RegionType::Heading { level: level as u8 }
|
||||||
|
} else {
|
||||||
|
RegionType::Paragraph
|
||||||
|
};
|
||||||
|
|
||||||
|
// Compute line bbox
|
||||||
|
let line_x_min = line.items.iter().map(|i| i.x).fold(f32::MAX, f32::min);
|
||||||
|
let line_x_max = line
|
||||||
|
.items
|
||||||
|
.iter()
|
||||||
|
.map(|i| i.x + i.width)
|
||||||
|
.fold(f32::MIN, f32::max);
|
||||||
|
let line_y = first.y;
|
||||||
|
let line_height = first.height;
|
||||||
|
|
||||||
|
// Check if we should start a new block:
|
||||||
|
// 1. Type changed (heading vs paragraph, or different heading level)
|
||||||
|
// 2. Large vertical gap (> 1.5x base size)
|
||||||
|
let should_break = match (¤t_type, &line_type) {
|
||||||
|
(None, _) => false, // first line, no break
|
||||||
|
(Some(cur), new) if cur != new => true,
|
||||||
|
(Some(_), _) => {
|
||||||
|
// Same type — check vertical gap
|
||||||
|
if let Some(py) = prev_y {
|
||||||
|
(py - line_y).abs() > base_size * 1.5
|
||||||
|
} else {
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
if should_break {
|
||||||
|
// Emit current block
|
||||||
|
if current_type.is_some() && block_x_max > block_x_min && block_y_max > block_y_min {
|
||||||
|
regions.push(LayoutRegion {
|
||||||
|
bbox: BBox {
|
||||||
|
x_min: block_x_min,
|
||||||
|
y_min: block_y_min,
|
||||||
|
x_max: block_x_max,
|
||||||
|
y_max: block_y_max,
|
||||||
|
},
|
||||||
|
region_type: current_type.take().unwrap(),
|
||||||
|
page,
|
||||||
|
confidence: 0.85,
|
||||||
|
needs_ocr: false,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
// Reset
|
||||||
|
block_x_min = f32::MAX;
|
||||||
|
block_x_max = f32::MIN;
|
||||||
|
block_y_min = f32::MAX;
|
||||||
|
block_y_max = f32::MIN;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Accumulate into current block
|
||||||
|
current_type = Some(line_type);
|
||||||
|
block_x_min = block_x_min.min(line_x_min);
|
||||||
|
block_x_max = block_x_max.max(line_x_max);
|
||||||
|
block_y_min = block_y_min.min(line_y);
|
||||||
|
block_y_max = block_y_max.max(line_y + line_height);
|
||||||
|
prev_y = Some(line_y);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Emit final block
|
||||||
|
if let Some(rt) = current_type {
|
||||||
|
if block_x_max > block_x_min && block_y_max > block_y_min {
|
||||||
|
regions.push(LayoutRegion {
|
||||||
|
bbox: BBox {
|
||||||
|
x_min: block_x_min,
|
||||||
|
y_min: block_y_min,
|
||||||
|
x_max: block_x_max,
|
||||||
|
y_max: block_y_max,
|
||||||
|
},
|
||||||
|
region_type: rt,
|
||||||
|
page,
|
||||||
|
confidence: 0.85,
|
||||||
|
needs_ocr: false,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
regions
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn make_item(text: &str, x: f32, y: f32, fs: f32) -> TextItem {
|
||||||
|
TextItem {
|
||||||
|
text: text.into(),
|
||||||
|
x,
|
||||||
|
y,
|
||||||
|
width: text.len() as f32 * fs * 0.5,
|
||||||
|
height: fs,
|
||||||
|
font: "Helvetica".into(),
|
||||||
|
font_size: fs,
|
||||||
|
page: 1,
|
||||||
|
is_bold: false,
|
||||||
|
is_italic: false,
|
||||||
|
item_type: ItemType::Text,
|
||||||
|
mcid: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heading_then_paragraph() {
|
||||||
|
let items = vec![
|
||||||
|
// Heading (larger font)
|
||||||
|
make_item("Introduction", 72.0, 700.0, 20.0),
|
||||||
|
// Body text (base size = 10pt)
|
||||||
|
make_item("This is the first paragraph of text.", 72.0, 670.0, 10.0),
|
||||||
|
make_item("It continues on this line here.", 72.0, 658.0, 10.0),
|
||||||
|
make_item("And this is the third line.", 72.0, 646.0, 10.0),
|
||||||
|
];
|
||||||
|
let regions = detect_text_block_regions(&items, 1, &HashSet::new());
|
||||||
|
assert!(regions.len() >= 2, "got {:?}", regions);
|
||||||
|
assert!(
|
||||||
|
regions
|
||||||
|
.iter()
|
||||||
|
.any(|r| matches!(r.region_type, RegionType::Heading { .. })),
|
||||||
|
"expected a heading region"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
regions
|
||||||
|
.iter()
|
||||||
|
.any(|r| r.region_type == RegionType::Paragraph),
|
||||||
|
"expected a paragraph region"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn excluded_indices_skip_items() {
|
||||||
|
let items = vec![
|
||||||
|
make_item("Hello", 72.0, 700.0, 10.0),
|
||||||
|
make_item("Table content", 72.0, 680.0, 10.0),
|
||||||
|
make_item("World", 72.0, 660.0, 10.0),
|
||||||
|
];
|
||||||
|
let mut excluded = HashSet::new();
|
||||||
|
excluded.insert(1); // table content excluded
|
||||||
|
let regions = detect_text_block_regions(&items, 1, &excluded);
|
||||||
|
// Should have regions, but the table item is excluded
|
||||||
|
assert!(!regions.is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
+33
-1
@@ -29,6 +29,7 @@ pub mod adobe_korea1;
|
|||||||
pub mod detector;
|
pub mod detector;
|
||||||
pub mod extractor;
|
pub mod extractor;
|
||||||
pub mod glyph_names;
|
pub mod glyph_names;
|
||||||
|
pub mod layout;
|
||||||
pub mod markdown;
|
pub mod markdown;
|
||||||
pub mod process_mode;
|
pub mod process_mode;
|
||||||
pub mod structure_tree;
|
pub mod structure_tree;
|
||||||
@@ -42,11 +43,14 @@ pub use detector::{
|
|||||||
detect_pdf_type_with_config, DetectionConfig, PdfType, PdfTypeResult, ScanStrategy,
|
detect_pdf_type_with_config, DetectionConfig, PdfType, PdfTypeResult, ScanStrategy,
|
||||||
};
|
};
|
||||||
pub use extractor::{extract_text, extract_text_with_positions, extract_text_with_positions_pages};
|
pub use extractor::{extract_text, extract_text_with_positions, extract_text_with_positions_pages};
|
||||||
|
pub use layout::detect_layout_regions;
|
||||||
pub use markdown::{
|
pub use markdown::{
|
||||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||||
};
|
};
|
||||||
pub use process_mode::ProcessMode;
|
pub use process_mode::ProcessMode;
|
||||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
pub use types::{
|
||||||
|
BBox, LayoutComplexity, LayoutRegion, PageLayout, PdfLine, PdfRect, RegionType, TextItem,
|
||||||
|
};
|
||||||
|
|
||||||
use lopdf::Document;
|
use lopdf::Document;
|
||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
@@ -218,6 +222,34 @@ pub fn process_pdf_mem_with_options(
|
|||||||
process_document(doc, page_count, options, start)
|
process_document(doc, page_count, options, start)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Layout region detection
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
/// Detect layout regions from a PDF file.
|
||||||
|
///
|
||||||
|
/// Extracts text positions and runs heuristic region detection (tables,
|
||||||
|
/// formulas, images, paragraphs, headings) without generating markdown.
|
||||||
|
pub fn detect_layout<P: AsRef<Path>>(path: P) -> Result<Vec<PageLayout>, PdfError> {
|
||||||
|
validate_pdf_file(&path)?;
|
||||||
|
let (doc, _page_count) = load_document_from_path(&path)?;
|
||||||
|
detect_layout_from_doc(&doc)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect layout regions from a memory buffer.
|
||||||
|
pub fn detect_layout_mem(buffer: &[u8]) -> Result<Vec<PageLayout>, PdfError> {
|
||||||
|
validate_pdf_bytes(buffer)?;
|
||||||
|
let (doc, _page_count) = load_document_from_mem(buffer)?;
|
||||||
|
detect_layout_from_doc(&doc)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn detect_layout_from_doc(doc: &Document) -> Result<Vec<PageLayout>, PdfError> {
|
||||||
|
let font_cmaps = FontCMaps::from_doc(doc);
|
||||||
|
let ((items, rects, lines), _thresholds, _gid_pages) =
|
||||||
|
extractor::extract_positioned_text_from_doc(doc, &font_cmaps, None)?;
|
||||||
|
Ok(layout::detect_layout_regions(&items, &rects, &lines))
|
||||||
|
}
|
||||||
|
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
// Deprecated compat shims
|
// Deprecated compat shims
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
|
|||||||
@@ -59,6 +59,57 @@ pub enum ItemType {
|
|||||||
FormField,
|
FormField,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── Layout region types ─────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// Type of layout region detected on a page.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum RegionType {
|
||||||
|
/// A table (bordered or borderless).
|
||||||
|
Table,
|
||||||
|
/// A mathematical formula or equation.
|
||||||
|
Formula,
|
||||||
|
/// An embedded image.
|
||||||
|
Image,
|
||||||
|
/// A paragraph of body text.
|
||||||
|
Paragraph,
|
||||||
|
/// A heading with a 1-based level.
|
||||||
|
Heading { level: u8 },
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A bounding box in PDF coordinate space (origin at bottom-left).
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct BBox {
|
||||||
|
pub x_min: f32,
|
||||||
|
pub y_min: f32,
|
||||||
|
pub x_max: f32,
|
||||||
|
pub y_max: f32,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A detected layout region on a page.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct LayoutRegion {
|
||||||
|
/// Bounding box in PDF points (bottom-left origin).
|
||||||
|
pub bbox: BBox,
|
||||||
|
/// The type of content in this region.
|
||||||
|
pub region_type: RegionType,
|
||||||
|
/// 1-indexed page number.
|
||||||
|
pub page: u32,
|
||||||
|
/// Detection confidence (0.0–1.0).
|
||||||
|
pub confidence: f32,
|
||||||
|
/// Whether an orchestrator should send this region to OCR
|
||||||
|
/// rather than using native text extraction.
|
||||||
|
pub needs_ocr: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Per-page layout detection result.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct PageLayout {
|
||||||
|
/// 1-indexed page number.
|
||||||
|
pub page: u32,
|
||||||
|
/// Detected regions, sorted top-to-bottom then left-to-right.
|
||||||
|
pub regions: Vec<LayoutRegion>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Layout complexity analysis result.
|
/// Layout complexity analysis result.
|
||||||
///
|
///
|
||||||
/// Callers can use this to decide whether the extracted markdown is reliable
|
/// Callers can use this to decide whether the extracted markdown is reliable
|
||||||
|
|||||||
Reference in New Issue
Block a user