feat(tables): add rect-guided calendar table builder and extractor improvements

Rect-guided tables:
- Add try_build_rect_guided_table() to build tables from rect cluster
  X positions as column boundaries, bypassing heuristic detection
- Split merged multi-number TextItems (e.g. "10 11 12...31") into
  individual day-column cells using column-advancing boundary assignment
- Interpolate missing column boundaries for holiday/non-work days
  that lack colored rects
- Strip tilde-leader noise from cells (legend text bleeding)
- Filter legend text beyond table area via max-X threshold
- Pass cluster_rects through RectHintRegion for downstream use

Extractor improvements:
- Deduplicate clip-path and fill-path rects before using as table hints
- Add font width fallback for missing glyph metrics via average width
- Improve column detection scoring to prefer balanced gutters
- Handle side-by-side layouts with hint-region derived split points

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Abimael Martell
2026-03-11 15:09:17 -07:00
co-authored by Claude Opus 4.6
parent 6b8f4d0295
commit e3990dc065
7 changed files with 1505 additions and 37 deletions
+114 -7
View File
@@ -831,15 +831,122 @@ pub(crate) fn extract_page_text_items(
}
}
// Only use clipping-path rects when no `re` rects exist on this page,
// to avoid diluting real table rects with decorative clip regions.
// Fill-path rects are third priority: only when both `re` and clip rects are empty.
if rects.is_empty() && !clip_rects.is_empty() {
rects = clip_rects;
} else if rects.is_empty() && clip_rects.is_empty() && !fill_rects.is_empty() {
rects = fill_rects;
// Only use clip/fill rects when no `re` rects exist on this page.
// Clip rects take priority over fill rects, but first we deduplicate
// them: some PDFs wrap every text block in a full-page W* clip path,
// producing thousands of identical rects that yield a degenerate grid.
// After dedup, if too few unique clip rects remain we fall through to
// fill rects (explicitly drawn visible rectangles).
if rects.is_empty() {
dedup_rects(&mut clip_rects);
if clip_rects.len() >= 4 {
rects = clip_rects;
} else if !fill_rects.is_empty() {
rects = fill_rects;
} else if !clip_rects.is_empty() {
rects = clip_rects;
}
}
let items = super::merge_text_items(items);
Ok((items, rects, lines))
}
/// Remove near-duplicate rects (same coordinates within 0.5 pt tolerance).
/// Some PDFs emit a full-page clip path for every text block, producing
/// thousands of identical rects. After dedup these collapse to one rect,
/// which is too few for table detection and gets naturally skipped.
fn dedup_rects(rects: &mut Vec<PdfRect>) {
if rects.len() <= 1 {
return;
}
// Round to 0.5-pt grid for tolerance, then sort and dedup.
rects.sort_by(|a, b| {
let ak = (
a.page,
(a.x * 2.0) as i32,
(a.y * 2.0) as i32,
(a.width * 2.0) as i32,
(a.height * 2.0) as i32,
);
let bk = (
b.page,
(b.x * 2.0) as i32,
(b.y * 2.0) as i32,
(b.width * 2.0) as i32,
(b.height * 2.0) as i32,
);
ak.cmp(&bk)
});
rects.dedup_by(|a, b| {
a.page == b.page
&& ((a.x - b.x).abs() < 0.5)
&& ((a.y - b.y).abs() < 0.5)
&& ((a.width - b.width).abs() < 0.5)
&& ((a.height - b.height).abs() < 0.5)
});
}
#[cfg(test)]
mod tests {
use super::*;
fn rect(x: f32, y: f32, w: f32, h: f32, page: u32) -> PdfRect {
PdfRect {
x,
y,
width: w,
height: h,
page,
}
}
#[test]
fn test_dedup_rects_identical() {
let mut rects = vec![rect(0.0, 0.0, 612.0, 792.0, 1); 3759];
dedup_rects(&mut rects);
assert_eq!(rects.len(), 1);
}
#[test]
fn test_dedup_rects_within_tolerance() {
let mut rects = vec![
rect(10.0, 20.0, 100.0, 50.0, 1),
rect(10.2, 20.1, 100.3, 50.4, 1),
];
dedup_rects(&mut rects);
assert_eq!(rects.len(), 1);
}
#[test]
fn test_dedup_rects_distinct_kept() {
let mut rects = vec![
rect(10.0, 20.0, 100.0, 50.0, 1),
rect(120.0, 20.0, 100.0, 50.0, 1),
rect(10.0, 80.0, 100.0, 50.0, 1),
];
dedup_rects(&mut rects);
assert_eq!(rects.len(), 3);
}
#[test]
fn test_dedup_rects_different_pages_kept() {
let mut rects = vec![
rect(0.0, 0.0, 612.0, 792.0, 1),
rect(0.0, 0.0, 612.0, 792.0, 2),
];
dedup_rects(&mut rects);
assert_eq!(rects.len(), 2);
}
#[test]
fn test_dedup_rects_empty_and_single() {
let mut empty: Vec<PdfRect> = vec![];
dedup_rects(&mut empty);
assert!(empty.is_empty());
let mut single = vec![rect(1.0, 2.0, 3.0, 4.0, 1)];
dedup_rects(&mut single);
assert_eq!(single.len(), 1);
}
}
+17 -1
View File
@@ -778,7 +778,9 @@ pub(crate) fn extract_text_from_operand(
None
};
let mut has_cmap = false;
if let Some(entry) = inline_cmaps.get(current_font) {
has_cmap = true;
if let Some(decoded) = decode_with_entry(entry) {
return Some(decoded);
}
@@ -787,12 +789,18 @@ pub(crate) fn extract_text_from_operand(
// Look up CMap by ToUnicode object reference
if let Some(&obj_num) = font_tounicode_refs.get(current_font) {
if let Some(entry) = font_cmaps.get_by_obj(obj_num) {
has_cmap = true;
if let Some(decoded) = decode_with_entry(entry) {
return Some(decoded);
}
}
}
// CID fonts with a CMap that couldn't decode: the CID is genuinely
// unmapped. Don't fall through to text-interpretation fallbacks
// (Latin-1, UTF-16, etc.) which would misinterpret CID bytes as
// character codes (e.g. CID 0x01A9 → Latin-1 "©").
// Try our custom encoding map from Differences arrays.
// The Differences array overrides specific codes in a base encoding (typically
// WinAnsiEncoding). We must combine Differences entries with the base encoding
@@ -883,8 +891,16 @@ pub(crate) fn extract_text_from_operand(
if let Some(symbol_text) = decode_symbol_fallback(bytes, base_font_name) {
return Some(symbol_text);
}
// For CID fonts (have ToUnicode CMap), the CID is
// genuinely unmapped — return None to avoid Latin-1
// fallback misinterpreting CID bytes as characters.
if has_cmap || font_tounicode_refs.contains_key(current_font) {
return None;
}
// Non-CID fonts: fall through to other methods
} else {
return Some(text);
}
return Some(text);
}
}
+131 -8
View File
@@ -121,7 +121,8 @@ pub(crate) fn detect_columns(items: &[TextItem], page: u32) -> Vec<ColumnRegion>
let y_range = y_max - y_min;
// Validate each valley with vertical consistency
let mut valid_valleys: Vec<(usize, usize)> = Vec::new();
// Each entry: (start_bin, end_bin, left_count, right_count)
let mut valid_valleys: Vec<(usize, usize, usize, usize)> = Vec::new();
for &(start, end) in &valleys {
let gutter_left = x_min + start as f32 * BIN_WIDTH;
let gutter_right = x_min + end as f32 * BIN_WIDTH;
@@ -164,7 +165,7 @@ pub(crate) fn detect_columns(items: &[TextItem], page: u32) -> Vec<ColumnRegion>
}
}
valid_valleys.push((start, end));
valid_valleys.push((start, end, left_items.len(), right_items.len()));
}
if valid_valleys.is_empty() {
@@ -177,16 +178,21 @@ pub(crate) fn detect_columns(items: &[TextItem], page: u32) -> Vec<ColumnRegion>
valid_valleys.len() + 1,
valid_valleys
.iter()
.map(|(s, e)| x_min + ((*s + *e) as f32 / 2.0) * BIN_WIDTH)
.map(|(s, e, _, _)| x_min + ((*s + *e) as f32 / 2.0) * BIN_WIDTH)
.collect::<Vec<_>>()
);
// Limit to at most 3 gutters (4 columns) — keep the widest if more found
// Limit to at most 3 gutters (4 columns).
// Score = width_in_bins * min(left_count, right_count)
// This prefers gutters that separate substantial content on both sides,
// rather than just the physically widest gaps (which may be intra-column).
if valid_valleys.len() > 3 {
valid_valleys.sort_by(|a, b| {
let wa = (a.1 - a.0) as f32;
let wb = (b.1 - b.0) as f32;
wb.partial_cmp(&wa).unwrap_or(std::cmp::Ordering::Equal)
let score_a = (a.1 - a.0) as f32 * (a.2.min(a.3) as f32);
let score_b = (b.1 - b.0) as f32 * (b.2.min(b.3) as f32);
score_b
.partial_cmp(&score_a)
.unwrap_or(std::cmp::Ordering::Equal)
});
valid_valleys.truncate(3);
// Re-sort by position (left to right)
@@ -196,7 +202,7 @@ pub(crate) fn detect_columns(items: &[TextItem], page: u32) -> Vec<ColumnRegion>
// Build column regions from gutter boundaries
let mut columns = Vec::new();
let mut col_start = x_min;
for &(start, end) in &valid_valleys {
for &(start, end, _, _) in &valid_valleys {
let gutter_center = x_min + ((start + end) as f32 / 2.0) * BIN_WIDTH;
columns.push(ColumnRegion {
x_min: col_start,
@@ -694,3 +700,120 @@ fn group_single_column(items: Vec<TextItem>) -> Vec<TextLine> {
lines
}
#[cfg(test)]
mod tests {
use super::*;
use crate::types::ItemType;
/// Helper: create a TextItem at given position with given width text.
fn make_item(page: u32, x: f32, y: f32, text: &str) -> TextItem {
TextItem {
text: text.to_string(),
x,
y,
width: text.len() as f32 * 6.0, // ~6pt per char
height: 12.0,
font_size: 12.0,
font: String::new(),
page,
is_bold: false,
is_italic: false,
item_type: ItemType::Text,
}
}
/// Generate dense items in a horizontal zone across many Y positions.
/// Items are placed with overlapping coverage so no intra-zone valleys appear.
fn fill_zone(page: u32, x_start: f32, x_end: f32, y_start: f32, y_end: f32) -> Vec<TextItem> {
let mut items = Vec::new();
let item_width = 60.0; // "SomeText__" = 10 chars * 6pt
let step = 55.0; // overlap slightly to avoid intra-zone histogram gaps
let mut y = y_start;
while y >= y_end {
let mut x = x_start;
while x + item_width <= x_end {
items.push(make_item(page, x, y, "SomeText__"));
x += step;
}
y -= 14.0;
}
items
}
#[test]
fn three_zone_layout_detected() {
// Left months (x=15..330), right months (x=345..660), sidebar (x=675..800)
// Each zone is >100pt wide so min_col_width won't reject any.
let mut items = Vec::new();
items.extend(fill_zone(1, 15.0, 330.0, 750.0, 50.0));
items.extend(fill_zone(1, 345.0, 660.0, 750.0, 50.0));
items.extend(fill_zone(1, 675.0, 800.0, 750.0, 50.0));
let cols = detect_columns(&items, 1);
assert_eq!(cols.len(), 3, "Expected 3 columns, got {}", cols.len());
// Gutter 1 should be in the gap between left and middle zones
let g1 = cols[0].x_max;
assert!(
(290.0..=350.0).contains(&g1),
"First gutter at {g1}, expected between left and middle zones"
);
// Gutter 2 should be in the gap between middle and right zones
let g2 = cols[1].x_max;
assert!(
(620.0..=680.0).contains(&g2),
"Second gutter at {g2}, expected between middle and right zones"
);
}
#[test]
fn two_column_regression_guard() {
// Standard 2-column layout with clear gutter at center
let mut items = Vec::new();
items.extend(fill_zone(1, 30.0, 280.0, 750.0, 50.0));
items.extend(fill_zone(1, 320.0, 570.0, 750.0, 50.0));
let cols = detect_columns(&items, 1);
assert_eq!(cols.len(), 2, "Expected 2 columns, got {}", cols.len());
let gutter = cols[0].x_max;
assert!(
(280.0..=320.0).contains(&gutter),
"Gutter at {gutter}, expected ~300"
);
}
#[test]
fn score_prefers_balanced_gutter_over_wide_gap() {
// 5 valid valleys: 2 are wide but split sparse content, 2 are narrower
// but separate dense zones. The dense-zone gutters should win.
let mut items = Vec::new();
// Dense left zone
items.extend(fill_zone(1, 15.0, 200.0, 750.0, 50.0));
// Dense middle zone
items.extend(fill_zone(1, 220.0, 400.0, 750.0, 50.0));
// Dense right zone
items.extend(fill_zone(1, 420.0, 600.0, 750.0, 50.0));
// Sparse far-right zone (few items)
for y_off in 0..12 {
items.push(make_item(
1,
700.0,
750.0 - y_off as f32 * 50.0,
"Sparse____",
));
}
let cols = detect_columns(&items, 1);
// Should detect the gutters between the 3 dense zones, not the wide gap
// before the sparse zone
assert!(
cols.len() >= 3,
"Expected >=3 columns for dense zones, got {}",
cols.len()
);
}
}
+13 -1
View File
@@ -181,6 +181,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
options: MarkdownOptions,
page_tables: std::collections::HashMap<u32, Vec<(f32, String)>>,
page_images: std::collections::HashMap<u32, Vec<(f32, String)>>,
band_split_pages: &HashSet<u32>,
) -> String {
if lines.is_empty() && page_tables.is_empty() && page_images.is_empty() {
return String::new();
@@ -210,6 +211,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
let mut output = String::new();
let mut current_page = 0u32;
let mut prev_y = f32::MAX;
let mut prev_x = 0.0f32;
let mut in_list = false;
let mut in_paragraph = false;
let mut last_list_x: Option<f32> = None;
@@ -274,6 +276,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
current_page = line.page;
prev_y = f32::MAX;
prev_x = 0.0;
if options.include_page_numbers {
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
@@ -317,14 +320,23 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// Paragraph break: large forward Y gap (normal) or large backward jump
// (newspaper columns emitted sequentially on the same page).
let y_gap = prev_y - line.y;
let line_x = line.items.first().map(|i| i.x).unwrap_or(0.0);
let is_para_break = y_gap.abs() > para_threshold;
if is_para_break && in_paragraph {
// Also break when X jumps significantly at the same Y level on
// pages with band-split side-by-side layout. This prevents
// interleaved left/right band lines from merging into one paragraph.
let is_band_switch = band_split_pages.contains(&line.page)
&& y_gap.abs() <= para_threshold
&& (prev_x - line_x).abs() > 50.0
&& prev_y < f32::MAX;
if (is_para_break || is_band_switch) && in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
}
// Don't immediately end list on paragraph break
// Let the continuation check below decide if we're still in a list
prev_y = line.y;
prev_x = line_x;
// Get text with optional bold/italic formatting
let text = line.text_with_formatting(options.detect_bold, options.detect_italic);
+311 -4
View File
@@ -127,6 +127,153 @@ pub(crate) fn split_side_by_side(items: &[TextItem]) -> Vec<(f32, f32)> {
vec![(x_min, best_split), (best_split, x_max)]
}
/// Derive a side-by-side split from rect hint regions.
///
/// When `split_side_by_side` doesn't detect a gap (e.g. the text gap is too
/// small), hint regions from large rect clusters can still reveal a left/right
/// zone layout (calendar months, form sections). This function checks if hint
/// regions pair up at the same Y bands and returns `[(x_min, split), (split,
/// x_max)]` if a consistent split exists.
fn split_from_hint_regions(items: &[TextItem], rects: &[PdfRect], page: u32) -> Vec<(f32, f32)> {
use crate::tables::{cluster_rects, RectHintRegion};
// Quick hint region computation (same logic as detect_tables_from_rects
// but without table detection).
let mut page_rects: Vec<(f32, f32, f32, f32)> = Vec::new();
for r in rects {
if r.page != page {
continue;
}
let (mut x, mut y, mut w, mut h) = (r.x, r.y, r.width, r.height);
if w < 0.0 {
x += w;
w = -w;
}
if h < 0.0 {
y += h;
h = -h;
}
if w < 5.0 || h < 5.0 {
continue;
}
page_rects.push((x, y, w, h));
}
if page_rects.len() < 60 {
return vec![];
}
// Width outlier filter (same as detect_tables_from_rects)
let mut widths: Vec<f32> = page_rects.iter().map(|&(_, _, w, _)| w).collect();
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
let median_width = widths[widths.len() / 2];
page_rects.retain(|&(_, _, w, _)| w <= median_width * 10.0);
let clusters = cluster_rects(&page_rects, 3.0, 6);
if clusters.len() < 4 {
return vec![];
}
// Build hint regions from large clusters
let mut hints: Vec<RectHintRegion> = Vec::new();
for cluster_indices in &clusters {
let group_rects: Vec<(f32, f32, f32, f32)> =
cluster_indices.iter().map(|&i| page_rects[i]).collect();
if group_rects.len() < 30 {
continue;
}
let x_left = group_rects.iter().map(|r| r.0).reduce(f32::min).unwrap();
let x_right = group_rects
.iter()
.map(|r| r.0 + r.2)
.reduce(f32::max)
.unwrap();
let y_bottom = group_rects.iter().map(|r| r.1).reduce(f32::min).unwrap();
let y_top = group_rects
.iter()
.map(|r| r.1 + r.3)
.reduce(f32::max)
.unwrap();
let w = x_right - x_left;
let h = y_top - y_bottom;
if (30.0..=400.0).contains(&w) && (10.0..=400.0).contains(&h) {
hints.push(RectHintRegion {
y_top,
y_bottom,
x_left,
x_right,
cluster_rects: Vec::new(),
});
}
}
if hints.len() < 4 {
return vec![];
}
// Check for left/right pairing: hints at the same Y band should split
// into distinct X groups. Count pairs where two hints share a Y band
// (>50% overlap) but occupy different X halves.
let page_x_mid = {
let x_min = items.iter().map(|i| i.x).reduce(f32::min).unwrap_or(0.0);
let x_max = items
.iter()
.map(|i| i.x + i.width)
.reduce(f32::max)
.unwrap_or(800.0);
(x_min + x_max) / 2.0
};
let mut pair_count = 0;
for (i, a) in hints.iter().enumerate() {
for b in hints.iter().skip(i + 1) {
let y_overlap = a.y_top.min(b.y_top) - a.y_bottom.max(b.y_bottom);
let y_min_span = (a.y_top - a.y_bottom).min(b.y_top - b.y_bottom);
if y_overlap > y_min_span * 0.5 {
let a_center = (a.x_left + a.x_right) / 2.0;
let b_center = (b.x_left + b.x_right) / 2.0;
if (a_center < page_x_mid) != (b_center < page_x_mid) {
pair_count += 1;
}
}
}
}
// Require at least 3 left/right pairs to confirm the layout
if pair_count < 3 {
return vec![];
}
// Find the split X: midpoint between the rightmost left-zone hint
// and the leftmost right-zone hint
let max_left_x = hints
.iter()
.filter(|h| (h.x_left + h.x_right) / 2.0 < page_x_mid)
.map(|h| h.x_right)
.reduce(f32::max);
let min_right_x = hints
.iter()
.filter(|h| (h.x_left + h.x_right) / 2.0 >= page_x_mid)
.map(|h| h.x_left)
.reduce(f32::min);
if let (Some(left_edge), Some(right_edge)) = (max_left_x, min_right_x) {
let split_x = (left_edge + right_edge) / 2.0;
let x_min = items.iter().map(|i| i.x).reduce(f32::min).unwrap_or(0.0);
let x_max = items
.iter()
.map(|i| i.x + i.width)
.reduce(f32::max)
.unwrap_or(800.0);
log::debug!(
"page {}: hint-derived side-by-side split at x={:.1}",
page,
split_x
);
vec![(x_min, split_x), (split_x, x_max)]
} else {
vec![]
}
}
/// Filter rects to those mostly contained within an X band.
///
/// Excludes rects that extend significantly beyond the band (e.g. page-wide
@@ -320,6 +467,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
) -> String {
use crate::tables::{
detect_tables, detect_tables_from_lines, detect_tables_from_rects, table_to_markdown,
try_build_rect_guided_table,
};
use crate::types::ItemType;
@@ -390,12 +538,27 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
pages.sort();
let page_count = pages.last().copied().unwrap_or(0) + 1;
// Track band splits per page so we can split non-table items later
let mut page_band_splits: HashMap<u32, Vec<(f32, f32)>> = HashMap::new();
for page in pages {
let group = page_groups.get(&page).unwrap();
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
// Check for side-by-side layout (e.g. two tables placed left and right)
let bands = split_side_by_side(&page_items);
let mut bands = split_side_by_side(&page_items);
// Fallback: use rect hint regions to detect side-by-side layout
// when the text gap is too narrow for split_side_by_side to detect
// (e.g. calendars with left/right month columns ~10pt apart).
if bands.is_empty() {
bands = split_from_hint_regions(&page_items, rects, page);
// Only track hint-derived splits for non-table line grouping.
// split_side_by_side splits already scope table detection and
// their non-table items should flow through normal line grouping.
if !bands.is_empty() {
page_band_splits.insert(page, bands.clone());
}
}
// Build list of (band_items, band_index_map, band_rects, band_lines).
// band_index_map[local_band_idx] → page_items index.
@@ -479,7 +642,52 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
}
}
// 3. Heuristic fallback on unclaimed items
// 3a. Try rect-guided table construction on hint regions before
// creating the heuristic closure (avoids borrow conflicts).
if rect_claimed.is_empty() && !hint_regions.is_empty() {
let padding = 15.0;
for hint in &hint_regions {
if hint.cluster_rects.is_empty() {
continue;
}
let (inside_items, inside_map): (Vec<TextItem>, Vec<usize>) = band_items
.iter()
.enumerate()
.filter(|(_, item)| {
item.y >= hint.y_bottom - padding
&& item.y <= hint.y_top + padding
&& item.x >= hint.x_left - padding
&& item.x <= hint.x_right + padding
})
.map(|(idx, item)| (item.clone(), idx))
.unzip();
if let Some(table) =
try_build_rect_guided_table(&inside_items, &hint.cluster_rects)
{
for &idx in &table.item_indices {
if let Some(&band_idx) = inside_map.get(idx) {
if let Some(&page_idx) = band_index_map.get(band_idx) {
if let Some(&(global_idx, _)) = group.get(page_idx) {
table_items.insert(global_idx);
}
}
}
}
let table_y = table.rows.first().copied().unwrap_or(0.0);
let table_md = table_to_markdown(&table);
page_tables
.entry(page)
.or_default()
.push((table_y, table_md));
for &band_idx in &inside_map {
rect_claimed.insert(band_idx);
}
}
}
}
// 3b. Heuristic fallback on unclaimed items
let mut run_heuristic =
|subset_items: &[TextItem], index_map: &[usize], min_items: usize| {
if subset_items.len() < min_items {
@@ -569,7 +777,50 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
// Merge continuation tables across page breaks, but only for table-only pages
merge_continuation_tables(&mut page_tables, &table_only_pages);
let lines = group_into_lines(non_table_items);
// Split non-table items by band boundaries before line grouping so that
// items from different side-by-side zones (e.g. left/right month columns
// in a calendar) don't merge into the same line.
let lines = if page_band_splits.is_empty() {
group_into_lines(non_table_items)
} else {
// Separate items into band-split pages and non-split pages
let mut split_page_items: HashMap<u32, Vec<TextItem>> = HashMap::new();
let mut unsplit_items: Vec<TextItem> = Vec::new();
for item in non_table_items {
if page_band_splits.contains_key(&item.page) {
split_page_items.entry(item.page).or_default().push(item);
} else {
unsplit_items.push(item);
}
}
// Process unsplit pages normally
let mut all_lines = group_into_lines(unsplit_items);
// Process each split page's bands independently, then interleave
// by Y position so paired zones (e.g. left/right months) appear together.
let mut split_pages: Vec<u32> = split_page_items.keys().copied().collect();
split_pages.sort();
for page in split_pages {
let items = split_page_items.remove(&page).unwrap();
let bands = &page_band_splits[&page];
let mut page_lines: Vec<crate::types::TextLine> = Vec::new();
for &(x_lo, x_hi) in bands {
let margin = 2.0;
let band_items: Vec<TextItem> = items
.iter()
.filter(|i| i.x >= x_lo - margin && i.x < x_hi + margin)
.cloned()
.collect();
if !band_items.is_empty() {
page_lines.extend(group_into_lines(band_items));
}
}
// Sort by Y descending (top to bottom) so left and right
// band lines interleave in visual reading order.
page_lines.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
all_lines.extend(page_lines);
}
all_lines
};
// Strip repeated headers/footers before conversion
let lines = if options.strip_headers_footers {
@@ -579,7 +830,14 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
};
// Convert to markdown, inserting tables and images at appropriate positions
to_markdown_from_lines_with_tables_and_images(lines, options, page_tables, page_images)
let band_split_page_set: HashSet<u32> = page_band_splits.keys().copied().collect();
to_markdown_from_lines_with_tables_and_images(
lines,
options,
page_tables,
page_images,
&band_split_page_set,
)
}
#[cfg(test)]
@@ -650,4 +908,53 @@ mod tests {
assert!(md.contains("- First item"));
assert!(md.contains("- Second item"));
}
fn make_item(x: f32, y: f32, page: u32) -> TextItem {
TextItem {
text: "A".into(),
x,
y,
width: 5.0,
height: 10.0,
font: String::new(),
font_size: 10.0,
page,
is_bold: false,
is_italic: false,
item_type: crate::types::ItemType::Text,
}
}
#[test]
fn split_from_hint_regions_too_few_rects() {
// Fewer than 60 rects → no split
let items = vec![make_item(10.0, 100.0, 1)];
let rects: Vec<PdfRect> = (0..30)
.map(|i| PdfRect {
x: 10.0 + (i % 7) as f32 * 15.0,
y: 100.0 + (i / 7) as f32 * 15.0,
width: 10.0,
height: 10.0,
page: 1,
})
.collect();
assert!(split_from_hint_regions(&items, &rects, 1).is_empty());
}
#[test]
fn split_from_hint_regions_no_pairs() {
// Enough rects but all in one X zone → no left/right pairs → no split
let items = vec![make_item(10.0, 100.0, 1)];
// 80 rects all in left half
let rects: Vec<PdfRect> = (0..80)
.map(|i| PdfRect {
x: 10.0 + (i % 10) as f32 * 15.0,
y: 100.0 + (i / 10) as f32 * 15.0,
width: 10.0,
height: 10.0,
page: 1,
})
.collect();
assert!(split_from_hint_regions(&items, &rects, 1).is_empty());
}
}
+515 -16
View File
@@ -97,6 +97,71 @@ pub(crate) fn cluster_rects(
result.into_iter().map(|(_, g)| g).collect()
}
/// Split a rect cluster at the widest X-gap when detection fails.
/// Returns sub-groups only if a gap >= `min_gap` exists and both sides have >= `min_group_size` rects.
#[allow(clippy::type_complexity)]
fn split_wide_cluster(
rects: &[(f32, f32, f32, f32)],
min_gap: f32,
min_group_size: usize,
) -> Option<(Vec<(f32, f32, f32, f32)>, Vec<(f32, f32, f32, f32)>)> {
if rects.len() < min_group_size * 2 {
return None;
}
// Build sorted list of X-intervals (x_left, x_right) from each rect
let mut intervals: Vec<(f32, f32)> = rects.iter().map(|&(x, _, w, _)| (x, x + w)).collect();
intervals.sort_by(|a, b| a.0.partial_cmp(&b.0).unwrap_or(std::cmp::Ordering::Equal));
// Merge overlapping intervals to find contiguous X-bands
let mut merged: Vec<(f32, f32)> = Vec::new();
for (start, end) in &intervals {
if let Some(last) = merged.last_mut() {
if *start <= last.1 + 1.0 {
last.1 = last.1.max(*end);
continue;
}
}
merged.push((*start, *end));
}
if merged.len() < 2 {
return None;
}
// Find the widest gap between consecutive merged intervals
let mut best_gap = 0.0_f32;
let mut best_split_x = 0.0_f32;
for i in 1..merged.len() {
let gap = merged[i].0 - merged[i - 1].1;
if gap > best_gap {
best_gap = gap;
best_split_x = (merged[i - 1].1 + merged[i].0) / 2.0;
}
}
if best_gap < min_gap {
return None;
}
let left: Vec<_> = rects
.iter()
.filter(|&&(x, _, w, _)| x + w / 2.0 < best_split_x)
.copied()
.collect();
let right: Vec<_> = rects
.iter()
.filter(|&&(x, _, w, _)| x + w / 2.0 >= best_split_x)
.copied()
.collect();
if left.len() >= min_group_size && right.len() >= min_group_size {
Some((left, right))
} else {
None
}
}
/// A bounding box hint from cell-border rects that failed full grid validation.
///
/// When a rect cluster contains cell-sized borders but they don't form a valid
@@ -110,6 +175,12 @@ pub struct RectHintRegion {
pub y_top: f32,
/// Y coordinate of the bottom edge (lowest value in PDF space)
pub y_bottom: f32,
/// X coordinate of the left edge
pub x_left: f32,
/// X coordinate of the right edge
pub x_right: f32,
/// Raw rects from the cluster (x, y, w, h) for rect-guided table building
pub cluster_rects: Vec<(f32, f32, f32, f32)>,
}
/// Detect tables from explicit rectangle (`re`) operators in the PDF.
@@ -227,6 +298,22 @@ pub fn detect_tables_from_rects(
tables.push(table);
} else if let Some(table) = detect_row_stripe_table(items, &group_rects, page) {
tables.push(table);
} else if let Some((left, right)) = split_wide_cluster(&group_rects, 15.0, 6) {
// Cluster was too wide — retry each half independently
debug!(
"page {}: splitting cluster of {} rects into {} + {} at x-gap",
page,
group_rects.len(),
left.len(),
right.len()
);
for sub in [&left, &right] {
if let Some(table) = detect_table_from_rect_group(items, sub, page) {
tables.push(table);
} else if let Some(table) = detect_row_stripe_table(items, sub, page) {
tables.push(table);
}
}
}
}
@@ -287,22 +374,80 @@ pub fn detect_tables_from_rects(
}
}
// On rect-sparse pages (≤ 6 rects), a few cell-border rects may define the
// table region even though they can't form a full grid (e.g. only horizontal
// row borders, no column dividers). Extract a hint region so the heuristic
// detector can be scoped to just that area, preventing nearby graph labels
// or other content from being merged into the table.
if tables.is_empty() && page_rects.len() >= 4 && page_rects.len() <= 6 {
let clusters = cluster_rects(&page_rects, 3.0, 4);
for cluster_indices in &clusters {
let group_rects: Vec<(f32, f32, f32, f32)> =
cluster_indices.iter().map(|&i| page_rects[i]).collect();
if let Some(hint) = extract_hint_region(&group_rects) {
if tables.is_empty() {
// When no tables detected but clusters exist, generate XY hint regions
// from cluster bounding boxes to scope heuristic table detection.
// This handles both large decorative-rect clusters (calendars, forms)
// and small cell-border clusters on rect-sparse pages.
if page_rects.len() >= 6 {
let clusters = cluster_rects(&page_rects, 3.0, 6);
for cluster_indices in &clusters {
let group_rects: Vec<(f32, f32, f32, f32)> =
cluster_indices.iter().map(|&i| page_rects[i]).collect();
if group_rects.len() < 30 {
continue;
}
let x_left = group_rects.iter().map(|r| r.0).reduce(f32::min).unwrap();
let x_right = group_rects
.iter()
.map(|r| r.0 + r.2)
.reduce(f32::max)
.unwrap();
let y_bottom = group_rects.iter().map(|r| r.1).reduce(f32::min).unwrap();
let y_top = group_rects
.iter()
.map(|r| r.1 + r.3)
.reduce(f32::max)
.unwrap();
let w = x_right - x_left;
let h = y_top - y_bottom;
if (30.0..=400.0).contains(&w) && (10.0..=400.0).contains(&h) {
debug!(
"page {}: hint candidate from {} rects: x={:.1}..{:.1} y={:.1}..{:.1} ({:.0}×{:.0})",
page, group_rects.len(), x_left, x_right, y_bottom, y_top, w, h
);
hint_regions.push(RectHintRegion {
y_top,
y_bottom,
x_left,
x_right,
cluster_rects: group_rects.clone(),
});
}
}
// Deduplicate overlapping hints
hint_regions = merge_overlapping_hints(hint_regions);
// Require multiple hint regions to confirm a multi-zone layout
// (calendars, forms). A single hint is likely a decorative cluster
// that would interfere with full-page heuristic detection.
if hint_regions.len() < 2 {
hint_regions.clear();
}
if !hint_regions.is_empty() {
debug!(
"page {}: hint region y={:.1}..{:.1}",
page, hint.y_bottom, hint.y_top
"page {}: {} XY hint regions from failed clusters",
page,
hint_regions.len()
);
hint_regions.push(hint);
}
}
// On rect-sparse pages (≤ 6 rects), a few cell-border rects may define the
// table region even though they can't form a full grid (e.g. only horizontal
// row borders, no column dividers). Extract a hint region so the heuristic
// detector can be scoped to just that area.
if hint_regions.is_empty() && page_rects.len() >= 4 && page_rects.len() <= 6 {
let small_clusters = cluster_rects(&page_rects, 3.0, 4);
for cluster_indices in &small_clusters {
let group_rects: Vec<(f32, f32, f32, f32)> =
cluster_indices.iter().map(|&i| page_rects[i]).collect();
if let Some(hint) = extract_hint_region(&group_rects) {
debug!(
"page {}: hint region y={:.1}..{:.1} x={:.1}..{:.1}",
page, hint.y_bottom, hint.y_top, hint.x_left, hint.x_right
);
hint_regions.push(hint);
}
}
}
}
@@ -310,6 +455,64 @@ pub fn detect_tables_from_rects(
(tables, hint_regions)
}
/// Merge nearby hint regions that share a Y band.
///
/// Two hints merge when they have substantial Y overlap (>50%) AND their X ranges
/// overlap or are close (gap < 50pt). This handles calendar-style layouts where a
/// month zone's decorative rects split into 2-3 adjacent clusters with small X gaps.
/// Runs iteratively until no more merges occur.
fn merge_overlapping_hints(mut hints: Vec<RectHintRegion>) -> Vec<RectHintRegion> {
if hints.len() <= 1 {
return hints;
}
loop {
hints.sort_by(|a, b| a.x_left.partial_cmp(&b.x_left).unwrap());
let mut merged: Vec<RectHintRegion> = Vec::new();
let mut any_merged = false;
for hint in &hints {
let mut did_merge = false;
for existing in merged.iter_mut() {
// Check Y overlap (>50% of smaller span)
let y_overlap =
existing.y_top.min(hint.y_top) - existing.y_bottom.max(hint.y_bottom);
let y_min_span =
(existing.y_top - existing.y_bottom).min(hint.y_top - hint.y_bottom);
if y_overlap <= y_min_span * 0.5 {
continue;
}
// Check X: overlapping or adjacent (gap < 50pt)
let x_gap = existing.x_left.max(hint.x_left) - existing.x_right.min(hint.x_right);
if x_gap < 50.0 {
// Don't merge if result would exceed max hint width (400pt)
let merged_left = existing.x_left.min(hint.x_left);
let merged_right = existing.x_right.max(hint.x_right);
if merged_right - merged_left > 400.0 {
continue;
}
existing.x_left = merged_left;
existing.x_right = merged_right;
existing.y_bottom = existing.y_bottom.min(hint.y_bottom);
existing.y_top = existing.y_top.max(hint.y_top);
existing
.cluster_rects
.extend_from_slice(&hint.cluster_rects);
did_merge = true;
any_merged = true;
break;
}
}
if !did_merge {
merged.push(hint.clone());
}
}
hints = merged;
if !any_merged {
break;
}
}
hints
}
/// Extract a hint region from a rect cluster that failed grid validation.
///
/// Only produces hints from small clusters (≤ 8 rects) where a few cell-border
@@ -340,12 +543,17 @@ fn extract_hint_region(group_rects: &[(f32, f32, f32, f32)]) -> Option<RectHintR
return None;
}
// Compute Y bounding box of cell-sized rects
// Compute bounding box of cell-sized rects
let y_bottom = cell_rects.iter().map(|(_, y, _, _)| *y).reduce(f32::min)?;
let y_top = cell_rects
.iter()
.map(|(_, y, _, h)| *y + *h)
.reduce(f32::max)?;
let x_left = cell_rects.iter().map(|(x, _, _, _)| *x).reduce(f32::min)?;
let x_right = cell_rects
.iter()
.map(|(x, _, w, _)| *x + *w)
.reduce(f32::max)?;
// The region must have meaningful height but not span an unreasonable area
let region_height = y_top - y_bottom;
@@ -353,7 +561,13 @@ fn extract_hint_region(group_rects: &[(f32, f32, f32, f32)]) -> Option<RectHintR
return None;
}
Some(RectHintRegion { y_top, y_bottom })
Some(RectHintRegion {
y_top,
y_bottom,
x_left,
x_right,
cluster_rects: Vec::new(),
})
}
/// Detect a single table from a cluster of spatially connected rects.
@@ -1794,4 +2008,289 @@ mod tests {
.collect();
assert!(extract_hint_region(&rects).is_none());
}
// --- split_wide_cluster ---
#[test]
fn split_at_wide_gap() {
// Left zone: x=10..50, Right zone: x=80..120 → gap of 30pt
let mut rects = Vec::new();
for i in 0..8 {
rects.push((10.0, i as f32 * 20.0, 40.0, 15.0)); // left
rects.push((80.0, i as f32 * 20.0, 40.0, 15.0)); // right
}
let result = split_wide_cluster(&rects, 15.0, 6);
assert!(result.is_some());
let (left, right) = result.unwrap();
assert!(left.iter().all(|&(x, _, _, _)| x < 60.0));
assert!(right.iter().all(|&(x, _, _, _)| x >= 60.0));
}
#[test]
fn no_split_narrow_gap() {
// Left zone: x=10..50, Right zone: x=55..95 → gap of only 5pt
let mut rects = Vec::new();
for i in 0..8 {
rects.push((10.0, i as f32 * 20.0, 40.0, 15.0));
rects.push((55.0, i as f32 * 20.0, 40.0, 15.0));
}
assert!(split_wide_cluster(&rects, 15.0, 6).is_none());
}
#[test]
fn no_split_small_subgroup() {
// Left zone: 2 rects, Right zone: 8 rects → left too small (< 6)
let mut rects = Vec::new();
for i in 0..2 {
rects.push((10.0, i as f32 * 20.0, 40.0, 15.0));
}
for i in 0..8 {
rects.push((80.0, i as f32 * 20.0, 40.0, 15.0));
}
// Also fails min total: 10 < 12 (min_group_size * 2 = 12)
assert!(split_wide_cluster(&rects, 15.0, 6).is_none());
}
#[test]
fn split_preserves_all_rects() {
let mut rects = Vec::new();
for i in 0..10 {
rects.push((10.0, i as f32 * 20.0, 40.0, 15.0));
rects.push((80.0, i as f32 * 20.0, 40.0, 15.0));
}
let (left, right) = split_wide_cluster(&rects, 15.0, 6).unwrap();
assert_eq!(left.len() + right.len(), rects.len());
}
#[test]
fn no_split_single_band() {
// All rects overlap in X → single merged interval, no gap
let rects: Vec<(f32, f32, f32, f32)> = (0..12)
.map(|i| (10.0 + i as f32 * 5.0, i as f32 * 20.0, 40.0, 15.0))
.collect();
assert!(split_wide_cluster(&rects, 15.0, 6).is_none());
}
// --- XY hint regions from failed clusters ---
#[test]
fn hint_from_failed_large_clusters() {
// Two separate clusters of 36 rects (6×6) each, placed side by side
// with a large gap so they form two distinct clusters.
// Requires ≥2 qualifying clusters to produce hints (multi-zone layout).
let mut page_rects: Vec<(f32, f32, f32, f32)> = Vec::new();
// Cluster 1: x=50..120, y=100..170
for row in 0..6 {
for col in 0..6 {
page_rects.push((
50.0 + col as f32 * 12.0,
100.0 + row as f32 * 12.0,
10.0,
10.0,
));
}
}
// Cluster 2: x=250..320, y=100..170 (130pt gap from cluster 1)
for row in 0..6 {
for col in 0..6 {
page_rects.push((
250.0 + col as f32 * 12.0,
100.0 + row as f32 * 12.0,
10.0,
10.0,
));
}
}
let items: Vec<TextItem> = vec![];
let rects: Vec<crate::types::PdfRect> = page_rects
.iter()
.map(|&(x, y, w, h)| crate::types::PdfRect {
x,
y,
width: w,
height: h,
page: 1,
})
.collect();
let (tables, hints) = detect_tables_from_rects(&items, &rects, 1);
assert!(tables.is_empty());
assert_eq!(hints.len(), 2);
// Cluster 1: x=50..120, y=100..170
assert!((hints[0].x_left - 50.0).abs() < 1.0);
assert!((hints[0].x_right - 120.0).abs() < 1.0);
assert!((hints[0].y_bottom - 100.0).abs() < 1.0);
assert!((hints[0].y_top - 170.0).abs() < 1.0);
// Cluster 2: x=250..320, y=100..170
assert!((hints[1].x_left - 250.0).abs() < 1.0);
assert!((hints[1].x_right - 320.0).abs() < 1.0);
}
#[test]
fn no_hint_single_large_cluster() {
// Single cluster of 36 rects — not enough (need ≥2 zones)
let mut page_rects: Vec<(f32, f32, f32, f32)> = Vec::new();
for row in 0..6 {
for col in 0..6 {
page_rects.push((
50.0 + col as f32 * 12.0,
100.0 + row as f32 * 12.0,
10.0,
10.0,
));
}
}
let items: Vec<TextItem> = vec![];
let rects: Vec<crate::types::PdfRect> = page_rects
.iter()
.map(|&(x, y, w, h)| crate::types::PdfRect {
x,
y,
width: w,
height: h,
page: 1,
})
.collect();
let (tables, hints) = detect_tables_from_rects(&items, &rects, 1);
assert!(tables.is_empty());
assert!(hints.is_empty());
}
#[test]
fn no_hint_too_few_rects() {
// 5 rects (< 10 threshold for large-cluster hints, also < 6 for clustering)
let rects: Vec<crate::types::PdfRect> = (0..5)
.map(|i| crate::types::PdfRect {
x: 50.0 + i as f32 * 30.0,
y: 100.0,
width: 20.0,
height: 20.0,
page: 1,
})
.collect();
let (tables, hints) = detect_tables_from_rects(&[], &rects, 1);
assert!(tables.is_empty());
// 5 rects: not enough for ≥6 clustering, and rect-sparse path needs 4-6
// but clusters of ≥4 won't form with disconnected rects (30pt gap > 3pt tol)
assert!(hints.is_empty());
}
#[test]
fn no_hint_page_spanning_width() {
// Rects spanning > 400pt width → no hint
let mut page_rects = Vec::new();
for i in 0..12 {
page_rects.push(crate::types::PdfRect {
x: i as f32 * 40.0,
y: 100.0,
width: 38.0,
height: 10.0,
page: 1,
});
}
let (tables, hints) = detect_tables_from_rects(&[], &page_rects, 1);
assert!(tables.is_empty());
assert!(hints.is_empty());
}
// --- merge_overlapping_hints ---
#[test]
fn merge_overlapping_hints_dedup() {
let hints = vec![
RectHintRegion {
x_left: 50.0,
x_right: 250.0,
y_bottom: 100.0,
y_top: 200.0,
cluster_rects: Vec::new(),
},
RectHintRegion {
x_left: 60.0,
x_right: 260.0,
y_bottom: 110.0,
y_top: 210.0,
cluster_rects: Vec::new(),
},
];
let merged = merge_overlapping_hints(hints);
assert_eq!(merged.len(), 1);
assert!((merged[0].x_left - 50.0).abs() < 0.01);
assert!((merged[0].x_right - 260.0).abs() < 0.01);
assert!((merged[0].y_bottom - 100.0).abs() < 0.01);
assert!((merged[0].y_top - 210.0).abs() < 0.01);
}
#[test]
fn merge_overlapping_hints_disjoint() {
let hints = vec![
RectHintRegion {
x_left: 50.0,
x_right: 200.0,
y_bottom: 100.0,
y_top: 200.0,
cluster_rects: Vec::new(),
},
RectHintRegion {
x_left: 350.0,
x_right: 500.0,
y_bottom: 100.0,
y_top: 200.0,
cluster_rects: Vec::new(),
},
];
let merged = merge_overlapping_hints(hints);
assert_eq!(merged.len(), 2);
}
#[test]
fn merge_hints_blocked_by_max_width() {
// Two hints in the same Y band with small X gap (8pt) but combined
// width > 400pt. Simulates left/right calendar month zones that
// should NOT merge.
let hints = vec![
RectHintRegion {
x_left: 20.0,
x_right: 340.0,
y_bottom: 100.0,
y_top: 170.0,
cluster_rects: Vec::new(),
},
RectHintRegion {
x_left: 348.0,
x_right: 668.0,
y_bottom: 100.0,
y_top: 170.0,
cluster_rects: Vec::new(),
},
];
let merged = merge_overlapping_hints(hints);
// Should remain separate: merged width would be 648pt > 400pt
assert_eq!(merged.len(), 2);
}
#[test]
fn merge_hints_adjacent_fragments() {
// Two fragments of the same zone with small gap, combined width < 400pt.
// Should merge.
let hints = vec![
RectHintRegion {
x_left: 20.0,
x_right: 266.0,
y_bottom: 100.0,
y_top: 170.0,
cluster_rects: Vec::new(),
},
RectHintRegion {
x_left: 276.0,
x_right: 340.0,
y_bottom: 100.0,
y_top: 170.0,
cluster_rects: Vec::new(),
},
];
let merged = merge_overlapping_hints(hints);
assert_eq!(merged.len(), 1);
assert!((merged[0].x_left - 20.0).abs() < 0.01);
assert!((merged[0].x_right - 340.0).abs() < 0.01);
}
}
+404
View File
@@ -11,9 +11,251 @@ mod grid;
pub use detect_heuristic::detect_tables;
pub use detect_lines::detect_tables_from_lines;
pub(crate) use detect_rects::cluster_rects;
pub use detect_rects::{detect_tables_from_rects, RectHintRegion};
pub use format::table_to_markdown;
use crate::types::TextItem;
/// Try to build a table from items + cluster rects (calendar-style layouts).
///
/// Uses rect X positions as column boundaries to directly construct a `Table`,
/// bypassing heuristic detection. Splits merged multi-number items first.
pub(crate) fn try_build_rect_guided_table(
items: &[TextItem],
cluster_rects: &[(f32, f32, f32, f32)],
) -> Option<Table> {
if items.is_empty() || cluster_rects.is_empty() {
return None;
}
// 1. Derive column boundaries from rect X positions (snapped to 2pt tolerance)
let mut x_lefts: Vec<f32> = cluster_rects.iter().map(|&(x, _, _, _)| x).collect();
x_lefts.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
// Snap: deduplicate within 2pt tolerance
let mut col_boundaries: Vec<f32> = Vec::new();
for x in &x_lefts {
if col_boundaries
.last()
.is_none_or(|last| (*x - *last).abs() > 2.0)
{
col_boundaries.push(*x);
}
}
if col_boundaries.len() < 5 {
return None;
}
// 1b. Interpolate missing boundaries: holidays/non-work days may not have
// rects, creating gaps. Fill gaps > 1.5× median spacing with evenly spaced
// boundaries so every day gets a column.
if col_boundaries.len() >= 2 {
let mut spacings: Vec<f32> = col_boundaries.windows(2).map(|w| w[1] - w[0]).collect();
spacings.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
let median_spacing = spacings[spacings.len() / 2];
let threshold = median_spacing * 1.5;
let mut filled: Vec<f32> = vec![col_boundaries[0]];
for i in 1..col_boundaries.len() {
let gap = col_boundaries[i] - col_boundaries[i - 1];
if gap > threshold {
// Insert interpolated boundaries
let n = (gap / median_spacing).round() as usize;
if n >= 2 {
let step = gap / n as f32;
for j in 1..n {
filled.push(col_boundaries[i - 1] + j as f32 * step);
}
}
}
filled.push(col_boundaries[i]);
}
col_boundaries = filled;
}
// 2. Split merged multi-number items
let mut expanded_items: Vec<(TextItem, usize)> = Vec::new();
for (idx, item) in items.iter().enumerate() {
let splits = split_merged_numbers(item, &col_boundaries);
for split_item in splits {
expanded_items.push((split_item, idx));
}
}
// 3. Derive row boundaries from item Y positions (5pt tolerance)
let mut y_values: Vec<f32> = expanded_items.iter().map(|(item, _)| item.y).collect();
y_values.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal)); // descending
let mut row_boundaries: Vec<f32> = Vec::new();
for y in &y_values {
if row_boundaries
.last()
.is_none_or(|last| (*last - *y).abs() > 5.0)
{
row_boundaries.push(*y);
}
}
if row_boundaries.is_empty() {
return None;
}
// 4. Assign items to cells
let n_rows = row_boundaries.len();
let n_cols = col_boundaries.len();
let mut cells: Vec<Vec<String>> = vec![vec![String::new(); n_cols]; n_rows];
let mut used_indices: Vec<usize> = Vec::new();
// Compute max X to exclude legend text beyond the table area
let col_spacing = if col_boundaries.len() >= 2 {
(col_boundaries.last().unwrap() - col_boundaries.first().unwrap())
/ (col_boundaries.len() - 1) as f32
} else {
20.0
};
let max_x = col_boundaries.last().unwrap() + col_spacing * 1.5;
for (item, orig_idx) in &expanded_items {
// Skip items beyond the table's rightmost column (legend text)
if item.x > max_x {
continue;
}
// Find row (nearest Y within tolerance)
let row = row_boundaries
.iter()
.position(|&ry| (ry - item.y).abs() <= 5.0);
// Find column: rightmost boundary ≤ item.x + tolerance.
// 4pt tolerance catches annotation items (e.g. "Memorial Day") that sit
// slightly before the next column boundary.
let col = col_boundaries.iter().rposition(|&cx| item.x >= cx - 4.0);
if let (Some(r), Some(c)) = (row, col) {
let cell = &mut cells[r][c];
if !cell.is_empty() {
cell.push(' ');
}
cell.push_str(item.text.trim());
used_indices.push(*orig_idx);
}
}
// 5. Clean up: strip tilde-leader noise from cells (legend text bleeding
// into the last column from the right side of the page)
for row in &mut cells {
for cell in row.iter_mut() {
if let Some(pos) = cell.find("~~~") {
cell.truncate(pos);
*cell = cell.trim_end().to_string();
}
}
}
// 6. Validate: at least one row should have ≥ 5 non-empty cells
let best_row_fill = cells
.iter()
.map(|row| row.iter().filter(|c| !c.is_empty()).count())
.max()
.unwrap_or(0);
if best_row_fill < 5 {
return None;
}
// Deduplicate used indices
used_indices.sort_unstable();
used_indices.dedup();
Some(Table {
columns: col_boundaries,
rows: row_boundaries,
cells,
item_indices: used_indices,
})
}
/// Split a TextItem whose text contains multiple whitespace-separated tokens
/// (like "10 11 12 ... 31") into individual TextItems, each assigned to the
/// nearest column boundary.
fn split_merged_numbers(item: &TextItem, col_boundaries: &[f32]) -> Vec<TextItem> {
let tokens: Vec<&str> = item.text.split_whitespace().collect();
if tokens.len() <= 1 {
return vec![item.clone()];
}
// Count consecutive leading numeric tokens (day numbers like "10 11 12")
let leading_numeric = tokens
.iter()
.take_while(|t| t.chars().all(|c| c.is_ascii_digit()))
.count();
// Need at least one leading number to split
if leading_numeric == 0 {
return vec![item.clone()];
}
let token_width = item.width / tokens.len() as f32;
let mut result = Vec::with_capacity(leading_numeric + 1);
// Find the enclosing column boundary (rightmost boundary ≤ item.x + 2pt),
// then advance through successive boundaries for each leading number.
// Using rposition avoids overshooting when item.x sits between boundaries.
let start_col = col_boundaries
.iter()
.rposition(|&cx| cx <= item.x + 2.0)
.unwrap_or(0);
// Split each leading numeric token into its own item at successive columns
for (i, token) in tokens.iter().enumerate().take(leading_numeric) {
let col_idx = start_col + i;
let snapped_x = if col_idx < col_boundaries.len() {
col_boundaries[col_idx]
} else {
// Fallback: distribute evenly if we run out of boundaries
let raw_x = item.x + i as f32 * token_width + token_width / 2.0;
col_boundaries
.iter()
.rev()
.find(|&&cx| cx <= raw_x + 2.0)
.copied()
.unwrap_or(raw_x)
};
result.push(TextItem {
text: token.to_string(),
x: snapped_x,
width: token_width,
y: item.y,
height: item.height,
font: item.font.clone(),
font_size: item.font_size,
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
item_type: item.item_type.clone(),
});
}
// Trailing non-numeric tokens become annotation placed at last numeric column
if leading_numeric < tokens.len() {
let annotation = tokens[leading_numeric..].join(" ");
let last_x = result.last().map(|i| i.x).unwrap_or(item.x);
result.push(TextItem {
text: annotation,
x: last_x,
width: token_width,
y: item.y,
height: item.height,
font: item.font.clone(),
font_size: item.font_size,
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
item_type: item.item_type.clone(),
});
}
result
}
/// Detection mode controls thresholds for table validation.
#[derive(Debug, Clone, Copy, PartialEq)]
pub(crate) enum TableDetectionMode {
@@ -500,4 +742,166 @@ mod tests {
md
);
}
// ── Rect-guided table builder tests ─────────────────────────────
#[test]
fn rect_guided_basic() {
// 7 column boundaries (like days of week), items "1"-"7" at matching X
let col_xs: Vec<f32> = (0..7).map(|i| 50.0 + i as f32 * 30.0).collect();
let cluster_rects: Vec<(f32, f32, f32, f32)> =
col_xs.iter().map(|&x| (x, 100.0, 28.0, 15.0)).collect();
let items: Vec<TextItem> = (1..=7)
.map(|i| make_item(&i.to_string(), col_xs[i - 1] + 2.0, 110.0, 7.0))
.collect();
let table = try_build_rect_guided_table(&items, &cluster_rects);
assert!(table.is_some(), "Should produce a table from 7 columns");
let table = table.unwrap();
assert_eq!(table.columns.len(), 7);
assert_eq!(table.rows.len(), 1);
for (i, cell) in table.cells[0].iter().enumerate() {
assert_eq!(cell, &(i + 1).to_string());
}
}
#[test]
fn rect_guided_split_merged() {
// One merged item "10 11 12" spanning 3 column boundaries
let col_xs: Vec<f32> = (0..7).map(|i| 50.0 + i as f32 * 30.0).collect();
let cluster_rects: Vec<(f32, f32, f32, f32)> =
col_xs.iter().map(|&x| (x, 100.0, 28.0, 15.0)).collect();
// Single items for cols 0-3, merged "4 5 6" spanning cols 4-6
let mut items = vec![
make_item("1", col_xs[0] + 2.0, 110.0, 7.0),
make_item("2", col_xs[1] + 2.0, 110.0, 7.0),
make_item("3", col_xs[2] + 2.0, 110.0, 7.0),
];
// Merged item spanning from col 3 to col 5 (width covers 3 columns)
let mut merged = make_item("4 5 6", col_xs[3], 110.0, 7.0);
merged.width = 3.0 * 30.0; // spans 3 column widths
items.push(merged);
let table = try_build_rect_guided_table(&items, &cluster_rects);
assert!(table.is_some(), "Should handle merged number items");
let table = table.unwrap();
// Check that "4", "5", "6" ended up in separate columns
let row = &table.cells[0];
assert!(
row.contains(&"4".to_string()),
"Should have '4' in a cell: {:?}",
row
);
assert!(
row.contains(&"5".to_string()),
"Should have '5' in a cell: {:?}",
row
);
assert!(
row.contains(&"6".to_string()),
"Should have '6' in a cell: {:?}",
row
);
}
#[test]
fn rect_guided_with_annotations() {
// Day numbers on one row, annotations on a second row
let col_xs: Vec<f32> = (0..7).map(|i| 50.0 + i as f32 * 30.0).collect();
let cluster_rects: Vec<(f32, f32, f32, f32)> =
col_xs.iter().map(|&x| (x, 100.0, 28.0, 15.0)).collect();
let mut items: Vec<TextItem> = (1..=7)
.map(|i| make_item(&i.to_string(), col_xs[i - 1] + 2.0, 115.0, 7.0))
.collect();
// Add annotation "Holiday" under day 4
items.push(make_item("Holiday", col_xs[3] + 2.0, 105.0, 6.0));
let table = try_build_rect_guided_table(&items, &cluster_rects);
assert!(table.is_some());
let table = table.unwrap();
assert_eq!(
table.rows.len(),
2,
"Should have 2 rows (days + annotations)"
);
// The annotation row should have "Holiday" in column 3
assert_eq!(table.cells[1][3], "Holiday");
}
#[test]
fn rect_guided_too_few_columns() {
// Only 3 column boundaries → should return None (need ≥ 5)
let cluster_rects = vec![
(50.0, 100.0, 28.0, 15.0),
(80.0, 100.0, 28.0, 15.0),
(110.0, 100.0, 28.0, 15.0),
];
let items = vec![
make_item("A", 52.0, 110.0, 7.0),
make_item("B", 82.0, 110.0, 7.0),
make_item("C", 112.0, 110.0, 7.0),
];
let table = try_build_rect_guided_table(&items, &cluster_rects);
assert!(table.is_none(), "Should reject fewer than 5 columns");
}
#[test]
fn split_merged_numbers_single_token() {
let col_boundaries = vec![50.0, 80.0, 110.0, 140.0, 170.0];
let item = make_item("Holiday", 52.0, 110.0, 7.0);
let result = split_merged_numbers(&item, &col_boundaries);
assert_eq!(result.len(), 1, "Single-token item should not be split");
assert_eq!(result[0].text, "Holiday");
}
#[test]
fn split_leading_numbers_with_annotation() {
// "11 Veterans Day" → "11" split off, "Veterans Day" as annotation
let col_boundaries = vec![50.0, 80.0, 110.0, 140.0, 170.0];
let mut item = make_item("11 Veterans Day", 110.0, 110.0, 7.0);
item.width = 90.0; // spans 3 tokens
let result = split_merged_numbers(&item, &col_boundaries);
assert_eq!(result.len(), 2, "Should split into number + annotation");
assert_eq!(result[0].text, "11");
assert_eq!(result[1].text, "Veterans Day");
}
#[test]
fn split_multiple_leading_numbers_with_annotation() {
// "24 25 Memorial Day" → "24", "25" split, "Memorial Day" trails
let col_xs: Vec<f32> = (0..7).map(|i| 50.0 + i as f32 * 30.0).collect();
let mut item = make_item("24 25 Memorial Day", col_xs[3], 110.0, 7.0);
item.width = 4.0 * 30.0; // spans 4 tokens
let result = split_merged_numbers(&item, &col_xs);
assert_eq!(result.len(), 3, "Should split into 2 numbers + annotation");
assert_eq!(result[0].text, "24");
assert_eq!(result[1].text, "25");
assert_eq!(result[2].text, "Memorial Day");
}
#[test]
fn split_no_leading_numbers() {
// "Memorial Day" → no leading numeric, returned as-is
let col_boundaries = vec![50.0, 80.0, 110.0, 140.0, 170.0];
let item = make_item("Memorial Day", 52.0, 110.0, 7.0);
let result = split_merged_numbers(&item, &col_boundaries);
assert_eq!(result.len(), 1);
assert_eq!(result[0].text, "Memorial Day");
}
#[test]
fn rect_guided_tilde_cleanup() {
// Items with tilde noise should have it stripped
let col_xs: Vec<f32> = (0..7).map(|i| 50.0 + i as f32 * 30.0).collect();
let cluster_rects: Vec<(f32, f32, f32, f32)> =
col_xs.iter().map(|&x| (x, 100.0, 28.0, 15.0)).collect();
let mut items: Vec<TextItem> = (1..=7)
.map(|i| make_item(&i.to_string(), col_xs[i - 1] + 2.0, 110.0, 7.0))
.collect();
// Day 7 has tilde-leader legend text bleeding in
items[6] = make_item("7 ~~~~~~~ Legend text here", col_xs[6] + 2.0, 110.0, 7.0);
let table = try_build_rect_guided_table(&items, &cluster_rects).unwrap();
assert_eq!(table.cells[0][6], "7", "Tilde noise should be stripped");
}
}