Compare commits

...
Author SHA1 Message Date
Abimael Martell b8249dfd61 docs: note multi-row expansion scope
Clarify that the row-band cap intentionally keeps v1 focused on common small row-loss cases while larger compressions continue to use heuristic fallback.

Made-with: Cursor
2026-04-29 08:08:45 -07:00
Abimael Martell c7612ceb16 tables: expand multi-row TSR cells in place
Recover row-under-counted TSR tables by splitting overstuffed cells with native PDF text bands before falling back to heuristic extraction.

Made-with: Cursor
2026-04-29 07:44:45 -07:00
3 changed files with 603 additions and 87 deletions
+9 -7
View File
@@ -482,9 +482,10 @@ pub fn extract_tables_with_structure_cells(
/// `fallbackReason` is `null` when the TSR-hybrid path produced the /// `fallbackReason` is `null` when the TSR-hybrid path produced the
/// markdown directly. When stage 1's quality check fires (the cells /// markdown directly. When stage 1's quality check fires (the cells
/// look like a SLANet detection pathology — phantom rows or multi-row /// look like a SLANet detection pathology — phantom rows or multi-row
/// content in a single cell), the heuristic table extractor is run on /// content in a single cell), the auto path may expand the TSR cells
/// the same region instead, and `fallbackReason` carries the diagnostic /// in-place or run the heuristic table extractor on the same region.
/// label (`"phantom_empty_row"`, `"multi_row_in_cell"`). /// `fallbackReason` carries the diagnostic label (for example
/// `"multi_row_in_cell_expanded"` or `"phantom_empty_row"`).
#[napi(object)] #[napi(object)]
pub struct TableExtractionResultJs { pub struct TableExtractionResultJs {
pub markdown: String, pub markdown: String,
@@ -494,13 +495,14 @@ pub struct TableExtractionResultJs {
/// Auto-fallback variant of [`extractTablesWithStructure`]. /// Auto-fallback variant of [`extractTablesWithStructure`].
/// ///
/// Runs the TSR-hybrid path, checks the resulting cells for known /// Runs the TSR-hybrid path, checks the resulting cells for known
/// SLANet detection pathologies, and falls back to the heuristic /// SLANet detection pathologies, expands multi-row cells in-place when
/// `extractTablesInRegions` for any input where the TSR path looks /// possible, and otherwise falls back to the heuristic
/// `extractTablesInRegions` for inputs where the TSR path looks
/// compromised. /// compromised.
/// ///
/// On clean inputs this returns identical markdown to /// On clean inputs this returns identical markdown to
/// `extractTablesWithStructure`; on flagged inputs the heuristic /// `extractTablesWithStructure`; on flagged inputs `fallbackReason` is
/// markdown replaces the TSR markdown and `fallbackReason` is set. /// set to the recovery path that produced the result.
#[napi] #[napi]
pub fn extract_tables_with_structure_auto( pub fn extract_tables_with_structure_auto(
buffer: Buffer, buffer: Buffer,
+428 -62
View File
@@ -55,7 +55,7 @@ pub use process_mode::ProcessMode;
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem}; pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
use lopdf::Document; use lopdf::Document;
use std::collections::{HashMap, HashSet}; use std::collections::{BTreeMap, HashMap, HashSet};
use std::path::Path; use std::path::Path;
use tounicode::FontCMaps; use tounicode::FontCMaps;
@@ -1775,20 +1775,339 @@ pub fn extract_tables_with_structure_mem(
/// ///
/// `fallback_reason` is `None` when the TSR-hybrid path produced the /// `fallback_reason` is `None` when the TSR-hybrid path produced the
/// markdown directly; `Some(<short identifier>)` when stage 1's quality /// markdown directly; `Some(<short identifier>)` when stage 1's quality
/// check fired and the heuristic `extract_tables_in_regions_mem` was /// check fired and either in-place expansion or the heuristic fallback
/// substituted instead. The reason string is stable enough to use as a /// produced the output. The reason string is stable enough to use as a
/// metric label (e.g. `phantom_empty_row`, `multi_row_in_cell`). /// metric label (e.g. `multi_row_in_cell_expanded`, `phantom_empty_row`).
#[derive(Debug, Clone)] #[derive(Debug, Clone)]
pub struct TableExtractionResult { pub struct TableExtractionResult {
pub markdown: String, pub markdown: String,
pub fallback_reason: Option<String>, pub fallback_reason: Option<String>,
} }
#[derive(Debug, Clone)]
enum TsrQualityIssue {
PhantomEmptyRow,
MultiRowInCell {
expanded_cells: Option<Vec<tables::StructuredCell>>,
},
}
impl TsrQualityIssue {
fn reason(&self) -> &'static str {
match self {
Self::PhantomEmptyRow => "phantom_empty_row",
Self::MultiRowInCell { .. } => "multi_row_in_cell",
}
}
}
#[derive(Clone)]
struct TsrCellTextLine {
center_y: f32,
half_height: f32,
items: Vec<TextItem>,
}
impl TsrCellTextLine {
fn new(item: TextItem) -> Self {
let center_y = item.y + item.height * 0.5;
let half_height = (item.height * 0.5).max(2.5);
Self {
center_y,
half_height,
items: vec![item],
}
}
fn add(&mut self, item: TextItem) {
let center_y = item.y + item.height * 0.5;
let existing = self.items.len() as f32;
self.center_y = (self.center_y * existing + center_y) / (existing + 1.0);
self.half_height = self.half_height.max((item.height * 0.5).max(2.5));
self.items.push(item);
}
fn bottom_y(&self) -> f32 {
self.items
.iter()
.map(|item| item.y)
.fold(f32::INFINITY, f32::min)
}
}
#[derive(Clone)]
struct TsrRowExpansion {
bands: Vec<f32>,
tolerance: f32,
}
fn collect_items_in_tsr_cell(
items: &[TextItem],
cell: &tables::StructuredCell,
page_height: f32,
coord_space: RegionCoordSpace,
) -> Vec<TextItem> {
let [x1, y1, x2, y2] = cell.page_pt_bbox;
if x1 >= x2 || y1 >= y2 {
return Vec::new();
}
let bounds = region_bounds(x1, y1, x2, y2, page_height, coord_space);
items
.iter()
.filter(|item| !item.text.trim().is_empty() && tsr_region_contains_item(item, bounds))
.cloned()
.collect()
}
fn cluster_tsr_cell_text_lines(mut items: Vec<TextItem>) -> Vec<TsrCellTextLine> {
if items.is_empty() {
return Vec::new();
}
items.sort_by(|a, b| {
let ay = a.y + a.height * 0.5;
let by = b.y + b.height * 0.5;
by.total_cmp(&ay).then(a.x.total_cmp(&b.x))
});
let mut lines: Vec<TsrCellTextLine> = Vec::new();
for item in items {
let item_top = item.y + item.height;
let item_half_height = (item.height * 0.5).max(2.5);
if let Some(last) = lines.last_mut() {
let gap = last.bottom_y() - item_top;
if gap <= last.half_height.max(item_half_height) {
last.add(item);
continue;
}
}
lines.push(TsrCellTextLine::new(item));
}
lines
}
fn build_tsr_row_expansion(
row_cells: &[usize],
cells: &[tables::StructuredCell],
cell_lines: &[Vec<TsrCellTextLine>],
) -> Option<TsrRowExpansion> {
if row_cells.is_empty() {
return None;
}
let has_spanning_cell = row_cells
.iter()
.any(|&idx| cells[idx].rowspan > 1 || cells[idx].colspan > 1);
if has_spanning_cell {
return None;
}
let multiline_cells = row_cells
.iter()
.filter(|&&idx| cell_lines[idx].len() >= 2)
.count();
if row_cells.len() >= 2 && multiline_cells < 2 {
return None;
}
if row_cells.len() == 1 && multiline_cells == 0 {
return None;
}
let mut centers: Vec<(f32, f32)> = row_cells
.iter()
.flat_map(|&idx| {
cell_lines[idx]
.iter()
.map(|line| (line.center_y, line.half_height))
})
.collect();
if centers.len() < 2 {
return None;
}
centers.sort_by(|a, b| b.0.total_cmp(&a.0));
let mut half_heights: Vec<f32> = centers.iter().map(|(_, h)| *h).collect();
half_heights.sort_by(|a, b| a.total_cmp(b));
let tolerance = (half_heights[half_heights.len() / 2] * 0.8).max(3.0);
let mut bands: Vec<(f32, usize)> = Vec::new();
for (center, _) in centers {
if let Some((band_center, count)) = bands
.iter_mut()
.find(|(band_center, _)| (*band_center - center).abs() <= tolerance)
{
*band_center = (*band_center * *count as f32 + center) / (*count as f32 + 1.0);
*count += 1;
} else {
bands.push((center, 1));
}
}
// V1 targets the common 1-2 lost-row cases; larger compressions stay on
// the existing heuristic fallback path until we have evidence to broaden it.
if !(2..=4).contains(&bands.len()) {
return None;
}
let min_support = if row_cells.len() >= 2 { 2 } else { 1 };
let supported = bands.iter().all(|(band, _)| {
row_cells
.iter()
.filter(|&&idx| {
cell_lines[idx]
.iter()
.any(|line| (line.center_y - *band).abs() <= tolerance)
})
.count()
>= min_support
});
if !supported {
return None;
}
bands.sort_by(|a, b| b.0.total_cmp(&a.0));
Some(TsrRowExpansion {
bands: bands.into_iter().map(|(center, _)| center).collect(),
tolerance,
})
}
fn text_for_tsr_band(
lines: &[TsrCellTextLine],
band: f32,
tolerance: f32,
adaptive_threshold: f32,
) -> String {
let mut matched: Vec<TextItem> = lines
.iter()
.filter(|line| (line.center_y - band).abs() <= tolerance)
.flat_map(|line| line.items.iter().cloned())
.collect();
if matched.is_empty() {
return String::new();
}
matched.sort_by(|a, b| b.y.total_cmp(&a.y).then(a.x.total_cmp(&b.x)));
collect_text_from_matched_items(matched, adaptive_threshold).replace(['\n', '\r'], " ")
}
fn slice_cell_bbox_for_expanded_row(
cell: &tables::StructuredCell,
row_idx: usize,
row_count: usize,
) -> [f32; 4] {
let mut bbox = cell.page_pt_bbox;
let top = bbox[1].min(bbox[3]);
let bottom = bbox[1].max(bbox[3]);
let height = bottom - top;
if height <= 0.0 || row_count == 0 {
return bbox;
}
let step = height / row_count as f32;
bbox[1] = top + step * row_idx as f32;
bbox[3] = if row_idx + 1 == row_count {
bottom
} else {
top + step * (row_idx + 1) as f32
};
bbox
}
fn try_expand_multi_row_cells(
cells: &[tables::StructuredCell],
items: &[TextItem],
page_height: f32,
coord_space: RegionCoordSpace,
adaptive_threshold: f32,
) -> Option<Vec<tables::StructuredCell>> {
if cells.is_empty() {
return None;
}
let cell_lines: Vec<Vec<TsrCellTextLine>> = cells
.iter()
.map(|cell| {
if cell.rowspan > 1 {
Vec::new()
} else {
cluster_tsr_cell_text_lines(collect_items_in_tsr_cell(
items,
cell,
page_height,
coord_space,
))
}
})
.collect();
let mut cells_by_row: BTreeMap<usize, Vec<usize>> = BTreeMap::new();
for (idx, cell) in cells.iter().enumerate() {
cells_by_row.entry(cell.row).or_default().push(idx);
}
let mut expansions: HashMap<usize, TsrRowExpansion> = HashMap::new();
for (&row, row_cells) in &cells_by_row {
let covered_by_rowspan = cells
.iter()
.any(|cell| cell.rowspan > 1 && cell.row <= row && row < cell.row + cell.rowspan);
if covered_by_rowspan {
continue;
}
if let Some(expansion) = build_tsr_row_expansion(row_cells, cells, &cell_lines) {
expansions.insert(row, expansion);
}
}
if expansions.is_empty() {
return None;
}
let mut expanded = Vec::with_capacity(cells.len() + expansions.len());
let mut row_shift = 0usize;
for (&row, row_cells) in &cells_by_row {
if let Some(expansion) = expansions.get(&row) {
for (band_idx, band) in expansion.bands.iter().enumerate() {
for &cell_idx in row_cells {
let mut cell = cells[cell_idx].clone();
cell.row = row + row_shift + band_idx;
cell.rowspan = 1;
cell.text = text_for_tsr_band(
&cell_lines[cell_idx],
*band,
expansion.tolerance,
adaptive_threshold,
);
cell.page_pt_bbox =
slice_cell_bbox_for_expanded_row(&cell, band_idx, expansion.bands.len());
expanded.push(cell);
}
}
row_shift += expansion.bands.len() - 1;
} else {
for &cell_idx in row_cells {
let mut cell = cells[cell_idx].clone();
cell.row += row_shift;
expanded.push(cell);
}
}
}
let original_rows = cells
.iter()
.map(|cell| cell.row + cell.rowspan.max(1))
.max()
.unwrap_or(0);
let expanded_rows = expanded
.iter()
.map(|cell| cell.row + cell.rowspan.max(1))
.max()
.unwrap_or(0);
(expanded_rows > original_rows).then_some(expanded)
}
/// Detect quality issues in the TSR-hybrid output for a single input. /// Detect quality issues in the TSR-hybrid output for a single input.
/// ///
/// Returns `Some(reason)` if the cells look like they reflect a known /// Returns `Some(issue)` if the cells look like they reflect a known
/// SLANet detection pathology that the heuristic table extractor would /// SLANet detection pathology. Reasons (also used as metric labels):
/// likely handle better. Reasons (also used as metric labels):
/// ///
/// * `phantom_empty_row` — a row whose every cell is empty, surrounded /// * `phantom_empty_row` — a row whose every cell is empty, surrounded
/// above and below by rows with content. SLANet sometimes emits an /// above and below by rows with content. SLANet sometimes emits an
@@ -1804,7 +2123,7 @@ fn detect_tsr_quality_issue(
buffer: &[u8], buffer: &[u8],
input: &TsrTableInput, input: &TsrTableInput,
cells: &[tables::StructuredCell], cells: &[tables::StructuredCell],
) -> Result<Option<String>, PdfError> { ) -> Result<Option<TsrQualityIssue>, PdfError> {
if cells.is_empty() { if cells.is_empty() {
return Ok(None); return Ok(None);
} }
@@ -1820,7 +2139,7 @@ fn detect_tsr_quality_issue(
} }
for r in 1..max_row { for r in 1..max_row {
if !row_has_content[r] && row_has_content[r - 1] && row_has_content[r + 1] { if !row_has_content[r] && row_has_content[r - 1] && row_has_content[r + 1] {
return Ok(Some("phantom_empty_row".to_string())); return Ok(Some(TsrQualityIssue::PhantomEmptyRow));
} }
} }
} }
@@ -1849,12 +2168,14 @@ fn detect_tsr_quality_issue(
&font_cmaps, &font_cmaps,
false, false,
)?; )?;
let _ = text_utils::fix_letterspaced_items(&mut items); let adaptive_threshold = text_utils::fix_letterspaced_items(&mut items);
let coords = if coords_rotated { let coords = if coords_rotated {
RegionCoordSpace::Rotated90Ccw RegionCoordSpace::Rotated90Ccw
} else { } else {
RegionCoordSpace::Standard RegionCoordSpace::Standard
}; };
let expanded_cells =
try_expand_multi_row_cells(cells, &items, page_h, coords, adaptive_threshold);
for cell in cells { for cell in cells {
// rowspan>1 cells are intentionally multi-line — skip them. // rowspan>1 cells are intentionally multi-line — skip them.
@@ -1864,57 +2185,12 @@ fn detect_tsr_quality_issue(
if cell.text.trim().is_empty() { if cell.text.trim().is_empty() {
continue; continue;
} }
let [x1, y1, x2, y2] = cell.page_pt_bbox; let cell_items = collect_items_in_tsr_cell(&items, cell, page_h, coords);
if x1 >= x2 || y1 >= y2 {
continue;
}
let bounds = region_bounds(x1, y1, x2, y2, page_h, coords);
// Collect the items inside this cell, with their y-centers and
// half-heights so we can cluster them into visual lines.
let mut cell_items: Vec<(f32, f32)> = Vec::new();
for item in &items {
if tsr_region_contains_item(item, bounds) {
let cy = item.y + item.height * 0.5;
let half_h = (item.height * 0.5).max(2.5);
cell_items.push((cy, half_h));
}
}
if cell_items.len() < 2 { if cell_items.len() < 2 {
continue; continue;
} }
// Sort by y-center descending (top-of-page first in PDF native if cluster_tsr_cell_text_lines(cell_items).len() >= 2 {
// coords where y grows upward) — direction doesn't matter, we return Ok(Some(TsrQualityIssue::MultiRowInCell { expanded_cells }));
// just need consecutive items to be neighbors in the sort.
cell_items.sort_by(|a, b| b.0.total_cmp(&a.0));
// Walk pairs and see if there's a real whitespace gap between
// any two adjacent items — defined as their bounding-box edges
// separated by more than half a line height. This rules out
// tall glyphs / superscripts / accents on a single visual line.
let max_half_h = cell_items
.iter()
.map(|(_, h)| *h)
.fold(0f32, f32::max)
.max(2.5);
let gap_threshold = max_half_h; // ≈ half a line height
let mut found_gap = false;
for w in cell_items.windows(2) {
let (cy_a, h_a) = w[0];
let (cy_b, h_b) = w[1];
// Gap = distance between the bottom of the upper item and
// the top of the lower item, measured in PDF-native coords
// (y grows upward, so the upper item has the larger cy).
let upper_bottom = cy_a - h_a;
let lower_top = cy_b + h_b;
let gap = upper_bottom - lower_top;
if gap > gap_threshold {
found_gap = true;
break;
}
}
if found_gap {
return Ok(Some("multi_row_in_cell".to_string()));
} }
} }
@@ -1927,9 +2203,11 @@ fn detect_tsr_quality_issue(
/// and falls back to the heuristic [`extract_tables_in_regions_mem`] /// and falls back to the heuristic [`extract_tables_in_regions_mem`]
/// for any input where the TSR path looks compromised. /// for any input where the TSR path looks compromised.
/// ///
/// On clean inputs this is identical to the markdown variant. /// On clean inputs this is identical to the markdown variant. On
/// On flagged inputs the heuristic markdown replaces the TSR markdown /// `multi_row_in_cell`, the wrapper first tries to expand over-stuffed
/// and the result's `fallback_reason` is set to the diagnostic label. /// rows in place; if that cannot produce a usable table, the heuristic
/// markdown replaces the TSR markdown and `fallback_reason` is set to
/// the diagnostic label.
/// ///
/// Two failure modes are guarded against per-input: /// Two failure modes are guarded against per-input:
/// ///
@@ -1939,6 +2217,9 @@ fn detect_tsr_quality_issue(
/// `_heuristic_empty` (e.g. `multi_row_in_cell_heuristic_empty`). /// `_heuristic_empty` (e.g. `multi_row_in_cell_heuristic_empty`).
/// This avoids replacing a usable wrong-but-non-empty TSR output /// This avoids replacing a usable wrong-but-non-empty TSR output
/// with literally nothing. /// with literally nothing.
/// * **Expanded multi-row cells**: when in-place recovery succeeds, the
/// result is labeled `multi_row_in_cell_expanded` and the heuristic is
/// not consulted.
/// * **Per-input errors**: any failure in detection or heuristic /// * **Per-input errors**: any failure in detection or heuristic
/// extraction for a single input is contained — that input /// extraction for a single input is contained — that input
/// returns the raw TSR markdown with `fallback_reason` set to /// returns the raw TSR markdown with `fallback_reason` set to
@@ -1983,7 +2264,21 @@ pub fn extract_tables_with_structure_auto_mem(
markdown: tsr_md, markdown: tsr_md,
fallback_reason: None, fallback_reason: None,
}, },
Some(reason) => { Some(issue) => {
let reason = issue.reason().to_string();
if let TsrQualityIssue::MultiRowInCell {
expanded_cells: Some(expanded_cells),
} = issue
{
let expanded_md = tables::cells_to_markdown(&expanded_cells);
if !expanded_md.trim().is_empty() {
results.push(TableExtractionResult {
markdown: expanded_md,
fallback_reason: Some("multi_row_in_cell_expanded".to_string()),
});
continue;
}
}
// Fall back to heuristic on the input's table region. // Fall back to heuristic on the input's table region.
// The crop's PDF-pt bbox IS the table region. // The crop's PDF-pt bbox IS the table region.
let heuristic_md = match extract_tables_in_regions_mem( let heuristic_md = match extract_tables_in_regions_mem(
@@ -3650,6 +3945,77 @@ mod tests {
assert!(!cells[2].text.contains("Boardwalk")); assert!(!cells[2].text.contains("Boardwalk"));
} }
#[test]
fn multi_row_expansion_splits_overstuffed_tsr_row() {
use crate::tables::{cells_to_markdown, StructuredCell};
let items = vec![
test_item("Branch Name", 20.0, 166.0, 55.0, 8.0),
test_item("Deposits", 120.0, 166.0, 36.0, 8.0),
test_item("Oak Street", 20.0, 136.0, 48.0, 8.0),
test_item("100", 120.0, 136.0, 18.0, 8.0),
test_item("Boardwalk", 20.0, 116.0, 46.0, 8.0),
test_item("200", 120.0, 116.0, 18.0, 8.0),
];
let cells = vec![
StructuredCell {
row: 0,
col: 0,
rowspan: 1,
colspan: 1,
is_header: true,
text: "Branch Name".into(),
page_pt_bbox: [10.0, 20.0, 100.0, 40.0],
},
StructuredCell {
row: 0,
col: 1,
rowspan: 1,
colspan: 1,
is_header: true,
text: "Deposits".into(),
page_pt_bbox: [110.0, 20.0, 190.0, 40.0],
},
StructuredCell {
row: 1,
col: 0,
rowspan: 1,
colspan: 1,
is_header: false,
text: "Oak Street Boardwalk".into(),
page_pt_bbox: [10.0, 40.0, 100.0, 100.0],
},
StructuredCell {
row: 1,
col: 1,
rowspan: 1,
colspan: 1,
is_header: false,
text: "100 200".into(),
page_pt_bbox: [110.0, 40.0, 190.0, 100.0],
},
];
let expanded =
try_expand_multi_row_cells(&cells, &items, 200.0, RegionCoordSpace::Standard, 0.10)
.expect("overstuffed data row should expand");
let md = cells_to_markdown(&expanded);
assert_eq!(expanded.iter().map(|c| c.row).max().unwrap() + 1, 3);
assert!(
md.contains("|Oak Street|100|"),
"missing first data row: {md}"
);
assert!(
md.contains("|Boardwalk|200|"),
"missing second data row: {md}"
);
assert!(
!md.contains("Oak Street Boardwalk"),
"compressed cell text should be replaced: {md}"
);
}
#[test] #[test]
fn tsr_assignment_caps_uses_median_geometry() { fn tsr_assignment_caps_uses_median_geometry() {
use crate::tables::StructuredCell; use crate::tables::StructuredCell;
+166 -18
View File
@@ -1812,6 +1812,93 @@ fn synthetic_vector_grid_pdf(two_tables: bool) -> Vec<u8> {
bytes bytes
} }
fn synthetic_vector_grid_three_row_pdf() -> Vec<u8> {
use lopdf::content::{Content, Operation};
use lopdf::{dictionary, Document, Object, Stream};
let mut doc = Document::with_version("1.5");
let pages_id = doc.new_object_id();
let page_id = doc.new_object_id();
let font_id = doc.new_object_id();
let content_id = doc.new_object_id();
doc.objects.insert(
font_id,
dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
}
.into(),
);
let mut operations = Vec::new();
for y in [740, 710, 680, 650] {
operations.push(Operation::new("m", vec![50.into(), y.into()]));
operations.push(Operation::new("l", vec![210.into(), y.into()]));
}
for x in [50, 130, 210] {
operations.push(Operation::new("m", vec![x.into(), 650.into()]));
operations.push(Operation::new("l", vec![x.into(), 740.into()]));
}
operations.push(Operation::new("S", vec![]));
operations.push(Operation::new("BT", vec![]));
operations.push(Operation::new("Tf", vec!["F1".into(), 10.into()]));
for (x, y, text) in [
(70, 724, "Branch"),
(150, 724, "Deposits"),
(70, 694, "Oak"),
(150, 694, "100"),
(70, 664, "Boardwalk"),
(150, 664, "200"),
] {
operations.push(Operation::new(
"Tm",
vec![1.into(), 0.into(), 0.into(), 1.into(), x.into(), y.into()],
));
operations.push(Operation::new("Tj", vec![Object::string_literal(text)]));
}
operations.push(Operation::new("ET", vec![]));
let content = Content { operations }.encode().unwrap();
doc.objects
.insert(content_id, Stream::new(dictionary! {}, content).into());
doc.objects.insert(
page_id,
dictionary! {
"Type" => "Page",
"Parent" => pages_id,
"MediaBox" => vec![0.into(), 0.into(), 300.into(), 800.into()],
"Resources" => dictionary! {
"Font" => dictionary! {
"F1" => font_id,
},
},
"Contents" => content_id,
}
.into(),
);
doc.objects.insert(
pages_id,
dictionary! {
"Type" => "Pages",
"Kids" => vec![page_id.into()],
"Count" => 1,
}
.into(),
);
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => pages_id,
});
doc.trailer.set("Root", catalog_id);
let mut bytes = Vec::new();
doc.save_to(&mut bytes).unwrap();
bytes
}
fn assert_close(actual: f32, expected: f32) { fn assert_close(actual: f32, expected: f32) {
assert!( assert!(
(actual - expected).abs() < 0.75, (actual - expected).abs() < 0.75,
@@ -2319,7 +2406,7 @@ fn test_auto_passes_through_clean_tsr_output() {
} }
#[test] #[test]
fn test_auto_falls_back_on_multi_row_in_cell() { fn test_auto_expands_multi_row_in_cell() {
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput}; use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
let buf = synthetic_dense_table_pdf(); let buf = synthetic_dense_table_pdf();
@@ -2370,16 +2457,79 @@ fn test_auto_falls_back_on_multi_row_in_cell() {
assert_eq!(results.len(), 1); assert_eq!(results.len(), 1);
assert_eq!( assert_eq!(
results[0].fallback_reason.as_deref(), results[0].fallback_reason.as_deref(),
Some("multi_row_in_cell"), Some("multi_row_in_cell_expanded"),
"expected multi_row_in_cell fallback, got {:?}", "expected multi_row_in_cell_expanded, got {:?}",
results[0].fallback_reason results[0].fallback_reason
); );
// The heuristic-fallback markdown should preserve all three PDF rows. // The in-place expansion should preserve all three PDF rows.
let md = &results[0].markdown; let md = &results[0].markdown;
assert!(md.contains("Oak Street"), "missing Oak Street: {md}"); assert!(md.contains("Oak Street"), "missing Oak Street: {md}");
assert!(md.contains("Boardwalk"), "missing Boardwalk: {md}"); assert!(md.contains("Boardwalk"), "missing Boardwalk: {md}");
assert!(md.contains("100"), "missing 100: {md}"); assert!(md.contains("100"), "missing 100: {md}");
assert!(md.contains("200"), "missing 200: {md}"); assert!(md.contains("200"), "missing 200: {md}");
assert!(
!md.contains("Oak Street Boardwalk"),
"rows should not remain compressed: {md}"
);
}
#[test]
fn test_auto_expands_under_counted_vector_grid_rows() {
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
let buf = synthetic_vector_grid_three_row_pdf();
let tokens: Vec<String> = [
"<table>",
"<thead>",
"<tr>",
"<th></th>",
"<th></th>",
"</tr>",
"</thead>",
"<tbody>",
"<tr>",
"<td></td>",
"<td></td>",
"</tr>",
"</tbody>",
"</table>",
]
.into_iter()
.map(String::from)
.collect();
let crop = [50.0, 60.0, 210.0, 150.0];
let cell_bboxes = vec![
poly(0.0, 0.0, 80.0, 30.0),
poly(80.0, 0.0, 160.0, 30.0),
poly(0.0, 30.0, 80.0, 90.0),
poly(80.0, 30.0, 160.0, 90.0),
];
let results = extract_tables_with_structure_auto_mem(
&buf,
&[TsrTableInput {
page: 0,
crop_pdf_pt_bbox: crop,
render_dpi: 72.0,
structure_tokens: tokens,
cell_bboxes,
}],
)
.unwrap();
assert_eq!(results.len(), 1);
assert_eq!(
results[0].fallback_reason.as_deref(),
Some("multi_row_in_cell_expanded")
);
let md = &results[0].markdown;
assert!(md.contains("|Branch|Deposits|"), "missing header: {md}");
assert!(md.contains("|Oak|100|"), "missing row 1: {md}");
assert!(md.contains("|Boardwalk|200|"), "missing row 2: {md}");
assert!(
!md.contains("Oak Boardwalk"),
"rows stayed compressed: {md}"
);
} }
#[test] #[test]
@@ -2456,15 +2606,14 @@ fn test_auto_does_not_fire_on_legit_rowspan_cell() {
} }
#[test] #[test]
fn test_auto_keeps_tsr_markdown_when_heuristic_returns_empty() { fn test_auto_expands_when_heuristic_region_is_empty() {
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput}; use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
let buf = synthetic_dense_table_pdf(); let buf = synthetic_dense_table_pdf();
// Same shape as the multi_row_in_cell regression — a tall data cell // Same shape as the multi_row_in_cell regression — a tall data cell
// that catches Oak Street + Boardwalk. But the crop bbox we pass // that catches Oak Street + Boardwalk. The crop bbox we pass points
// points at a strip of the page that has NO text items, so the // at a strip of the page that has NO text items, so the old heuristic
// heuristic's region will be empty when it tries to extract there. // fallback would be empty. Expansion uses the cell bboxes directly.
// The auto wrapper must keep the TSR markdown rather than ship "".
let tokens: Vec<String> = [ let tokens: Vec<String> = [
"<table>", "<table>",
"<thead>", "<thead>",
@@ -2510,20 +2659,19 @@ fn test_auto_keeps_tsr_markdown_when_heuristic_returns_empty() {
let r = &results[0]; let r = &results[0];
assert_eq!( assert_eq!(
r.fallback_reason.as_deref(), r.fallback_reason.as_deref(),
Some("multi_row_in_cell_heuristic_empty"), Some("multi_row_in_cell_expanded"),
"expected _heuristic_empty suffix, got {:?}", "expected expansion despite empty heuristic region, got {:?}",
r.fallback_reason, r.fallback_reason,
); );
// TSR markdown should be preserved — non-empty, contains the cell
// text we know was assigned by the TSR path.
assert!( assert!(
!r.markdown.trim().is_empty(), r.markdown.contains("|Oak Street|100|"),
"expected TSR markdown to be preserved, got empty", "missing row 1: {}",
r.markdown
); );
assert!( assert!(
r.markdown.contains("Oak Street") || r.markdown.contains("Boardwalk"), r.markdown.contains("|Boardwalk|200|"),
"expected TSR markdown to contain at least one row, got: {}", "missing row 2: {}",
r.markdown, r.markdown
); );
} }