Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
66103742c3 | ||
|
|
97fc32ac70 | ||
|
|
a4161c8392 | ||
|
|
5b1fe30c66 |
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.8.0",
|
||||
"version": "1.8.3",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
+9
-7
@@ -482,9 +482,10 @@ pub fn extract_tables_with_structure_cells(
|
||||
/// `fallbackReason` is `null` when the TSR-hybrid path produced the
|
||||
/// markdown directly. When stage 1's quality check fires (the cells
|
||||
/// look like a SLANet detection pathology — phantom rows or multi-row
|
||||
/// content in a single cell), the heuristic table extractor is run on
|
||||
/// the same region instead, and `fallbackReason` carries the diagnostic
|
||||
/// label (`"phantom_empty_row"`, `"multi_row_in_cell"`).
|
||||
/// content in a single cell), the auto path may expand the TSR cells
|
||||
/// in-place or run the heuristic table extractor on the same region.
|
||||
/// `fallbackReason` carries the diagnostic label (for example
|
||||
/// `"multi_row_in_cell_expanded"` or `"phantom_empty_row"`).
|
||||
#[napi(object)]
|
||||
pub struct TableExtractionResultJs {
|
||||
pub markdown: String,
|
||||
@@ -494,13 +495,14 @@ pub struct TableExtractionResultJs {
|
||||
/// Auto-fallback variant of [`extractTablesWithStructure`].
|
||||
///
|
||||
/// Runs the TSR-hybrid path, checks the resulting cells for known
|
||||
/// SLANet detection pathologies, and falls back to the heuristic
|
||||
/// `extractTablesInRegions` for any input where the TSR path looks
|
||||
/// SLANet detection pathologies, expands multi-row cells in-place when
|
||||
/// possible, and otherwise falls back to the heuristic
|
||||
/// `extractTablesInRegions` for inputs where the TSR path looks
|
||||
/// compromised.
|
||||
///
|
||||
/// On clean inputs this returns identical markdown to
|
||||
/// `extractTablesWithStructure`; on flagged inputs the heuristic
|
||||
/// markdown replaces the TSR markdown and `fallbackReason` is set.
|
||||
/// `extractTablesWithStructure`; on flagged inputs `fallbackReason` is
|
||||
/// set to the recovery path that produced the result.
|
||||
#[napi]
|
||||
pub fn extract_tables_with_structure_auto(
|
||||
buffer: Buffer,
|
||||
|
||||
+492
-66
@@ -55,7 +55,7 @@ pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
|
||||
use lopdf::Document;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::collections::{BTreeMap, HashMap, HashSet};
|
||||
use std::path::Path;
|
||||
use tounicode::FontCMaps;
|
||||
|
||||
@@ -995,6 +995,46 @@ fn crop_px_bbox_is_plausible(
|
||||
mod vector_grid_tests {
|
||||
use super::crop_px_bbox_is_plausible;
|
||||
|
||||
/// Regression for `forecast_table_chart.pdf` (doc 128 from the
|
||||
/// opendataloader-bench corpus). The table has six visual columns, but
|
||||
/// text X-clustering in the cell-rect fallback previously split wide
|
||||
/// columns into ten spurious columns.
|
||||
#[test]
|
||||
fn forecast_table_chart_six_cols() {
|
||||
use crate::extractor::content_stream::extract_page_text_items;
|
||||
use crate::tables::detect_tables_from_rects;
|
||||
use crate::tounicode::FontCMaps;
|
||||
use lopdf::Document;
|
||||
use std::collections::HashSet;
|
||||
use std::fs;
|
||||
|
||||
let path = "tests/fixtures/forecast_table_chart.pdf";
|
||||
let buf = fs::read(path).unwrap();
|
||||
let doc = Document::load_mem(&buf).unwrap();
|
||||
let pages = doc.get_pages();
|
||||
let &page_id = pages.get(&1).unwrap();
|
||||
let needed: HashSet<u32> = HashSet::from([1]);
|
||||
let cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed));
|
||||
let ((items, rects, _lines), _has_gid, _rotated) =
|
||||
extract_page_text_items(&doc, page_id, 1, &cmaps, false).unwrap();
|
||||
|
||||
let (rect_tables, _) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert_eq!(rect_tables.len(), 1, "expected one rect-detected table");
|
||||
let t = &rect_tables[0];
|
||||
assert_eq!(
|
||||
t.columns.len(),
|
||||
6,
|
||||
"doc 128 has a 6-column table; got {} edges: {:?}",
|
||||
t.columns.len(),
|
||||
t.columns
|
||||
);
|
||||
assert!(
|
||||
t.rows.len() >= 14 && t.rows.len() <= 17,
|
||||
"row count drift: {}",
|
||||
t.rows.len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_crop_px_bbox_is_plausible_bounds() {
|
||||
let crop = [10.0, 20.0, 110.0, 220.0];
|
||||
@@ -1775,36 +1815,357 @@ pub fn extract_tables_with_structure_mem(
|
||||
///
|
||||
/// `fallback_reason` is `None` when the TSR-hybrid path produced the
|
||||
/// markdown directly; `Some(<short identifier>)` when stage 1's quality
|
||||
/// check fired and the heuristic `extract_tables_in_regions_mem` was
|
||||
/// substituted instead. The reason string is stable enough to use as a
|
||||
/// metric label (e.g. `phantom_empty_row`, `multi_row_in_cell`).
|
||||
/// check fired and either in-place expansion or the heuristic fallback
|
||||
/// produced the output. The reason string is stable enough to use as a
|
||||
/// metric label (e.g. `multi_row_in_cell_expanded`, `phantom_empty_row`).
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct TableExtractionResult {
|
||||
pub markdown: String,
|
||||
pub fallback_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
enum TsrQualityIssue {
|
||||
PhantomEmptyRow,
|
||||
MultiRowInCell {
|
||||
expanded_cells: Option<Vec<tables::StructuredCell>>,
|
||||
},
|
||||
}
|
||||
|
||||
impl TsrQualityIssue {
|
||||
fn reason(&self) -> &'static str {
|
||||
match self {
|
||||
Self::PhantomEmptyRow => "phantom_empty_row",
|
||||
Self::MultiRowInCell { .. } => "multi_row_in_cell",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct TsrCellTextLine {
|
||||
center_y: f32,
|
||||
half_height: f32,
|
||||
items: Vec<TextItem>,
|
||||
}
|
||||
|
||||
impl TsrCellTextLine {
|
||||
fn new(item: TextItem) -> Self {
|
||||
let center_y = item.y + item.height * 0.5;
|
||||
let half_height = (item.height * 0.5).max(2.5);
|
||||
Self {
|
||||
center_y,
|
||||
half_height,
|
||||
items: vec![item],
|
||||
}
|
||||
}
|
||||
|
||||
fn add(&mut self, item: TextItem) {
|
||||
let center_y = item.y + item.height * 0.5;
|
||||
let existing = self.items.len() as f32;
|
||||
self.center_y = (self.center_y * existing + center_y) / (existing + 1.0);
|
||||
self.half_height = self.half_height.max((item.height * 0.5).max(2.5));
|
||||
self.items.push(item);
|
||||
}
|
||||
|
||||
fn bottom_y(&self) -> f32 {
|
||||
self.items
|
||||
.iter()
|
||||
.map(|item| item.y)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct TsrRowExpansion {
|
||||
bands: Vec<f32>,
|
||||
tolerance: f32,
|
||||
}
|
||||
|
||||
fn collect_items_in_tsr_cell(
|
||||
items: &[TextItem],
|
||||
cell: &tables::StructuredCell,
|
||||
page_height: f32,
|
||||
coord_space: RegionCoordSpace,
|
||||
) -> Vec<TextItem> {
|
||||
let [x1, y1, x2, y2] = cell.page_pt_bbox;
|
||||
if x1 >= x2 || y1 >= y2 {
|
||||
return Vec::new();
|
||||
}
|
||||
let bounds = region_bounds(x1, y1, x2, y2, page_height, coord_space);
|
||||
items
|
||||
.iter()
|
||||
.filter(|item| !item.text.trim().is_empty() && tsr_region_contains_item(item, bounds))
|
||||
.cloned()
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn cluster_tsr_cell_text_lines(mut items: Vec<TextItem>) -> Vec<TsrCellTextLine> {
|
||||
if items.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
items.sort_by(|a, b| {
|
||||
let ay = a.y + a.height * 0.5;
|
||||
let by = b.y + b.height * 0.5;
|
||||
by.total_cmp(&ay).then(a.x.total_cmp(&b.x))
|
||||
});
|
||||
|
||||
let mut lines: Vec<TsrCellTextLine> = Vec::new();
|
||||
for item in items {
|
||||
let item_top = item.y + item.height;
|
||||
let item_half_height = (item.height * 0.5).max(2.5);
|
||||
if let Some(last) = lines.last_mut() {
|
||||
let gap = last.bottom_y() - item_top;
|
||||
if gap <= last.half_height.max(item_half_height) {
|
||||
last.add(item);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
lines.push(TsrCellTextLine::new(item));
|
||||
}
|
||||
|
||||
lines
|
||||
}
|
||||
|
||||
fn build_tsr_row_expansion(
|
||||
row_cells: &[usize],
|
||||
cells: &[tables::StructuredCell],
|
||||
cell_lines: &[Vec<TsrCellTextLine>],
|
||||
) -> Option<TsrRowExpansion> {
|
||||
if row_cells.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let has_spanning_cell = row_cells
|
||||
.iter()
|
||||
.any(|&idx| cells[idx].rowspan > 1 || cells[idx].colspan > 1);
|
||||
if has_spanning_cell {
|
||||
return None;
|
||||
}
|
||||
|
||||
let multiline_cells = row_cells
|
||||
.iter()
|
||||
.filter(|&&idx| cell_lines[idx].len() >= 2)
|
||||
.count();
|
||||
if row_cells.len() >= 2 && multiline_cells < 2 {
|
||||
return None;
|
||||
}
|
||||
if row_cells.len() == 1 && multiline_cells == 0 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut centers: Vec<(f32, f32)> = row_cells
|
||||
.iter()
|
||||
.flat_map(|&idx| {
|
||||
cell_lines[idx]
|
||||
.iter()
|
||||
.map(|line| (line.center_y, line.half_height))
|
||||
})
|
||||
.collect();
|
||||
if centers.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
centers.sort_by(|a, b| b.0.total_cmp(&a.0));
|
||||
|
||||
let mut half_heights: Vec<f32> = centers.iter().map(|(_, h)| *h).collect();
|
||||
half_heights.sort_by(|a, b| a.total_cmp(b));
|
||||
let tolerance = (half_heights[half_heights.len() / 2] * 0.8).max(3.0);
|
||||
|
||||
let mut bands: Vec<(f32, usize)> = Vec::new();
|
||||
for (center, _) in centers {
|
||||
if let Some((band_center, count)) = bands
|
||||
.iter_mut()
|
||||
.find(|(band_center, _)| (*band_center - center).abs() <= tolerance)
|
||||
{
|
||||
*band_center = (*band_center * *count as f32 + center) / (*count as f32 + 1.0);
|
||||
*count += 1;
|
||||
} else {
|
||||
bands.push((center, 1));
|
||||
}
|
||||
}
|
||||
|
||||
// V1 targets the common 1-2 lost-row cases; larger compressions stay on
|
||||
// the existing heuristic fallback path until we have evidence to broaden it.
|
||||
if !(2..=4).contains(&bands.len()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let min_support = if row_cells.len() >= 2 { 2 } else { 1 };
|
||||
let supported = bands.iter().all(|(band, _)| {
|
||||
row_cells
|
||||
.iter()
|
||||
.filter(|&&idx| {
|
||||
cell_lines[idx]
|
||||
.iter()
|
||||
.any(|line| (line.center_y - *band).abs() <= tolerance)
|
||||
})
|
||||
.count()
|
||||
>= min_support
|
||||
});
|
||||
if !supported {
|
||||
return None;
|
||||
}
|
||||
|
||||
bands.sort_by(|a, b| b.0.total_cmp(&a.0));
|
||||
Some(TsrRowExpansion {
|
||||
bands: bands.into_iter().map(|(center, _)| center).collect(),
|
||||
tolerance,
|
||||
})
|
||||
}
|
||||
|
||||
fn text_for_tsr_band(
|
||||
lines: &[TsrCellTextLine],
|
||||
band: f32,
|
||||
tolerance: f32,
|
||||
adaptive_threshold: f32,
|
||||
) -> String {
|
||||
let mut matched: Vec<TextItem> = lines
|
||||
.iter()
|
||||
.filter(|line| (line.center_y - band).abs() <= tolerance)
|
||||
.flat_map(|line| line.items.iter().cloned())
|
||||
.collect();
|
||||
if matched.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
matched.sort_by(|a, b| b.y.total_cmp(&a.y).then(a.x.total_cmp(&b.x)));
|
||||
collect_text_from_matched_items(matched, adaptive_threshold).replace(['\n', '\r'], " ")
|
||||
}
|
||||
|
||||
fn slice_cell_bbox_for_expanded_row(
|
||||
cell: &tables::StructuredCell,
|
||||
row_idx: usize,
|
||||
row_count: usize,
|
||||
) -> [f32; 4] {
|
||||
let mut bbox = cell.page_pt_bbox;
|
||||
let top = bbox[1].min(bbox[3]);
|
||||
let bottom = bbox[1].max(bbox[3]);
|
||||
let height = bottom - top;
|
||||
if height <= 0.0 || row_count == 0 {
|
||||
return bbox;
|
||||
}
|
||||
let step = height / row_count as f32;
|
||||
bbox[1] = top + step * row_idx as f32;
|
||||
bbox[3] = if row_idx + 1 == row_count {
|
||||
bottom
|
||||
} else {
|
||||
top + step * (row_idx + 1) as f32
|
||||
};
|
||||
bbox
|
||||
}
|
||||
|
||||
fn try_expand_multi_row_cells(
|
||||
cells: &[tables::StructuredCell],
|
||||
items: &[TextItem],
|
||||
page_height: f32,
|
||||
coord_space: RegionCoordSpace,
|
||||
adaptive_threshold: f32,
|
||||
) -> Option<Vec<tables::StructuredCell>> {
|
||||
if cells.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let cell_lines: Vec<Vec<TsrCellTextLine>> = cells
|
||||
.iter()
|
||||
.map(|cell| {
|
||||
if cell.rowspan > 1 {
|
||||
Vec::new()
|
||||
} else {
|
||||
cluster_tsr_cell_text_lines(collect_items_in_tsr_cell(
|
||||
items,
|
||||
cell,
|
||||
page_height,
|
||||
coord_space,
|
||||
))
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut cells_by_row: BTreeMap<usize, Vec<usize>> = BTreeMap::new();
|
||||
for (idx, cell) in cells.iter().enumerate() {
|
||||
cells_by_row.entry(cell.row).or_default().push(idx);
|
||||
}
|
||||
|
||||
let mut expansions: HashMap<usize, TsrRowExpansion> = HashMap::new();
|
||||
for (&row, row_cells) in &cells_by_row {
|
||||
let covered_by_rowspan = cells
|
||||
.iter()
|
||||
.any(|cell| cell.rowspan > 1 && cell.row <= row && row < cell.row + cell.rowspan);
|
||||
if covered_by_rowspan {
|
||||
continue;
|
||||
}
|
||||
if let Some(expansion) = build_tsr_row_expansion(row_cells, cells, &cell_lines) {
|
||||
expansions.insert(row, expansion);
|
||||
}
|
||||
}
|
||||
|
||||
if expansions.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut expanded = Vec::with_capacity(cells.len() + expansions.len());
|
||||
let mut row_shift = 0usize;
|
||||
for (&row, row_cells) in &cells_by_row {
|
||||
if let Some(expansion) = expansions.get(&row) {
|
||||
for (band_idx, band) in expansion.bands.iter().enumerate() {
|
||||
for &cell_idx in row_cells {
|
||||
let mut cell = cells[cell_idx].clone();
|
||||
cell.row = row + row_shift + band_idx;
|
||||
cell.rowspan = 1;
|
||||
cell.text = text_for_tsr_band(
|
||||
&cell_lines[cell_idx],
|
||||
*band,
|
||||
expansion.tolerance,
|
||||
adaptive_threshold,
|
||||
);
|
||||
cell.page_pt_bbox =
|
||||
slice_cell_bbox_for_expanded_row(&cell, band_idx, expansion.bands.len());
|
||||
expanded.push(cell);
|
||||
}
|
||||
}
|
||||
row_shift += expansion.bands.len() - 1;
|
||||
} else {
|
||||
for &cell_idx in row_cells {
|
||||
let mut cell = cells[cell_idx].clone();
|
||||
cell.row += row_shift;
|
||||
expanded.push(cell);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let original_rows = cells
|
||||
.iter()
|
||||
.map(|cell| cell.row + cell.rowspan.max(1))
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
let expanded_rows = expanded
|
||||
.iter()
|
||||
.map(|cell| cell.row + cell.rowspan.max(1))
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
(expanded_rows > original_rows).then_some(expanded)
|
||||
}
|
||||
|
||||
/// Detect quality issues in the TSR-hybrid output for a single input.
|
||||
///
|
||||
/// Returns `Some(reason)` if the cells look like they reflect a known
|
||||
/// SLANet detection pathology that the heuristic table extractor would
|
||||
/// likely handle better. Reasons (also used as metric labels):
|
||||
/// Returns `Some(issue)` if the cells look like they reflect a known
|
||||
/// SLANet detection pathology. Reasons (also used as metric labels):
|
||||
///
|
||||
/// * `phantom_empty_row` — a row whose every cell is empty, surrounded
|
||||
/// above and below by rows with content. SLANet sometimes emits an
|
||||
/// extra row that doesn't correspond to any visible PDF row.
|
||||
/// * `multi_row_in_cell` — at least one `rowspan==1` cell encloses
|
||||
/// PDF text items that cluster into two distinct visual lines
|
||||
/// * `multi_row_in_cell` — at least one non-label `rowspan==1` cell
|
||||
/// encloses PDF text items that cluster into two distinct visual lines
|
||||
/// separated by a whitespace gap larger than the line height. Cells
|
||||
/// declared as `rowspan>1` are excluded since they are *expected*
|
||||
/// to span multiple lines. SLANet's row under-detection on
|
||||
/// tightly-packed tables produces the rowspan==1-but-multi-line
|
||||
/// pattern (the FNBO failure mode).
|
||||
/// to span multiple lines. First-row/first-column wraps are ignored
|
||||
/// unless the in-place row expansion has enough support to repair them,
|
||||
/// because those are often legitimate wrapped headers or row labels.
|
||||
/// SLANet's row under-detection on tightly-packed tables produces the
|
||||
/// rowspan==1-but-multi-line pattern (the FNBO failure mode).
|
||||
fn detect_tsr_quality_issue(
|
||||
buffer: &[u8],
|
||||
input: &TsrTableInput,
|
||||
cells: &[tables::StructuredCell],
|
||||
) -> Result<Option<String>, PdfError> {
|
||||
) -> Result<Option<TsrQualityIssue>, PdfError> {
|
||||
if cells.is_empty() {
|
||||
return Ok(None);
|
||||
}
|
||||
@@ -1820,7 +2181,7 @@ fn detect_tsr_quality_issue(
|
||||
}
|
||||
for r in 1..max_row {
|
||||
if !row_has_content[r] && row_has_content[r - 1] && row_has_content[r + 1] {
|
||||
return Ok(Some("phantom_empty_row".to_string()));
|
||||
return Ok(Some(TsrQualityIssue::PhantomEmptyRow));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1849,12 +2210,16 @@ fn detect_tsr_quality_issue(
|
||||
&font_cmaps,
|
||||
false,
|
||||
)?;
|
||||
let _ = text_utils::fix_letterspaced_items(&mut items);
|
||||
let adaptive_threshold = text_utils::fix_letterspaced_items(&mut items);
|
||||
let coords = if coords_rotated {
|
||||
RegionCoordSpace::Rotated90Ccw
|
||||
} else {
|
||||
RegionCoordSpace::Standard
|
||||
};
|
||||
let expanded_cells =
|
||||
try_expand_multi_row_cells(cells, &items, page_h, coords, adaptive_threshold);
|
||||
let first_row = cells.iter().map(|cell| cell.row).min().unwrap_or(0);
|
||||
let first_col = cells.iter().map(|cell| cell.col).min().unwrap_or(0);
|
||||
|
||||
for cell in cells {
|
||||
// rowspan>1 cells are intentionally multi-line — skip them.
|
||||
@@ -1864,72 +2229,45 @@ fn detect_tsr_quality_issue(
|
||||
if cell.text.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
let [x1, y1, x2, y2] = cell.page_pt_bbox;
|
||||
if x1 >= x2 || y1 >= y2 {
|
||||
continue;
|
||||
}
|
||||
let bounds = region_bounds(x1, y1, x2, y2, page_h, coords);
|
||||
|
||||
// Collect the items inside this cell, with their y-centers and
|
||||
// half-heights so we can cluster them into visual lines.
|
||||
let mut cell_items: Vec<(f32, f32)> = Vec::new();
|
||||
for item in &items {
|
||||
if tsr_region_contains_item(item, bounds) {
|
||||
let cy = item.y + item.height * 0.5;
|
||||
let half_h = (item.height * 0.5).max(2.5);
|
||||
cell_items.push((cy, half_h));
|
||||
}
|
||||
}
|
||||
let cell_items = collect_items_in_tsr_cell(&items, cell, page_h, coords);
|
||||
if cell_items.len() < 2 {
|
||||
continue;
|
||||
}
|
||||
// Sort by y-center descending (top-of-page first in PDF native
|
||||
// coords where y grows upward) — direction doesn't matter, we
|
||||
// just need consecutive items to be neighbors in the sort.
|
||||
cell_items.sort_by(|a, b| b.0.total_cmp(&a.0));
|
||||
|
||||
// Walk pairs and see if there's a real whitespace gap between
|
||||
// any two adjacent items — defined as their bounding-box edges
|
||||
// separated by more than half a line height. This rules out
|
||||
// tall glyphs / superscripts / accents on a single visual line.
|
||||
let max_half_h = cell_items
|
||||
.iter()
|
||||
.map(|(_, h)| *h)
|
||||
.fold(0f32, f32::max)
|
||||
.max(2.5);
|
||||
let gap_threshold = max_half_h; // ≈ half a line height
|
||||
let mut found_gap = false;
|
||||
for w in cell_items.windows(2) {
|
||||
let (cy_a, h_a) = w[0];
|
||||
let (cy_b, h_b) = w[1];
|
||||
// Gap = distance between the bottom of the upper item and
|
||||
// the top of the lower item, measured in PDF-native coords
|
||||
// (y grows upward, so the upper item has the larger cy).
|
||||
let upper_bottom = cy_a - h_a;
|
||||
let lower_top = cy_b + h_b;
|
||||
let gap = upper_bottom - lower_top;
|
||||
if gap > gap_threshold {
|
||||
found_gap = true;
|
||||
break;
|
||||
}
|
||||
if cluster_tsr_cell_text_lines(cell_items).len() < 2 {
|
||||
continue;
|
||||
}
|
||||
if found_gap {
|
||||
return Ok(Some("multi_row_in_cell".to_string()));
|
||||
if expanded_cells.is_some() {
|
||||
return Ok(Some(TsrQualityIssue::MultiRowInCell { expanded_cells }));
|
||||
}
|
||||
if !is_wrapped_tsr_label_cell(cell, first_row, first_col) {
|
||||
return Ok(Some(TsrQualityIssue::MultiRowInCell {
|
||||
expanded_cells: None,
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
fn is_wrapped_tsr_label_cell(
|
||||
cell: &tables::StructuredCell,
|
||||
first_row: usize,
|
||||
first_col: usize,
|
||||
) -> bool {
|
||||
cell.is_header || cell.row == first_row || cell.col == first_col
|
||||
}
|
||||
|
||||
/// Auto-fallback variant of [`extract_tables_with_structure_mem`]:
|
||||
/// runs the TSR-hybrid path, checks the resulting cells for known
|
||||
/// SLANet detection pathologies (phantom rows, multi-row-in-cell text),
|
||||
/// and falls back to the heuristic [`extract_tables_in_regions_mem`]
|
||||
/// for any input where the TSR path looks compromised.
|
||||
///
|
||||
/// On clean inputs this is identical to the markdown variant.
|
||||
/// On flagged inputs the heuristic markdown replaces the TSR markdown
|
||||
/// and the result's `fallback_reason` is set to the diagnostic label.
|
||||
/// On clean inputs this is identical to the markdown variant. On
|
||||
/// `multi_row_in_cell`, the wrapper first tries to expand over-stuffed
|
||||
/// rows in place; if that cannot produce a usable table, the heuristic
|
||||
/// markdown replaces the TSR markdown and `fallback_reason` is set to
|
||||
/// the diagnostic label.
|
||||
///
|
||||
/// Two failure modes are guarded against per-input:
|
||||
///
|
||||
@@ -1939,6 +2277,9 @@ fn detect_tsr_quality_issue(
|
||||
/// `_heuristic_empty` (e.g. `multi_row_in_cell_heuristic_empty`).
|
||||
/// This avoids replacing a usable wrong-but-non-empty TSR output
|
||||
/// with literally nothing.
|
||||
/// * **Expanded multi-row cells**: when in-place recovery succeeds, the
|
||||
/// result is labeled `multi_row_in_cell_expanded` and the heuristic is
|
||||
/// not consulted.
|
||||
/// * **Per-input errors**: any failure in detection or heuristic
|
||||
/// extraction for a single input is contained — that input
|
||||
/// returns the raw TSR markdown with `fallback_reason` set to
|
||||
@@ -1983,7 +2324,21 @@ pub fn extract_tables_with_structure_auto_mem(
|
||||
markdown: tsr_md,
|
||||
fallback_reason: None,
|
||||
},
|
||||
Some(reason) => {
|
||||
Some(issue) => {
|
||||
let reason = issue.reason().to_string();
|
||||
if let TsrQualityIssue::MultiRowInCell {
|
||||
expanded_cells: Some(expanded_cells),
|
||||
} = issue
|
||||
{
|
||||
let expanded_md = tables::cells_to_markdown(&expanded_cells);
|
||||
if !expanded_md.trim().is_empty() {
|
||||
results.push(TableExtractionResult {
|
||||
markdown: expanded_md,
|
||||
fallback_reason: Some("multi_row_in_cell_expanded".to_string()),
|
||||
});
|
||||
continue;
|
||||
}
|
||||
}
|
||||
// Fall back to heuristic on the input's table region.
|
||||
// The crop's PDF-pt bbox IS the table region.
|
||||
let heuristic_md = match extract_tables_in_regions_mem(
|
||||
@@ -3650,6 +4005,77 @@ mod tests {
|
||||
assert!(!cells[2].text.contains("Boardwalk"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multi_row_expansion_splits_overstuffed_tsr_row() {
|
||||
use crate::tables::{cells_to_markdown, StructuredCell};
|
||||
|
||||
let items = vec![
|
||||
test_item("Branch Name", 20.0, 166.0, 55.0, 8.0),
|
||||
test_item("Deposits", 120.0, 166.0, 36.0, 8.0),
|
||||
test_item("Oak Street", 20.0, 136.0, 48.0, 8.0),
|
||||
test_item("100", 120.0, 136.0, 18.0, 8.0),
|
||||
test_item("Boardwalk", 20.0, 116.0, 46.0, 8.0),
|
||||
test_item("200", 120.0, 116.0, 18.0, 8.0),
|
||||
];
|
||||
let cells = vec![
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: true,
|
||||
text: "Branch Name".into(),
|
||||
page_pt_bbox: [10.0, 20.0, 100.0, 40.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: true,
|
||||
text: "Deposits".into(),
|
||||
page_pt_bbox: [110.0, 20.0, 190.0, 40.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "Oak Street Boardwalk".into(),
|
||||
page_pt_bbox: [10.0, 40.0, 100.0, 100.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "100 200".into(),
|
||||
page_pt_bbox: [110.0, 40.0, 190.0, 100.0],
|
||||
},
|
||||
];
|
||||
|
||||
let expanded =
|
||||
try_expand_multi_row_cells(&cells, &items, 200.0, RegionCoordSpace::Standard, 0.10)
|
||||
.expect("overstuffed data row should expand");
|
||||
let md = cells_to_markdown(&expanded);
|
||||
|
||||
assert_eq!(expanded.iter().map(|c| c.row).max().unwrap() + 1, 3);
|
||||
assert!(
|
||||
md.contains("|Oak Street|100|"),
|
||||
"missing first data row: {md}"
|
||||
);
|
||||
assert!(
|
||||
md.contains("|Boardwalk|200|"),
|
||||
"missing second data row: {md}"
|
||||
);
|
||||
assert!(
|
||||
!md.contains("Oak Street Boardwalk"),
|
||||
"compressed cell text should be replaced: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tsr_assignment_caps_uses_median_geometry() {
|
||||
use crate::tables::StructuredCell;
|
||||
|
||||
+59
-15
@@ -1587,25 +1587,69 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
return None;
|
||||
}
|
||||
|
||||
// Derive columns from text X-position clustering
|
||||
// Derive columns from text X-position clustering, but prefer rect
|
||||
// X-edges when they already provide a tighter scaffold. Some PDFs draw
|
||||
// only the row-index cells in the body plus a full header row; that is
|
||||
// not dense enough for `try_build_grid`, but the header rects still define
|
||||
// the real columns. Text starts inside wide cells can otherwise split the
|
||||
// table into spurious sub-columns.
|
||||
let columns = cluster_x_positions(&page_items, 15.0);
|
||||
if columns.len() < 2 {
|
||||
let text_col_edges = if columns.len() >= 2 {
|
||||
let mut edges: Vec<f32> = Vec::with_capacity(columns.len() + 1);
|
||||
let min_x = page_items.iter().map(|(_, i)| i.x).reduce(f32::min)?;
|
||||
edges.push(min_x - 5.0);
|
||||
for pair in columns.windows(2) {
|
||||
edges.push((pair[0] + pair[1]) / 2.0);
|
||||
}
|
||||
let max_x_right = page_items
|
||||
.iter()
|
||||
.map(|(_, i)| i.x + i.width)
|
||||
.reduce(f32::max)?;
|
||||
edges.push(max_x_right + 5.0);
|
||||
Some(edges)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let rect_col_edges = {
|
||||
let mut x_vals = Vec::with_capacity(content_rects.len() * 2);
|
||||
for &&(x, _, w, _) in &content_rects {
|
||||
x_vals.push(x);
|
||||
x_vals.push(x + w);
|
||||
}
|
||||
let mut edges = snap_edges(&x_vals, 6.0);
|
||||
edges.sort_by(|a, b| a.total_cmp(b));
|
||||
if (3..=26).contains(&edges.len()) {
|
||||
Some(edges)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
let col_edges = match (rect_col_edges, text_col_edges) {
|
||||
(Some(rect_edges), Some(text_edges)) if rect_edges.len() <= text_edges.len() => {
|
||||
debug!(
|
||||
" cell-rect using {} rect-derived columns over {} text clusters",
|
||||
rect_edges.len() - 1,
|
||||
text_edges.len() - 1
|
||||
);
|
||||
rect_edges
|
||||
}
|
||||
(_, Some(text_edges)) => text_edges,
|
||||
(Some(rect_edges), None) => rect_edges,
|
||||
(None, None) => {
|
||||
debug!(
|
||||
" cell-rect rejected: only {} columns from text clustering",
|
||||
columns.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
};
|
||||
|
||||
if col_edges.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Build column edges
|
||||
let mut col_edges: Vec<f32> = Vec::with_capacity(columns.len() + 1);
|
||||
let min_x = page_items.iter().map(|(_, i)| i.x).reduce(f32::min)?;
|
||||
col_edges.push(min_x - 5.0);
|
||||
for pair in columns.windows(2) {
|
||||
col_edges.push((pair[0] + pair[1]) / 2.0);
|
||||
}
|
||||
let max_x_right = page_items
|
||||
.iter()
|
||||
.map(|(_, i)| i.x + i.width)
|
||||
.reduce(f32::max)?;
|
||||
col_edges.push(max_x_right + 5.0);
|
||||
|
||||
let num_cols = col_edges.len() - 1;
|
||||
let num_rows = row_edges.len() - 1;
|
||||
|
||||
|
||||
BIN
Binary file not shown.
BIN
Binary file not shown.
+221
-18
@@ -1812,6 +1812,93 @@ fn synthetic_vector_grid_pdf(two_tables: bool) -> Vec<u8> {
|
||||
bytes
|
||||
}
|
||||
|
||||
fn synthetic_vector_grid_three_row_pdf() -> Vec<u8> {
|
||||
use lopdf::content::{Content, Operation};
|
||||
use lopdf::{dictionary, Document, Object, Stream};
|
||||
|
||||
let mut doc = Document::with_version("1.5");
|
||||
let pages_id = doc.new_object_id();
|
||||
let page_id = doc.new_object_id();
|
||||
let font_id = doc.new_object_id();
|
||||
let content_id = doc.new_object_id();
|
||||
|
||||
doc.objects.insert(
|
||||
font_id,
|
||||
dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
}
|
||||
.into(),
|
||||
);
|
||||
|
||||
let mut operations = Vec::new();
|
||||
for y in [740, 710, 680, 650] {
|
||||
operations.push(Operation::new("m", vec![50.into(), y.into()]));
|
||||
operations.push(Operation::new("l", vec![210.into(), y.into()]));
|
||||
}
|
||||
for x in [50, 130, 210] {
|
||||
operations.push(Operation::new("m", vec![x.into(), 650.into()]));
|
||||
operations.push(Operation::new("l", vec![x.into(), 740.into()]));
|
||||
}
|
||||
operations.push(Operation::new("S", vec![]));
|
||||
|
||||
operations.push(Operation::new("BT", vec![]));
|
||||
operations.push(Operation::new("Tf", vec!["F1".into(), 10.into()]));
|
||||
for (x, y, text) in [
|
||||
(70, 724, "Branch"),
|
||||
(150, 724, "Deposits"),
|
||||
(70, 694, "Oak"),
|
||||
(150, 694, "100"),
|
||||
(70, 664, "Boardwalk"),
|
||||
(150, 664, "200"),
|
||||
] {
|
||||
operations.push(Operation::new(
|
||||
"Tm",
|
||||
vec![1.into(), 0.into(), 0.into(), 1.into(), x.into(), y.into()],
|
||||
));
|
||||
operations.push(Operation::new("Tj", vec![Object::string_literal(text)]));
|
||||
}
|
||||
operations.push(Operation::new("ET", vec![]));
|
||||
|
||||
let content = Content { operations }.encode().unwrap();
|
||||
doc.objects
|
||||
.insert(content_id, Stream::new(dictionary! {}, content).into());
|
||||
doc.objects.insert(
|
||||
page_id,
|
||||
dictionary! {
|
||||
"Type" => "Page",
|
||||
"Parent" => pages_id,
|
||||
"MediaBox" => vec![0.into(), 0.into(), 300.into(), 800.into()],
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => font_id,
|
||||
},
|
||||
},
|
||||
"Contents" => content_id,
|
||||
}
|
||||
.into(),
|
||||
);
|
||||
doc.objects.insert(
|
||||
pages_id,
|
||||
dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Kids" => vec![page_id.into()],
|
||||
"Count" => 1,
|
||||
}
|
||||
.into(),
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => pages_id,
|
||||
});
|
||||
doc.trailer.set("Root", catalog_id);
|
||||
|
||||
let mut bytes = Vec::new();
|
||||
doc.save_to(&mut bytes).unwrap();
|
||||
bytes
|
||||
}
|
||||
|
||||
fn assert_close(actual: f32, expected: f32) {
|
||||
assert!(
|
||||
(actual - expected).abs() < 0.75,
|
||||
@@ -2319,7 +2406,7 @@ fn test_auto_passes_through_clean_tsr_output() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_auto_falls_back_on_multi_row_in_cell() {
|
||||
fn test_auto_expands_multi_row_in_cell() {
|
||||
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
|
||||
|
||||
let buf = synthetic_dense_table_pdf();
|
||||
@@ -2370,16 +2457,134 @@ fn test_auto_falls_back_on_multi_row_in_cell() {
|
||||
assert_eq!(results.len(), 1);
|
||||
assert_eq!(
|
||||
results[0].fallback_reason.as_deref(),
|
||||
Some("multi_row_in_cell"),
|
||||
"expected multi_row_in_cell fallback, got {:?}",
|
||||
Some("multi_row_in_cell_expanded"),
|
||||
"expected multi_row_in_cell_expanded, got {:?}",
|
||||
results[0].fallback_reason
|
||||
);
|
||||
// The heuristic-fallback markdown should preserve all three PDF rows.
|
||||
// The in-place expansion should preserve all three PDF rows.
|
||||
let md = &results[0].markdown;
|
||||
assert!(md.contains("Oak Street"), "missing Oak Street: {md}");
|
||||
assert!(md.contains("Boardwalk"), "missing Boardwalk: {md}");
|
||||
assert!(md.contains("100"), "missing 100: {md}");
|
||||
assert!(md.contains("200"), "missing 200: {md}");
|
||||
assert!(
|
||||
!md.contains("Oak Street Boardwalk"),
|
||||
"rows should not remain compressed: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_auto_expands_under_counted_vector_grid_rows() {
|
||||
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
|
||||
|
||||
let buf = synthetic_vector_grid_three_row_pdf();
|
||||
let tokens: Vec<String> = [
|
||||
"<table>",
|
||||
"<thead>",
|
||||
"<tr>",
|
||||
"<th></th>",
|
||||
"<th></th>",
|
||||
"</tr>",
|
||||
"</thead>",
|
||||
"<tbody>",
|
||||
"<tr>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"</tbody>",
|
||||
"</table>",
|
||||
]
|
||||
.into_iter()
|
||||
.map(String::from)
|
||||
.collect();
|
||||
let crop = [50.0, 60.0, 210.0, 150.0];
|
||||
let cell_bboxes = vec![
|
||||
poly(0.0, 0.0, 80.0, 30.0),
|
||||
poly(80.0, 0.0, 160.0, 30.0),
|
||||
poly(0.0, 30.0, 80.0, 90.0),
|
||||
poly(80.0, 30.0, 160.0, 90.0),
|
||||
];
|
||||
|
||||
let results = extract_tables_with_structure_auto_mem(
|
||||
&buf,
|
||||
&[TsrTableInput {
|
||||
page: 0,
|
||||
crop_pdf_pt_bbox: crop,
|
||||
render_dpi: 72.0,
|
||||
structure_tokens: tokens,
|
||||
cell_bboxes,
|
||||
}],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
assert_eq!(
|
||||
results[0].fallback_reason.as_deref(),
|
||||
Some("multi_row_in_cell_expanded")
|
||||
);
|
||||
let md = &results[0].markdown;
|
||||
assert!(md.contains("|Branch|Deposits|"), "missing header: {md}");
|
||||
assert!(md.contains("|Oak|100|"), "missing row 1: {md}");
|
||||
assert!(md.contains("|Boardwalk|200|"), "missing row 2: {md}");
|
||||
assert!(
|
||||
!md.contains("Oak Boardwalk"),
|
||||
"rows stayed compressed: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_auto_keeps_wrapped_header_vector_grid_doc51() {
|
||||
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
|
||||
|
||||
let buf = std::fs::read("tests/fixtures/government_positions_women.pdf").unwrap();
|
||||
let crop = [0.0, 0.0, 612.0, 792.0];
|
||||
let grid = detect_vector_grid_in_region_mem(&buf, 0, crop, 200.0)
|
||||
.unwrap()
|
||||
.expect("expected doc 51 vector grid");
|
||||
assert_eq!(
|
||||
grid.cell_bboxes.len(),
|
||||
36,
|
||||
"doc 51 should have a 9x4 vector grid"
|
||||
);
|
||||
|
||||
let results = extract_tables_with_structure_auto_mem(
|
||||
&buf,
|
||||
&[TsrTableInput {
|
||||
page: 0,
|
||||
crop_pdf_pt_bbox: crop,
|
||||
render_dpi: 200.0,
|
||||
structure_tokens: grid.structure_tokens,
|
||||
cell_bboxes: grid.cell_bboxes,
|
||||
}],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
let r = &results[0];
|
||||
assert!(
|
||||
r.fallback_reason.is_none(),
|
||||
"wrapped header/label text should not trigger heuristic fallback: {:?}\n{}",
|
||||
r.fallback_reason,
|
||||
r.markdown
|
||||
);
|
||||
let md = &r.markdown;
|
||||
assert!(md.contains("Government Position"), "missing header: {md}");
|
||||
assert!(
|
||||
md.contains("Aquino Administration"),
|
||||
"missing Aquino header: {md}"
|
||||
);
|
||||
assert!(
|
||||
md.contains("Ramos Administration"),
|
||||
"missing Ramos header: {md}"
|
||||
);
|
||||
assert!(
|
||||
md.contains("City Municipal Councilor"),
|
||||
"row label was truncated: {md}"
|
||||
);
|
||||
assert!(
|
||||
!md.contains("|Position||Administration"),
|
||||
"heuristic fallback split the header row: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2456,15 +2661,14 @@ fn test_auto_does_not_fire_on_legit_rowspan_cell() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_auto_keeps_tsr_markdown_when_heuristic_returns_empty() {
|
||||
fn test_auto_expands_when_heuristic_region_is_empty() {
|
||||
use pdf_inspector::{extract_tables_with_structure_auto_mem, TsrTableInput};
|
||||
|
||||
let buf = synthetic_dense_table_pdf();
|
||||
// Same shape as the multi_row_in_cell regression — a tall data cell
|
||||
// that catches Oak Street + Boardwalk. But the crop bbox we pass
|
||||
// points at a strip of the page that has NO text items, so the
|
||||
// heuristic's region will be empty when it tries to extract there.
|
||||
// The auto wrapper must keep the TSR markdown rather than ship "".
|
||||
// that catches Oak Street + Boardwalk. The crop bbox we pass points
|
||||
// at a strip of the page that has NO text items, so the old heuristic
|
||||
// fallback would be empty. Expansion uses the cell bboxes directly.
|
||||
let tokens: Vec<String> = [
|
||||
"<table>",
|
||||
"<thead>",
|
||||
@@ -2510,20 +2714,19 @@ fn test_auto_keeps_tsr_markdown_when_heuristic_returns_empty() {
|
||||
let r = &results[0];
|
||||
assert_eq!(
|
||||
r.fallback_reason.as_deref(),
|
||||
Some("multi_row_in_cell_heuristic_empty"),
|
||||
"expected _heuristic_empty suffix, got {:?}",
|
||||
Some("multi_row_in_cell_expanded"),
|
||||
"expected expansion despite empty heuristic region, got {:?}",
|
||||
r.fallback_reason,
|
||||
);
|
||||
// TSR markdown should be preserved — non-empty, contains the cell
|
||||
// text we know was assigned by the TSR path.
|
||||
assert!(
|
||||
!r.markdown.trim().is_empty(),
|
||||
"expected TSR markdown to be preserved, got empty",
|
||||
r.markdown.contains("|Oak Street|100|"),
|
||||
"missing row 1: {}",
|
||||
r.markdown
|
||||
);
|
||||
assert!(
|
||||
r.markdown.contains("Oak Street") || r.markdown.contains("Boardwalk"),
|
||||
"expected TSR markdown to contain at least one row, got: {}",
|
||||
r.markdown,
|
||||
r.markdown.contains("|Boardwalk|200|"),
|
||||
"missing row 2: {}",
|
||||
r.markdown
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user