Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bc25730fb0 | ||
|
|
0c56791b0f | ||
|
|
e8a7529e27 | ||
|
|
5ade93440b | ||
|
|
cbd45e96d9 |
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "@firecrawl/pdf-inspector",
|
"name": "@firecrawl/pdf-inspector",
|
||||||
"version": "1.6.0",
|
"version": "1.6.1",
|
||||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"types": "index.d.ts",
|
"types": "index.d.ts",
|
||||||
|
|||||||
+172
-20
@@ -839,7 +839,7 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
inputs: &[TsrTableInput],
|
inputs: &[TsrTableInput],
|
||||||
) -> Result<Vec<Vec<tables::StructuredCell>>, PdfError> {
|
) -> Result<Vec<Vec<tables::StructuredCell>>, PdfError> {
|
||||||
use tables::structured::{
|
use tables::structured::{
|
||||||
cell_px_to_page_pt, parse_structure, polygon_to_aabb, StructuredCell,
|
cell_px_to_page_pt, normalize_cell_bands, parse_structure, polygon_to_aabb, StructuredCell,
|
||||||
};
|
};
|
||||||
|
|
||||||
validate_pdf_bytes(buffer)?;
|
validate_pdf_bytes(buffer)?;
|
||||||
@@ -906,32 +906,15 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
|
|
||||||
let mut cells: Vec<StructuredCell> = Vec::with_capacity(slots.len());
|
let mut cells: Vec<StructuredCell> = Vec::with_capacity(slots.len());
|
||||||
for slot in &slots {
|
for slot in &slots {
|
||||||
let cell_text;
|
|
||||||
let page_pt_bbox;
|
let page_pt_bbox;
|
||||||
|
|
||||||
if let Some(coords_arr) = input.cell_bboxes.get(slot.bbox_idx) {
|
if let Some(coords_arr) = input.cell_bboxes.get(slot.bbox_idx) {
|
||||||
if let Some(aabb_px) = polygon_to_aabb(coords_arr) {
|
if let Some(aabb_px) = polygon_to_aabb(coords_arr) {
|
||||||
let aabb_pt = cell_px_to_page_pt(aabb_px, input.render_dpi, crop_origin);
|
page_pt_bbox = cell_px_to_page_pt(aabb_px, input.render_dpi, crop_origin);
|
||||||
let raw = collect_text_in_region_with_options(
|
|
||||||
items,
|
|
||||||
aabb_pt[0],
|
|
||||||
aabb_pt[1],
|
|
||||||
aabb_pt[2],
|
|
||||||
aabb_pt[3],
|
|
||||||
page_h,
|
|
||||||
coords,
|
|
||||||
adaptive_threshold,
|
|
||||||
);
|
|
||||||
// Markdown cells must be one line — collapse line breaks
|
|
||||||
// produced by the line-grouping pass.
|
|
||||||
cell_text = raw.replace(['\n', '\r'], " ");
|
|
||||||
page_pt_bbox = aabb_pt;
|
|
||||||
} else {
|
} else {
|
||||||
cell_text = String::new();
|
|
||||||
page_pt_bbox = [0.0, 0.0, 0.0, 0.0];
|
page_pt_bbox = [0.0, 0.0, 0.0, 0.0];
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
cell_text = String::new();
|
|
||||||
page_pt_bbox = [0.0, 0.0, 0.0, 0.0];
|
page_pt_bbox = [0.0, 0.0, 0.0, 0.0];
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -941,11 +924,21 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
rowspan: slot.rowspan,
|
rowspan: slot.rowspan,
|
||||||
colspan: slot.colspan,
|
colspan: slot.colspan,
|
||||||
is_header: slot.is_header,
|
is_header: slot.is_header,
|
||||||
text: cell_text,
|
text: String::new(),
|
||||||
page_pt_bbox,
|
page_pt_bbox,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
normalize_cell_bands(&mut cells);
|
||||||
|
for cell in &mut cells {
|
||||||
|
let [x1, y1, x2, y2] = cell.page_pt_bbox;
|
||||||
|
let raw =
|
||||||
|
collect_text_in_tsr_cell(items, x1, y1, x2, y2, page_h, coords, adaptive_threshold);
|
||||||
|
// Markdown cells must be one line — collapse line breaks produced
|
||||||
|
// by the line-grouping pass.
|
||||||
|
cell.text = raw.replace(['\n', '\r'], " ");
|
||||||
|
}
|
||||||
|
|
||||||
results.push(cells);
|
results.push(cells);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1062,6 +1055,30 @@ fn collect_text_in_region_with_options(
|
|||||||
.filter(|item| region_overlaps_item(item, bounds))
|
.filter(|item| region_overlaps_item(item, bounds))
|
||||||
.cloned()
|
.cloned()
|
||||||
.collect();
|
.collect();
|
||||||
|
collect_text_from_matched_items(matched, adaptive_threshold)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn collect_text_in_tsr_cell(
|
||||||
|
items: &[TextItem],
|
||||||
|
rx1: f32,
|
||||||
|
ry1: f32,
|
||||||
|
rx2: f32,
|
||||||
|
ry2: f32,
|
||||||
|
page_height: f32,
|
||||||
|
coord_space: RegionCoordSpace,
|
||||||
|
adaptive_threshold: f32,
|
||||||
|
) -> String {
|
||||||
|
let bounds = region_bounds(rx1, ry1, rx2, ry2, page_height, coord_space);
|
||||||
|
let matched: Vec<TextItem> = items
|
||||||
|
.iter()
|
||||||
|
.filter(|item| tsr_region_contains_item(item, bounds))
|
||||||
|
.cloned()
|
||||||
|
.collect();
|
||||||
|
collect_text_from_matched_items(matched, adaptive_threshold)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn collect_text_from_matched_items(matched: Vec<TextItem>, adaptive_threshold: f32) -> String {
|
||||||
if matched.is_empty() {
|
if matched.is_empty() {
|
||||||
return String::new();
|
return String::new();
|
||||||
}
|
}
|
||||||
@@ -1163,6 +1180,30 @@ fn region_overlaps_item(item: &TextItem, bounds: RegionBounds) -> bool {
|
|||||||
x_overlap > 0.0 && y_overlap > 0.0
|
x_overlap > 0.0 && y_overlap > 0.0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn tsr_region_contains_item(item: &TextItem, bounds: RegionBounds) -> bool {
|
||||||
|
let item_x_min = item.x;
|
||||||
|
let item_x_max = item.x + text_utils::effective_width(item);
|
||||||
|
let item_y_min = item.y;
|
||||||
|
let item_y_max = item.y + item.height;
|
||||||
|
|
||||||
|
let center_x = (item_x_min + item_x_max) * 0.5;
|
||||||
|
let center_y = (item_y_min + item_y_max) * 0.5;
|
||||||
|
if center_x >= bounds.x_min
|
||||||
|
&& center_x <= bounds.x_max
|
||||||
|
&& center_y >= bounds.y_min
|
||||||
|
&& center_y <= bounds.y_max
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
let x_overlap = (item_x_max.min(bounds.x_max) - item_x_min.max(bounds.x_min)).max(0.0);
|
||||||
|
let y_overlap = (item_y_max.min(bounds.y_max) - item_y_min.max(bounds.y_min)).max(0.0);
|
||||||
|
let item_width = (item_x_max - item_x_min).max(0.1);
|
||||||
|
let item_height = (item_y_max - item_y_min).max(0.1);
|
||||||
|
|
||||||
|
x_overlap / item_width >= 0.6 && y_overlap / item_height >= 0.6
|
||||||
|
}
|
||||||
|
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
// Internal: single-load document pipeline
|
// Internal: single-load document pipeline
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
@@ -2217,6 +2258,24 @@ pub(crate) fn validate_pdf_file<P: AsRef<Path>>(path: P) -> Result<(), PdfError>
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
use crate::types::ItemType;
|
||||||
|
|
||||||
|
fn test_item(text: &str, x: f32, y: f32, width: f32, height: f32) -> TextItem {
|
||||||
|
TextItem {
|
||||||
|
text: text.to_string(),
|
||||||
|
x,
|
||||||
|
y,
|
||||||
|
width,
|
||||||
|
height,
|
||||||
|
font: "Helvetica".to_string(),
|
||||||
|
font_size: height,
|
||||||
|
page: 1,
|
||||||
|
is_bold: false,
|
||||||
|
is_italic: false,
|
||||||
|
item_type: ItemType::Text,
|
||||||
|
mcid: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_detect_encoding_issues_fffd() {
|
fn test_detect_encoding_issues_fffd() {
|
||||||
@@ -2297,4 +2356,97 @@ mod tests {
|
|||||||
"Valid Japanese text should not be flagged as garbage"
|
"Valid Japanese text should not be flagged as garbage"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tsr_text_fill_does_not_pull_neighboring_overlapping_rows() {
|
||||||
|
use crate::tables::structured::normalize_cell_bands;
|
||||||
|
use crate::tables::StructuredCell;
|
||||||
|
|
||||||
|
let items = vec![
|
||||||
|
test_item("Branch Name", 12.0, 88.0, 55.0, 8.0),
|
||||||
|
test_item("Deposits", 112.0, 88.0, 36.0, 8.0),
|
||||||
|
test_item("Oak Street", 12.0, 72.0, 48.0, 8.0),
|
||||||
|
test_item("100", 112.0, 72.0, 18.0, 8.0),
|
||||||
|
test_item("Boardwalk", 12.0, 55.2, 46.0, 8.0),
|
||||||
|
test_item("200", 112.0, 55.2, 18.0, 8.0),
|
||||||
|
];
|
||||||
|
let mut cells = vec![
|
||||||
|
StructuredCell {
|
||||||
|
row: 0,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: true,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [10.0, 100.0, 100.0, 125.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 0,
|
||||||
|
col: 1,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: true,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [100.0, 100.0, 170.0, 125.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 1,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [10.0, 116.0, 100.0, 141.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 1,
|
||||||
|
col: 1,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [100.0, 116.0, 170.0, 141.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 2,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [10.0, 132.8, 100.0, 157.8],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 2,
|
||||||
|
col: 1,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [100.0, 132.8, 170.0, 157.8],
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
normalize_cell_bands(&mut cells);
|
||||||
|
for cell in &mut cells {
|
||||||
|
let [x1, y1, x2, y2] = cell.page_pt_bbox;
|
||||||
|
cell.text = collect_text_in_tsr_cell(
|
||||||
|
&items,
|
||||||
|
x1,
|
||||||
|
y1,
|
||||||
|
x2,
|
||||||
|
y2,
|
||||||
|
200.0,
|
||||||
|
RegionCoordSpace::Standard,
|
||||||
|
0.10,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
assert_eq!(cells[0].text, "Branch Name");
|
||||||
|
assert_eq!(cells[2].text, "Oak Street");
|
||||||
|
assert_eq!(cells[4].text, "Boardwalk");
|
||||||
|
assert!(!cells[0].text.contains("Oak Street"));
|
||||||
|
assert!(!cells[2].text.contains("Branch Name"));
|
||||||
|
assert!(!cells[2].text.contains("Boardwalk"));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+235
-1
@@ -12,7 +12,7 @@
|
|||||||
//! Cell text is supplied separately by the caller (typically by overlap-
|
//! Cell text is supplied separately by the caller (typically by overlap-
|
||||||
//! testing PDF text items against each cell's page-PDF-pt bbox).
|
//! testing PDF text items against each cell's page-PDF-pt bbox).
|
||||||
|
|
||||||
use std::collections::HashSet;
|
use std::collections::{HashMap, HashSet};
|
||||||
|
|
||||||
/// A single resolved cell, with both structural metadata and its bbox in
|
/// A single resolved cell, with both structural metadata and its bbox in
|
||||||
/// page PDF-points (top-left origin).
|
/// page PDF-points (top-left origin).
|
||||||
@@ -219,6 +219,144 @@ pub(crate) fn cell_px_to_page_pt(
|
|||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Refine TSR cell bboxes into non-overlapping row/column bands.
|
||||||
|
///
|
||||||
|
/// SLANet-style bboxes are often plausible but too tall on dense borderless
|
||||||
|
/// tables. Native PDF text assignment is more reliable when each parsed row
|
||||||
|
/// owns the band between neighboring row centers instead of the full model box.
|
||||||
|
pub(crate) fn normalize_cell_bands(cells: &mut [StructuredCell]) {
|
||||||
|
if cells.len() < 2 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
let row_bands = derive_axis_bands(cells, Axis::Y);
|
||||||
|
let col_bands = derive_axis_bands(cells, Axis::X);
|
||||||
|
|
||||||
|
for cell in cells {
|
||||||
|
let row_end = cell.row + cell.rowspan.max(1).saturating_sub(1);
|
||||||
|
if let (Some(&(y1, _)), Some(&(_, y2))) =
|
||||||
|
(row_bands.get(&cell.row), row_bands.get(&row_end))
|
||||||
|
{
|
||||||
|
let clamped_y1 = cell.page_pt_bbox[1].max(y1);
|
||||||
|
let clamped_y2 = cell.page_pt_bbox[3].min(y2);
|
||||||
|
if clamped_y1 < clamped_y2 {
|
||||||
|
cell.page_pt_bbox[1] = clamped_y1;
|
||||||
|
cell.page_pt_bbox[3] = clamped_y2;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let col_end = cell.col + cell.colspan.max(1).saturating_sub(1);
|
||||||
|
if let (Some(&(x1, _)), Some(&(_, x2))) =
|
||||||
|
(col_bands.get(&cell.col), col_bands.get(&col_end))
|
||||||
|
{
|
||||||
|
let clamped_x1 = cell.page_pt_bbox[0].max(x1);
|
||||||
|
let clamped_x2 = cell.page_pt_bbox[2].min(x2);
|
||||||
|
if clamped_x1 < clamped_x2 {
|
||||||
|
cell.page_pt_bbox[0] = clamped_x1;
|
||||||
|
cell.page_pt_bbox[2] = clamped_x2;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Copy)]
|
||||||
|
enum Axis {
|
||||||
|
X,
|
||||||
|
Y,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn derive_axis_bands(cells: &[StructuredCell], axis: Axis) -> HashMap<usize, (f32, f32)> {
|
||||||
|
let mut by_index: HashMap<usize, Vec<(f32, f32)>> = HashMap::new();
|
||||||
|
|
||||||
|
// Prefer non-spanning cells so colspan/rowspan boxes do not skew a single
|
||||||
|
// column/row center. If an axis has no non-spanning examples for an index,
|
||||||
|
// fall back to anchored cells below.
|
||||||
|
for cell in cells {
|
||||||
|
let span = match axis {
|
||||||
|
Axis::X => cell.colspan.max(1),
|
||||||
|
Axis::Y => cell.rowspan.max(1),
|
||||||
|
};
|
||||||
|
if span == 1 {
|
||||||
|
let idx = match axis {
|
||||||
|
Axis::X => cell.col,
|
||||||
|
Axis::Y => cell.row,
|
||||||
|
};
|
||||||
|
by_index
|
||||||
|
.entry(idx)
|
||||||
|
.or_default()
|
||||||
|
.push(axis_bounds(cell.page_pt_bbox, axis));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for cell in cells {
|
||||||
|
let idx = match axis {
|
||||||
|
Axis::X => cell.col,
|
||||||
|
Axis::Y => cell.row,
|
||||||
|
};
|
||||||
|
if !by_index.contains_key(&idx) {
|
||||||
|
by_index
|
||||||
|
.entry(idx)
|
||||||
|
.or_default()
|
||||||
|
.push(axis_bounds(cell.page_pt_bbox, axis));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut rows: Vec<(usize, f32, f32, f32)> = by_index
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|(idx, bounds)| {
|
||||||
|
let mut min_edge = f32::INFINITY;
|
||||||
|
let mut max_edge = f32::NEG_INFINITY;
|
||||||
|
let mut center_sum = 0.0;
|
||||||
|
let mut count = 0usize;
|
||||||
|
for (lo, hi) in bounds {
|
||||||
|
if lo.is_finite() && hi.is_finite() && lo < hi {
|
||||||
|
min_edge = min_edge.min(lo);
|
||||||
|
max_edge = max_edge.max(hi);
|
||||||
|
center_sum += (lo + hi) * 0.5;
|
||||||
|
count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
(count > 0).then_some((idx, center_sum / count as f32, min_edge, max_edge))
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
if rows.len() < 2 {
|
||||||
|
return rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|(idx, _center, lo, hi)| (idx, (lo, hi)))
|
||||||
|
.collect();
|
||||||
|
}
|
||||||
|
|
||||||
|
rows.sort_by_key(|(idx, _, _, _)| *idx);
|
||||||
|
|
||||||
|
let mut bands = HashMap::new();
|
||||||
|
for i in 0..rows.len() {
|
||||||
|
let (idx, _center, min_edge, max_edge) = rows[i];
|
||||||
|
let lo = if i == 0 {
|
||||||
|
min_edge
|
||||||
|
} else {
|
||||||
|
(rows[i - 1].1 + rows[i].1) * 0.5
|
||||||
|
};
|
||||||
|
let hi = if i + 1 == rows.len() {
|
||||||
|
max_edge
|
||||||
|
} else {
|
||||||
|
(rows[i].1 + rows[i + 1].1) * 0.5
|
||||||
|
};
|
||||||
|
if lo.is_finite() && hi.is_finite() && lo < hi {
|
||||||
|
bands.insert(idx, (lo, hi));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
bands
|
||||||
|
}
|
||||||
|
|
||||||
|
fn axis_bounds(bbox: [f32; 4], axis: Axis) -> (f32, f32) {
|
||||||
|
match axis {
|
||||||
|
Axis::X => (bbox[0].min(bbox[2]), bbox[0].max(bbox[2])),
|
||||||
|
Axis::Y => (bbox[1].min(bbox[3]), bbox[1].max(bbox[3])),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Sanitize cell text for inclusion in a markdown pipe-table cell:
|
/// Sanitize cell text for inclusion in a markdown pipe-table cell:
|
||||||
/// collapse whitespace runs, drop newlines/tabs (cells must be one line),
|
/// collapse whitespace runs, drop newlines/tabs (cells must be one line),
|
||||||
/// and escape pipes that would otherwise break the table.
|
/// and escape pipes that would otherwise break the table.
|
||||||
@@ -448,6 +586,102 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn normalize_cell_bands_splits_overlapping_slanet_rows() {
|
||||||
|
let mut cells = vec![
|
||||||
|
StructuredCell {
|
||||||
|
row: 0,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: true,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [10.0, 100.0, 90.0, 120.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 0,
|
||||||
|
col: 1,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: true,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [90.0, 100.0, 170.0, 120.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 1,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [10.0, 116.0, 90.0, 136.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 1,
|
||||||
|
col: 1,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [90.0, 116.0, 170.0, 136.0],
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
normalize_cell_bands(&mut cells);
|
||||||
|
|
||||||
|
assert_eq!(cells[0].page_pt_bbox[3], cells[2].page_pt_bbox[1]);
|
||||||
|
assert_eq!(cells[1].page_pt_bbox[3], cells[3].page_pt_bbox[1]);
|
||||||
|
assert!(
|
||||||
|
(cells[0].page_pt_bbox[3] - 118.0).abs() < 0.01,
|
||||||
|
"row separator should be midpoint between row centers: {:?}",
|
||||||
|
cells
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn normalize_cell_bands_preserves_colspan_extent() {
|
||||||
|
let mut cells = vec![
|
||||||
|
StructuredCell {
|
||||||
|
row: 0,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 2,
|
||||||
|
is_header: true,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [8.0, 80.0, 172.0, 98.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 1,
|
||||||
|
col: 0,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [10.0, 96.0, 90.0, 114.0],
|
||||||
|
},
|
||||||
|
StructuredCell {
|
||||||
|
row: 1,
|
||||||
|
col: 1,
|
||||||
|
rowspan: 1,
|
||||||
|
colspan: 1,
|
||||||
|
is_header: false,
|
||||||
|
text: String::new(),
|
||||||
|
page_pt_bbox: [88.0, 96.0, 170.0, 114.0],
|
||||||
|
},
|
||||||
|
];
|
||||||
|
|
||||||
|
normalize_cell_bands(&mut cells);
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
cells[0].page_pt_bbox[0] <= cells[1].page_pt_bbox[0],
|
||||||
|
"spanning cell should retain the first column's left edge"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
cells[0].page_pt_bbox[2] >= cells[2].page_pt_bbox[2],
|
||||||
|
"spanning cell should retain the last column's right edge"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_int_attr_basic() {
|
fn parse_int_attr_basic() {
|
||||||
assert_eq!(parse_int_attr(" colspan=\"4\"", "colspan"), Some(4));
|
assert_eq!(parse_int_attr(" colspan=\"4\"", "colspan"), Some(4));
|
||||||
|
|||||||
@@ -1484,6 +1484,82 @@ fn poly(x1: f32, y1: f32, x2: f32, y2: f32) -> Vec<f32> {
|
|||||||
vec![x1, y1, x2, y1, x2, y2, x1, y2]
|
vec![x1, y1, x2, y1, x2, y2, x1, y2]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn synthetic_dense_table_pdf() -> Vec<u8> {
|
||||||
|
use lopdf::content::{Content, Operation};
|
||||||
|
use lopdf::{dictionary, Document, Object, Stream};
|
||||||
|
|
||||||
|
let mut doc = Document::with_version("1.5");
|
||||||
|
let pages_id = doc.new_object_id();
|
||||||
|
let page_id = doc.new_object_id();
|
||||||
|
let font_id = doc.new_object_id();
|
||||||
|
let content_id = doc.new_object_id();
|
||||||
|
|
||||||
|
doc.objects.insert(
|
||||||
|
font_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Font",
|
||||||
|
"Subtype" => "Type1",
|
||||||
|
"BaseFont" => "Helvetica",
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let operations = vec![
|
||||||
|
Operation::new("BT", vec![]),
|
||||||
|
Operation::new("Tf", vec!["F1".into(), 10.into()]),
|
||||||
|
Operation::new("Td", vec![20.into(), 700.into()]),
|
||||||
|
Operation::new("Tj", vec![Object::string_literal("Branch Name")]),
|
||||||
|
Operation::new("Td", vec![100.into(), 0.into()]),
|
||||||
|
Operation::new("Tj", vec![Object::string_literal("Deposits")]),
|
||||||
|
Operation::new("Td", vec![Object::Integer(-100), Object::Real(-16.8)]),
|
||||||
|
Operation::new("Tj", vec![Object::string_literal("Oak Street")]),
|
||||||
|
Operation::new("Td", vec![100.into(), 0.into()]),
|
||||||
|
Operation::new("Tj", vec![Object::string_literal("100")]),
|
||||||
|
Operation::new("Td", vec![Object::Integer(-100), Object::Real(-16.8)]),
|
||||||
|
Operation::new("Tj", vec![Object::string_literal("Boardwalk")]),
|
||||||
|
Operation::new("Td", vec![100.into(), 0.into()]),
|
||||||
|
Operation::new("Tj", vec![Object::string_literal("200")]),
|
||||||
|
Operation::new("ET", vec![]),
|
||||||
|
];
|
||||||
|
let content = Content { operations }.encode().unwrap();
|
||||||
|
doc.objects
|
||||||
|
.insert(content_id, Stream::new(dictionary! {}, content).into());
|
||||||
|
|
||||||
|
doc.objects.insert(
|
||||||
|
page_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Page",
|
||||||
|
"Parent" => pages_id,
|
||||||
|
"MediaBox" => vec![0.into(), 0.into(), 200.into(), 800.into()],
|
||||||
|
"Resources" => dictionary! {
|
||||||
|
"Font" => dictionary! {
|
||||||
|
"F1" => font_id,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"Contents" => content_id,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
doc.objects.insert(
|
||||||
|
pages_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Pages",
|
||||||
|
"Kids" => vec![page_id.into()],
|
||||||
|
"Count" => 1,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
let catalog_id = doc.add_object(dictionary! {
|
||||||
|
"Type" => "Catalog",
|
||||||
|
"Pages" => pages_id,
|
||||||
|
});
|
||||||
|
doc.trailer.set("Root", catalog_id);
|
||||||
|
|
||||||
|
let mut bytes = Vec::new();
|
||||||
|
doc.save_to(&mut bytes).unwrap();
|
||||||
|
bytes
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_extract_tables_with_structure_real_pdf_bits_pilani() {
|
fn test_extract_tables_with_structure_real_pdf_bits_pilani() {
|
||||||
use pdf_inspector::{extract_tables_with_structure_mem, TsrTableInput};
|
use pdf_inspector::{extract_tables_with_structure_mem, TsrTableInput};
|
||||||
@@ -1564,6 +1640,64 @@ fn test_extract_tables_with_structure_real_pdf_bits_pilani() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_with_structure_dense_overlapping_slanet_boxes() {
|
||||||
|
use pdf_inspector::{extract_tables_with_structure_mem, TsrTableInput};
|
||||||
|
|
||||||
|
let buf = synthetic_dense_table_pdf();
|
||||||
|
let tokens: Vec<String> = [
|
||||||
|
"<table>",
|
||||||
|
"<thead>",
|
||||||
|
"<tr>",
|
||||||
|
"<th></th>",
|
||||||
|
"<th></th>",
|
||||||
|
"</tr>",
|
||||||
|
"</thead>",
|
||||||
|
"<tbody>",
|
||||||
|
"<tr>",
|
||||||
|
"<td></td>",
|
||||||
|
"<td></td>",
|
||||||
|
"</tr>",
|
||||||
|
"<tr>",
|
||||||
|
"<td></td>",
|
||||||
|
"<td></td>",
|
||||||
|
"</tr>",
|
||||||
|
"</tbody>",
|
||||||
|
"</table>",
|
||||||
|
]
|
||||||
|
.into_iter()
|
||||||
|
.map(String::from)
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// Rows are spaced 16.8pt apart, while the SLANet-style boxes are 40pt
|
||||||
|
// tall and overlap adjacent rows. Text must still land in only one row.
|
||||||
|
let cell_bboxes = vec![
|
||||||
|
poly(10.0, 72.0, 100.0, 112.0),
|
||||||
|
poly(90.0, 72.0, 180.0, 112.0),
|
||||||
|
poly(10.0, 88.8, 100.0, 128.8),
|
||||||
|
poly(90.0, 88.8, 180.0, 128.8),
|
||||||
|
poly(10.0, 105.6, 100.0, 145.6),
|
||||||
|
poly(90.0, 105.6, 180.0, 145.6),
|
||||||
|
];
|
||||||
|
|
||||||
|
let mds = extract_tables_with_structure_mem(
|
||||||
|
&buf,
|
||||||
|
&[TsrTableInput {
|
||||||
|
page: 0,
|
||||||
|
crop_pdf_pt_bbox: [0.0, 0.0, 200.0, 800.0],
|
||||||
|
render_dpi: 72.0,
|
||||||
|
structure_tokens: tokens,
|
||||||
|
cell_bboxes,
|
||||||
|
}],
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let expected = "|Branch Name|Deposits|\n|---|---|\n|Oak Street|100|\n|Boardwalk|200|\n";
|
||||||
|
assert_eq!(mds[0], expected);
|
||||||
|
assert!(!mds[0].contains("Branch Name Oak Street"));
|
||||||
|
assert!(!mds[0].contains("Oak Street Boardwalk"));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_extract_tables_with_structure_input_order_preserved() {
|
fn test_extract_tables_with_structure_input_order_preserved() {
|
||||||
use pdf_inspector::{extract_tables_with_structure_mem, TsrTableInput};
|
use pdf_inspector::{extract_tables_with_structure_mem, TsrTableInput};
|
||||||
|
|||||||
Reference in New Issue
Block a user