Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3081f94e72 | ||
|
|
455dfe5a74 |
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.9.1",
|
||||
"version": "1.9.3",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
+94
-8
@@ -823,7 +823,8 @@ pub fn extract_tables_in_regions_mem(
|
||||
// symmetrically low under font-decode failure — this
|
||||
// guard breaks that symmetry by comparing against
|
||||
// bbox area, which is independent of extraction.
|
||||
if region_text_density_too_low(region_text_chars, region_area)
|
||||
if source != TableCandidateSource::KeyValue
|
||||
&& region_text_density_too_low(region_text_chars, region_area)
|
||||
&& !markdown_table_body_is_dense(&md)
|
||||
{
|
||||
return None;
|
||||
@@ -836,8 +837,10 @@ pub fn extract_tables_in_regions_mem(
|
||||
Some(TableCandidateIssue::LineRowUndercount)
|
||||
} else if wide_table_sparse_prefix_undercount(&md) {
|
||||
Some(TableCandidateIssue::SparseWideUndercount)
|
||||
} else if source != TableCandidateSource::Line
|
||||
&& text_cluster_column_undercount(&matched, shape)
|
||||
} else if !matches!(
|
||||
source,
|
||||
TableCandidateSource::Line | TableCandidateSource::KeyValue
|
||||
) && text_cluster_column_undercount(&matched, shape)
|
||||
{
|
||||
Some(TableCandidateIssue::TextColumnUndercount)
|
||||
} else if prose_grid_fragment_needs_ocr(&md) {
|
||||
@@ -881,6 +884,16 @@ pub fn extract_tables_in_regions_mem(
|
||||
{
|
||||
candidates.push(candidate);
|
||||
}
|
||||
if let Some(table) = tables::try_build_table_from_columns(&matched, page_1idx) {
|
||||
if let Some(candidate) = evaluate(TableCandidateSource::Column, &table) {
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
if let Some(table) = tables::try_build_key_value_table_from_rows(&matched, page_1idx) {
|
||||
if let Some(candidate) = evaluate(TableCandidateSource::KeyValue, &table) {
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
match select_table_candidate(&candidates) {
|
||||
Some(candidate) => page_results.push(RegionText {
|
||||
@@ -3752,6 +3765,8 @@ enum TableCandidateSource {
|
||||
Rect,
|
||||
Line,
|
||||
Heuristic,
|
||||
Column,
|
||||
KeyValue,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
@@ -3786,8 +3801,12 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
|
||||
// serving a tidy-looking fragment.
|
||||
if first.issue == Some(TableCandidateIssue::LineRowUndercount) {
|
||||
return candidates.iter().find(|candidate| {
|
||||
candidate.source == TableCandidateSource::Heuristic
|
||||
&& candidate.issue.is_none()
|
||||
matches!(
|
||||
candidate.source,
|
||||
TableCandidateSource::Heuristic
|
||||
| TableCandidateSource::Column
|
||||
| TableCandidateSource::KeyValue
|
||||
) && candidate.issue.is_none()
|
||||
&& candidate.shape.cols * 10 >= first.shape.cols * 13
|
||||
});
|
||||
}
|
||||
@@ -3805,14 +3824,31 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
|
||||
TableCandidateSource::Rect | TableCandidateSource::Line
|
||||
) {
|
||||
if let Some(heuristic) = candidates.iter().find(|candidate| {
|
||||
candidate.source == TableCandidateSource::Heuristic
|
||||
&& candidate.issue.is_none()
|
||||
matches!(
|
||||
candidate.source,
|
||||
TableCandidateSource::Heuristic
|
||||
| TableCandidateSource::Column
|
||||
| TableCandidateSource::KeyValue
|
||||
) && candidate.issue.is_none()
|
||||
&& heuristic_substantially_better(candidate.shape, accepted.shape)
|
||||
}) {
|
||||
accepted = heuristic;
|
||||
}
|
||||
}
|
||||
|
||||
if accepted.source == TableCandidateSource::Heuristic {
|
||||
if let Some(layout_candidate) = candidates.iter().find(|candidate| {
|
||||
matches!(
|
||||
candidate.source,
|
||||
TableCandidateSource::Column | TableCandidateSource::KeyValue
|
||||
) && candidate.issue.is_none()
|
||||
&& candidate.shape.cols >= accepted.shape.cols
|
||||
&& candidate.shape.rows > accepted.shape.rows
|
||||
}) {
|
||||
accepted = layout_candidate;
|
||||
}
|
||||
}
|
||||
|
||||
Some(accepted)
|
||||
}
|
||||
|
||||
@@ -4314,7 +4350,9 @@ fn looks_like_partial_table_ex(markdown: &str, layout_assisted: bool) -> bool {
|
||||
cell.trim().is_empty() && header_empty_indices.contains(&idx)
|
||||
})
|
||||
&& layout_assisted_empty_header_has_dense_body(markdown, n_cols);
|
||||
if !sparse_row_shares_header_spacer {
|
||||
let sparse_row_is_section_label = layout_assisted
|
||||
&& layout_assisted_sparse_section_row_is_ok(data_inner, markdown, n_cols);
|
||||
if !sparse_row_shares_header_spacer && !sparse_row_is_section_label {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -4406,6 +4444,26 @@ fn layout_assisted_empty_header_has_dense_body(markdown: &str, n_cols: usize) ->
|
||||
&& filled_cells * 100 >= total_cells * 45
|
||||
}
|
||||
|
||||
fn layout_assisted_sparse_section_row_is_ok(row: &[&str], markdown: &str, n_cols: usize) -> bool {
|
||||
let labels: Vec<&str> = row
|
||||
.iter()
|
||||
.map(|cell| cell.trim())
|
||||
.filter(|cell| !cell.is_empty())
|
||||
.collect();
|
||||
if labels.len() != 1 {
|
||||
return false;
|
||||
}
|
||||
let label = labels[0];
|
||||
if label.len() > 40 || !label.chars().any(|ch| ch.is_alphabetic()) {
|
||||
return false;
|
||||
}
|
||||
if label.ends_with('.') || label.ends_with('!') || label.ends_with('?') || label.ends_with(':')
|
||||
{
|
||||
return false;
|
||||
}
|
||||
layout_assisted_empty_header_has_dense_body(markdown, n_cols)
|
||||
}
|
||||
|
||||
fn markdown_table_body_is_dense(markdown: &str) -> bool {
|
||||
let rows = markdown_pipe_rows(markdown);
|
||||
let data_rows: Vec<&Vec<&str>> = rows
|
||||
@@ -4787,6 +4845,16 @@ mod table_candidate_selection_tests {
|
||||
assert_eq!(selected.source, TableCandidateSource::Heuristic);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefers_clean_column_fallback_when_it_recovers_more_rows() {
|
||||
let candidates = vec![
|
||||
candidate(TableCandidateSource::Heuristic, 6, 5, None),
|
||||
candidate(TableCandidateSource::Column, 7, 5, None),
|
||||
];
|
||||
let selected = select_table_candidate(&candidates).unwrap();
|
||||
assert_eq!(selected.source, TableCandidateSource::Column);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_candidate_collapsing_captured_y_clusters_is_suspicious() {
|
||||
let long = "value value value value value value value value value value value value";
|
||||
@@ -5091,6 +5159,24 @@ mod looks_like_partial_table_tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_section_row_passes_when_layout_assisted_body_is_dense() {
|
||||
let md = "|Properties|Conditions|Method|Typical values|Units|\n\
|
||||
|---|---|---|---|---|\n\
|
||||
|Rheology|||||\n\
|
||||
|Melt Flow Rate|230 C/2.16 kg|ASTM D1238|3.0|g/10 min|\n\
|
||||
|Tensile Stress at Yield|50 mm/min|ASTM D638|31|MPa|\n\
|
||||
|Elongation at Yield|50 mm/min|ASTM D638|8|%|";
|
||||
assert!(
|
||||
looks_like_partial_table(md),
|
||||
"strict mode rejects the sparse first row"
|
||||
);
|
||||
assert!(
|
||||
!looks_like_partial_table_ex(md, true),
|
||||
"layout-assisted should allow a short section label above dense table rows"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paragraph_still_rejected_when_layout_assisted() {
|
||||
// Paragraph detection is not relaxed — it's a genuine extraction issue.
|
||||
|
||||
+65
-4
@@ -208,6 +208,24 @@ fn looks_like_compact_entry_label(cell: &str) -> bool {
|
||||
(1..=6).contains(&words)
|
||||
}
|
||||
|
||||
fn looks_like_plain_section_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 4 || trimmed.len() > 40 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.ends_with(['.', ',', ';', ':']) || trimmed.contains(|ch: char| ch.is_ascii_digit()) {
|
||||
return false;
|
||||
}
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|ch| !ch.is_lowercase()) {
|
||||
return false;
|
||||
}
|
||||
trimmed
|
||||
.chars()
|
||||
.all(|ch| ch.is_alphabetic() || ch.is_whitespace() || matches!(ch, '&' | '/' | '-'))
|
||||
&& starts_with_uppercase_alpha(trimmed)
|
||||
&& (1..=4).contains(&alpha_word_count(trimmed))
|
||||
}
|
||||
|
||||
fn ends_like_incomplete_phrase(cell: &str) -> bool {
|
||||
let lower = cell.trim_end().to_ascii_lowercase();
|
||||
lower.ends_with(" and")
|
||||
@@ -305,6 +323,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
.and_then(|r| r.first())
|
||||
.map(|c| c.trim())
|
||||
.unwrap_or("");
|
||||
let header_filled = cleaned
|
||||
.first()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(num_cols);
|
||||
let looks_like_spanning_first_column_row = first_cell.is_empty()
|
||||
&& row.len() >= 4
|
||||
&& non_first_cells.len() == row.len().saturating_sub(1)
|
||||
@@ -328,6 +350,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
&& non_first_cells
|
||||
.iter()
|
||||
.any(|cell| looks_like_compact_entry_label(cell));
|
||||
let looks_like_section_label_row = !first_cell.is_empty()
|
||||
&& filled_cells == 1
|
||||
&& header_filled >= 3
|
||||
&& looks_like_plain_section_label(first_cell);
|
||||
// Classic continuation: first cell empty, content in other cells
|
||||
let is_classic_continuation = first_cell.is_empty()
|
||||
&& !non_first_cells.is_empty()
|
||||
@@ -344,10 +370,6 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
.last()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(0);
|
||||
let header_filled = cleaned
|
||||
.first()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(num_cols);
|
||||
// Merge when the row has significantly fewer filled cells than header.
|
||||
// For wide tables (5+ cols), require ≤50% of header cells.
|
||||
// For narrow tables (2-4 cols), require fewer than header cells.
|
||||
@@ -369,6 +391,7 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
&& !looks_like_spanning_first_column_row
|
||||
&& !looks_like_hierarchical_subrow
|
||||
&& !looks_like_new_first_column_entry
|
||||
&& !looks_like_section_label_row
|
||||
&& !is_short_subheader;
|
||||
|
||||
let is_continuation = is_classic_continuation || is_wrapped_continuation;
|
||||
@@ -514,6 +537,44 @@ mod tests {
|
||||
assert!(cleaned[1][1].contains("continued text here"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_first_column_section_label_not_merged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Properties".into(),
|
||||
"Conditions".into(),
|
||||
"Method".into(),
|
||||
"Typical values".into(),
|
||||
"Units".into(),
|
||||
],
|
||||
vec![
|
||||
"Melt Flow Rate".into(),
|
||||
"230 C/2.16 kg".into(),
|
||||
"ASTM D1238".into(),
|
||||
"3.0".into(),
|
||||
"g/10 min".into(),
|
||||
],
|
||||
vec![
|
||||
"Mechanical".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"Tensile Stress at Yield".into(),
|
||||
"50 mm/min".into(),
|
||||
"ASTM D638".into(),
|
||||
"31".into(),
|
||||
"MPa".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 4);
|
||||
assert_eq!(cleaned[2][0], "Mechanical");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_short_subheader_not_merged() {
|
||||
let cells = vec![
|
||||
|
||||
@@ -460,6 +460,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
item_indices.push(item_idx);
|
||||
}
|
||||
}
|
||||
merge_superscript_marker_rows(&mut row_ys, &mut cells);
|
||||
|
||||
// Validate: need reasonable fill rate
|
||||
let total_cells = row_ys.len() * columns.len();
|
||||
@@ -547,6 +548,536 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
Some(Table::new(col_xs, row_ys, cells, item_indices))
|
||||
}
|
||||
|
||||
/// Build a region-scoped two-column key/value table from text baselines.
|
||||
///
|
||||
/// This intentionally lives outside the full-page heuristic detector. Layout
|
||||
/// callers already supplied a table-shaped bbox, and some real table regions
|
||||
/// are plain product/spec forms with only two visual columns. The main column
|
||||
/// fallback starts at four columns to avoid newspaper/prose false positives;
|
||||
/// this path keeps tighter key/value-specific guards instead.
|
||||
pub(crate) fn try_build_key_value_table_from_rows(items: &[TextItem], page: u32) -> Option<Table> {
|
||||
let page_items: Vec<RowItem> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| item.page == page && !item.text.trim().is_empty())
|
||||
.map(|(idx, item)| RowItem {
|
||||
index: idx,
|
||||
item: item.clone(),
|
||||
})
|
||||
.collect();
|
||||
|
||||
if page_items.len() < 4 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let median_font_size = median_f32(page_items.iter().map(|ri| ri.item.font_size).collect())
|
||||
.unwrap_or(10.0)
|
||||
.max(1.0);
|
||||
let y_tol = (median_font_size * 0.75).clamp(4.0, 9.0);
|
||||
let rows = group_key_value_visual_rows(page_items, y_tol);
|
||||
if rows.len() < 2 || rows.len() > 80 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let split_x = infer_key_value_split_x(&rows, median_font_size)?;
|
||||
let mut kv_rows: Vec<KeyValueRow> = Vec::new();
|
||||
let mut paired_rows = 0usize;
|
||||
let mut section_rows = 0usize;
|
||||
let mut left_label_like = 0usize;
|
||||
let mut left_starts = Vec::new();
|
||||
let mut right_starts = Vec::new();
|
||||
|
||||
for row in &rows {
|
||||
let mut left_items = Vec::new();
|
||||
let mut right_items = Vec::new();
|
||||
for item in &row.items {
|
||||
if item.item.x < split_x {
|
||||
left_items.push(item);
|
||||
} else {
|
||||
right_items.push(item);
|
||||
}
|
||||
}
|
||||
|
||||
let left = join_row_item_text(&left_items);
|
||||
let right = join_row_item_text(&right_items);
|
||||
if left.is_empty() && right.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut item_indices: Vec<usize> = row.items.iter().map(|ri| ri.index).collect();
|
||||
item_indices.sort_unstable();
|
||||
item_indices.dedup();
|
||||
|
||||
if !left.is_empty() && !right.is_empty() {
|
||||
paired_rows += 1;
|
||||
if looks_like_key_value_label(&left) {
|
||||
left_label_like += 1;
|
||||
}
|
||||
if let Some(x) = left_items.first().map(|ri| ri.item.x) {
|
||||
left_starts.push(x);
|
||||
}
|
||||
if let Some(x) = right_items.first().map(|ri| ri.item.x) {
|
||||
right_starts.push(x);
|
||||
}
|
||||
} else if !left.is_empty() {
|
||||
section_rows += 1;
|
||||
}
|
||||
|
||||
kv_rows.push(KeyValueRow {
|
||||
y: row.y,
|
||||
left,
|
||||
right,
|
||||
item_indices,
|
||||
});
|
||||
}
|
||||
|
||||
if kv_rows.len() < 2 || paired_rows < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let header_inferred = key_value_first_pair_is_header(&kv_rows);
|
||||
let data_pairs = if header_inferred {
|
||||
paired_rows.saturating_sub(1)
|
||||
} else {
|
||||
paired_rows
|
||||
};
|
||||
if data_pairs < 1 {
|
||||
return None;
|
||||
}
|
||||
|
||||
if section_rows > paired_rows * 2 + 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let label_rows_for_score = if header_inferred {
|
||||
paired_rows.saturating_sub(1)
|
||||
} else {
|
||||
paired_rows
|
||||
};
|
||||
let label_like_for_score = if header_inferred && !kv_rows.is_empty() {
|
||||
left_label_like.saturating_sub(1)
|
||||
} else {
|
||||
left_label_like
|
||||
};
|
||||
if label_rows_for_score >= 2 && label_like_for_score * 2 < label_rows_for_score {
|
||||
return None;
|
||||
}
|
||||
|
||||
let left_x = median_f32(left_starts).unwrap_or_else(|| {
|
||||
rows.iter()
|
||||
.flat_map(|row| row.items.iter().map(|ri| ri.item.x))
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
});
|
||||
let right_x = median_f32(right_starts).unwrap_or(split_x);
|
||||
if !left_x.is_finite() || !right_x.is_finite() || right_x - left_x < 40.0 {
|
||||
return None;
|
||||
}
|
||||
let right_cluster_count = significant_side_x_clusters(&rows, split_x, false);
|
||||
let marker_rows = marker_matrix_value_rows(&kv_rows);
|
||||
if (right_cluster_count >= 5 && paired_rows >= 3)
|
||||
|| (right_cluster_count >= 3 && marker_rows >= 3 && marker_rows * 2 >= paired_rows)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
|
||||
if key_value_rows_look_like_prose(&kv_rows, header_inferred) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut table_rows = Vec::new();
|
||||
let mut cells = Vec::new();
|
||||
let mut item_indices = Vec::new();
|
||||
|
||||
let mut start_idx = 0usize;
|
||||
if header_inferred {
|
||||
let header = &kv_rows[0];
|
||||
table_rows.push(header.y);
|
||||
cells.push(vec![header.left.clone(), header.right.clone()]);
|
||||
item_indices.extend(header.item_indices.iter().copied());
|
||||
start_idx = 1;
|
||||
} else {
|
||||
table_rows.push(kv_rows.first().map(|row| row.y + y_tol).unwrap_or(0.0));
|
||||
cells.push(vec!["Field".to_string(), "Value".to_string()]);
|
||||
}
|
||||
|
||||
for row in kv_rows.iter().skip(start_idx) {
|
||||
if !row.left.is_empty() && !row.right.is_empty() {
|
||||
table_rows.push(row.y);
|
||||
cells.push(vec![row.left.clone(), row.right.clone()]);
|
||||
item_indices.extend(row.item_indices.iter().copied());
|
||||
} else if !row.left.is_empty() {
|
||||
table_rows.push(row.y);
|
||||
cells.push(vec!["Section".to_string(), row.left.clone()]);
|
||||
item_indices.extend(row.item_indices.iter().copied());
|
||||
} else if !row.right.is_empty() {
|
||||
if let Some(last) = cells.last_mut() {
|
||||
if let Some(value) = last.get_mut(1) {
|
||||
if !value.trim().is_empty() {
|
||||
value.push(' ');
|
||||
}
|
||||
value.push_str(&row.right);
|
||||
item_indices.extend(row.item_indices.iter().copied());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if cells.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
item_indices.sort_unstable();
|
||||
item_indices.dedup();
|
||||
|
||||
log::debug!(
|
||||
"key-value table: {} rows, pairs={}, sections={}, split_x={:.1}",
|
||||
cells.len(),
|
||||
paired_rows,
|
||||
section_rows,
|
||||
split_x
|
||||
);
|
||||
|
||||
Some(Table::new(
|
||||
vec![left_x, right_x],
|
||||
table_rows,
|
||||
cells,
|
||||
item_indices,
|
||||
))
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct RowItem {
|
||||
index: usize,
|
||||
item: TextItem,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct VisualRow {
|
||||
y: f32,
|
||||
items: Vec<RowItem>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct KeyValueRow {
|
||||
y: f32,
|
||||
left: String,
|
||||
right: String,
|
||||
item_indices: Vec<usize>,
|
||||
}
|
||||
|
||||
fn group_key_value_visual_rows(mut items: Vec<RowItem>, y_tol: f32) -> Vec<VisualRow> {
|
||||
items.sort_by(|a, b| {
|
||||
b.item
|
||||
.y
|
||||
.total_cmp(&a.item.y)
|
||||
.then_with(|| a.item.x.total_cmp(&b.item.x))
|
||||
});
|
||||
|
||||
let mut rows: Vec<VisualRow> = Vec::new();
|
||||
for row_item in items {
|
||||
if let Some(row) = rows
|
||||
.iter_mut()
|
||||
.find(|row| (row.y - row_item.item.y).abs() <= y_tol)
|
||||
{
|
||||
let len = row.items.len() as f32;
|
||||
row.y = (row.y * len + row_item.item.y) / (len + 1.0);
|
||||
row.items.push(row_item);
|
||||
continue;
|
||||
}
|
||||
|
||||
rows.push(VisualRow {
|
||||
y: row_item.item.y,
|
||||
items: vec![row_item],
|
||||
});
|
||||
}
|
||||
|
||||
for row in &mut rows {
|
||||
row.items.sort_by(|a, b| a.item.x.total_cmp(&b.item.x));
|
||||
}
|
||||
rows.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
rows
|
||||
}
|
||||
|
||||
fn infer_key_value_split_x(rows: &[VisualRow], median_font_size: f32) -> Option<f32> {
|
||||
let min_gap = (median_font_size * 2.0).max(24.0);
|
||||
let mut splits = Vec::new();
|
||||
|
||||
for row in rows {
|
||||
if row.items.len() < 2 {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut best_gap = 0.0f32;
|
||||
let mut best_split = None;
|
||||
for pair in row.items.windows(2) {
|
||||
let left = &pair[0].item;
|
||||
let right = &pair[1].item;
|
||||
let left_right = left.x + left.width.max(0.0);
|
||||
let gap = right.x - left_right;
|
||||
if gap > best_gap {
|
||||
best_gap = gap;
|
||||
best_split = Some(left_right + gap / 2.0);
|
||||
}
|
||||
}
|
||||
|
||||
if best_gap >= min_gap {
|
||||
if let Some(split) = best_split {
|
||||
splits.push(split);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if splits.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
median_f32(splits)
|
||||
}
|
||||
|
||||
fn join_row_item_text(items: &[&RowItem]) -> String {
|
||||
let mut parts = Vec::new();
|
||||
for item in items {
|
||||
let trimmed = item.item.text.trim();
|
||||
if !trimmed.is_empty() {
|
||||
parts.push(trimmed);
|
||||
}
|
||||
}
|
||||
normalize_cell_text(&parts.join(" "))
|
||||
}
|
||||
|
||||
fn normalize_cell_text(text: &str) -> String {
|
||||
text.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
|
||||
fn key_value_first_pair_is_header(rows: &[KeyValueRow]) -> bool {
|
||||
let Some(first) = rows.first() else {
|
||||
return false;
|
||||
};
|
||||
if first.left.is_empty() || first.right.is_empty() {
|
||||
return false;
|
||||
}
|
||||
if !looks_like_key_value_header_cell(&first.left)
|
||||
|| !looks_like_key_value_header_cell(&first.right)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
rows.iter()
|
||||
.skip(1)
|
||||
.any(|row| !row.left.is_empty() && !row.right.is_empty())
|
||||
}
|
||||
|
||||
fn looks_like_key_value_header_cell(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 2 || trimmed.len() > 40 {
|
||||
return false;
|
||||
}
|
||||
let words = word_count_simple(trimmed);
|
||||
if !(1..=4).contains(&words) {
|
||||
return false;
|
||||
}
|
||||
let lower = trimmed.to_ascii_lowercase();
|
||||
if matches!(
|
||||
lower.as_str(),
|
||||
"yes" | "no" | "true" | "false" | "none" | "n/a" | "na"
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
trimmed.chars().any(|c| c.is_alphabetic())
|
||||
&& !trimmed.chars().any(|c| c.is_ascii_digit())
|
||||
&& !trimmed.ends_with(['.', ',', ';', ':'])
|
||||
}
|
||||
|
||||
fn looks_like_key_value_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 2 || trimmed.len() > 90 {
|
||||
return false;
|
||||
}
|
||||
let words = word_count_simple(trimmed);
|
||||
if words == 0 || words > 10 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.ends_with(['.', ',', ';']) {
|
||||
return false;
|
||||
}
|
||||
trimmed.chars().any(|c| c.is_alphabetic())
|
||||
}
|
||||
|
||||
fn key_value_rows_look_like_prose(rows: &[KeyValueRow], header_inferred: bool) -> bool {
|
||||
let mut long_sentence_cells = 0usize;
|
||||
let mut total_cells = 0usize;
|
||||
let mut total_chars = 0usize;
|
||||
let mut paired_rows = 0usize;
|
||||
let mut solo_prose_rows = 0usize;
|
||||
|
||||
for row in rows.iter().skip(usize::from(header_inferred)) {
|
||||
if !row.left.is_empty() && !row.right.is_empty() {
|
||||
paired_rows += 1;
|
||||
} else {
|
||||
let solo = if row.left.is_empty() {
|
||||
row.right.trim()
|
||||
} else {
|
||||
row.left.trim()
|
||||
};
|
||||
if solo.chars().count() > 70
|
||||
|| word_count_simple(solo) > 9
|
||||
|| (solo.chars().count() > 35 && solo.ends_with(['.', '!', '?']))
|
||||
{
|
||||
solo_prose_rows += 1;
|
||||
}
|
||||
}
|
||||
for cell in [&row.left, &row.right] {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.is_empty() {
|
||||
continue;
|
||||
}
|
||||
total_cells += 1;
|
||||
total_chars += trimmed.chars().count();
|
||||
if trimmed.chars().count() > 100
|
||||
|| (trimmed.chars().count() > 55 && trimmed.ends_with(['.', '!', '?']))
|
||||
{
|
||||
long_sentence_cells += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if paired_rows < 1 || total_cells == 0 {
|
||||
return true;
|
||||
}
|
||||
if solo_prose_rows >= 3 {
|
||||
return true;
|
||||
}
|
||||
|
||||
let avg_chars = total_chars as f32 / total_cells as f32;
|
||||
avg_chars > 75.0 || long_sentence_cells * 2 >= total_cells
|
||||
}
|
||||
|
||||
fn marker_matrix_value_rows(rows: &[KeyValueRow]) -> usize {
|
||||
rows.iter()
|
||||
.filter(|row| !row.left.is_empty() && compact_marker_value(&row.right))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn compact_marker_value(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.is_empty() || trimmed.chars().count() > 80 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.chars().any(|ch| ch.is_alphabetic()) {
|
||||
return false;
|
||||
}
|
||||
trimmed
|
||||
.chars()
|
||||
.any(|ch| ch.is_ascii_digit() || matches!(ch, '•' | '●' | '·'))
|
||||
}
|
||||
|
||||
fn significant_side_x_clusters(rows: &[VisualRow], split_x: f32, left_side: bool) -> usize {
|
||||
let mut xs = Vec::new();
|
||||
for row in rows {
|
||||
for item in &row.items {
|
||||
let is_left = item.item.x < split_x;
|
||||
if is_left == left_side {
|
||||
xs.push(item.item.x);
|
||||
}
|
||||
}
|
||||
}
|
||||
xs.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
let mut counts = Vec::new();
|
||||
let mut center = None::<f32>;
|
||||
let mut count = 0usize;
|
||||
for x in xs {
|
||||
match center {
|
||||
Some(current) if (x - current).abs() <= 8.0 => {
|
||||
center = Some((current * count as f32 + x) / (count as f32 + 1.0));
|
||||
count += 1;
|
||||
}
|
||||
Some(_) => {
|
||||
counts.push(count);
|
||||
center = Some(x);
|
||||
count = 1;
|
||||
}
|
||||
None => {
|
||||
center = Some(x);
|
||||
count = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
if count > 0 {
|
||||
counts.push(count);
|
||||
}
|
||||
|
||||
counts.into_iter().filter(|&count| count >= 2).count()
|
||||
}
|
||||
|
||||
fn word_count_simple(cell: &str) -> usize {
|
||||
cell.split_whitespace()
|
||||
.filter(|word| word.chars().any(|c| c.is_alphanumeric()))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn median_f32(mut values: Vec<f32>) -> Option<f32> {
|
||||
values.retain(|value| value.is_finite());
|
||||
if values.is_empty() {
|
||||
return None;
|
||||
}
|
||||
values.sort_by(|a, b| a.total_cmp(b));
|
||||
Some(values[values.len() / 2])
|
||||
}
|
||||
|
||||
fn merge_superscript_marker_rows(row_ys: &mut Vec<f32>, cells: &mut Vec<Vec<String>>) {
|
||||
let mut row_idx = 0;
|
||||
while row_idx < cells.len() {
|
||||
let non_empty: Vec<(usize, String)> = cells[row_idx]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(col_idx, cell)| {
|
||||
let trimmed = cell.trim();
|
||||
(!trimmed.is_empty()).then_some((col_idx, trimmed.to_string()))
|
||||
})
|
||||
.collect();
|
||||
|
||||
if non_empty.len() != 1 || !is_superscript_marker_cell(&non_empty[0].1) {
|
||||
row_idx += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
let (marker_col, marker) = &non_empty[0];
|
||||
let prev =
|
||||
(row_idx > 0).then(|| (row_idx - 1, (row_ys[row_idx - 1] - row_ys[row_idx]).abs()));
|
||||
let next = (row_idx + 1 < cells.len())
|
||||
.then(|| (row_idx + 1, (row_ys[row_idx] - row_ys[row_idx + 1]).abs()));
|
||||
let target = [prev, next]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter(|(_, gap)| *gap <= 10.0)
|
||||
.min_by(|(_, gap_a), (_, gap_b)| gap_a.total_cmp(gap_b))
|
||||
.map(|(idx, _)| idx);
|
||||
|
||||
let Some(target_idx) = target else {
|
||||
row_idx += 1;
|
||||
continue;
|
||||
};
|
||||
|
||||
let target_cell = &mut cells[target_idx][*marker_col];
|
||||
if target_cell.trim().is_empty() {
|
||||
*target_cell = marker.to_string();
|
||||
} else {
|
||||
target_cell.push_str(marker);
|
||||
}
|
||||
cells.remove(row_idx);
|
||||
row_ys.remove(row_idx);
|
||||
}
|
||||
}
|
||||
|
||||
fn is_superscript_marker_cell(value: &str) -> bool {
|
||||
let trimmed = value.trim();
|
||||
!trimmed.is_empty()
|
||||
&& trimmed.chars().count() <= 2
|
||||
&& trimmed
|
||||
.chars()
|
||||
.all(|ch| matches!(ch, '*' | '#' | 'o' | 'O' | '°' | 'º' | '†' | '‡'))
|
||||
}
|
||||
|
||||
/// What kind of structure a detected `Table` represents. Classification is
|
||||
/// computed once at construction so consumers don't have to re-analyze the
|
||||
/// cells (and stay consistent across detection backends).
|
||||
@@ -689,6 +1220,190 @@ mod tests {
|
||||
assert!(md.contains("|Cell 1|"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_superscript_marker_rows() {
|
||||
let mut rows = vec![506.0, 500.0, 480.0];
|
||||
let mut cells = vec![
|
||||
vec!["".into(), "".into(), "*".into()],
|
||||
vec!["Name".into(), "Method".into(), "Typical values".into()],
|
||||
vec!["Flow".into(), "ASTM D1238".into(), "3.0".into()],
|
||||
];
|
||||
|
||||
merge_superscript_marker_rows(&mut rows, &mut cells);
|
||||
|
||||
assert_eq!(rows, vec![500.0, 480.0]);
|
||||
assert_eq!(cells[0][2], "Typical values*");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_column_builder_handles_borderless_specs_table() {
|
||||
let items = vec![
|
||||
make_char("*", 458.1, 544.2, 8.0, 4.4),
|
||||
make_char("Properties", 36.0, 538.6, 12.0, 53.1),
|
||||
make_char("Conditions", 195.8, 538.6, 12.0, 55.0),
|
||||
make_char("Method", 297.2, 538.6, 12.0, 39.4),
|
||||
make_char("Typical values", 384.1, 538.6, 12.0, 74.0),
|
||||
make_char("Units", 510.6, 538.6, 8.0, 17.9),
|
||||
make_char("Rheology", 36.0, 508.3, 10.0, 40.6),
|
||||
make_char("o", 209.8, 492.5, 6.5, 3.5),
|
||||
make_char("Melt Flow Rate", 36.0, 488.0, 10.0, 65.2),
|
||||
make_char("230 ", 190.4, 488.0, 10.0, 19.4),
|
||||
make_char("C/2.16 kg", 213.3, 488.0, 10.0, 42.8),
|
||||
make_char("ASTM D1238", 288.4, 488.0, 10.0, 56.8),
|
||||
make_char("3.0 ", 416.4, 488.0, 10.0, 16.9),
|
||||
make_char("g/10 min", 504.1, 488.0, 10.0, 39.5),
|
||||
make_char("Mechanical", 36.0, 451.5, 10.0, 48.3),
|
||||
make_char("Tensile Stress at Yield", 36.0, 431.3, 10.0, 96.7),
|
||||
make_char("50 mm/min", 197.9, 431.3, 10.0, 50.8),
|
||||
make_char("ASTM D638", 291.2, 431.3, 10.0, 51.3),
|
||||
make_char("31 ", 417.9, 431.3, 10.0, 13.9),
|
||||
make_char("MPa", 514.7, 431.3, 10.0, 18.4),
|
||||
make_char("Elongation at Yield", 36.0, 403.0, 10.0, 82.2),
|
||||
make_char("50 mm/min", 197.9, 403.0, 10.0, 50.8),
|
||||
make_char("ASTM D638", 291.2, 403.0, 10.0, 51.3),
|
||||
make_char("8 ", 420.6, 403.0, 10.0, 8.5),
|
||||
make_char("%", 519.1, 403.0, 10.0, 9.7),
|
||||
make_char("Flexural Modulus", 36.0, 374.6, 10.0, 74.0),
|
||||
make_char("ASTM D790", 291.2, 374.6, 10.0, 51.3),
|
||||
make_char("1400", 412.4, 374.6, 10.0, 21.8),
|
||||
make_char("MPa", 514.7, 374.6, 10.0, 18.4),
|
||||
];
|
||||
|
||||
let table = try_build_table_from_columns(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(
|
||||
md.contains("|Properties|Conditions|Method|Typical values*|Units|"),
|
||||
"{md}"
|
||||
);
|
||||
assert!(md.contains("|Mechanical|||||"), "{md}");
|
||||
assert!(
|
||||
md.contains("|Flexural Modulus||ASTM D790|1400|MPa|"),
|
||||
"{md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_recovers_sectioned_specs_table() {
|
||||
let items = vec![
|
||||
make_char("Ordering Information", 69.0, 700.0, 9.0, 96.0),
|
||||
make_char("Package Contents", 69.0, 680.0, 9.0, 82.0),
|
||||
make_char(
|
||||
"CCH Adapter Panel with 3 m pigtail; installation guide",
|
||||
200.0,
|
||||
680.0,
|
||||
9.0,
|
||||
245.0,
|
||||
),
|
||||
make_char("Units per Delivery", 69.0, 660.0, 9.0, 78.0),
|
||||
make_char("1/1", 200.0, 660.0, 9.0, 18.0),
|
||||
];
|
||||
|
||||
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(md.contains("|Field|Value|"), "{md}");
|
||||
assert!(md.contains("|Section|Ordering Information|"), "{md}");
|
||||
assert!(
|
||||
md.contains(
|
||||
"|Package Contents|CCH Adapter Panel with 3 m pigtail; installation guide|"
|
||||
),
|
||||
"{md}"
|
||||
);
|
||||
assert!(md.contains("|Units per Delivery|1/1|"), "{md}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_preserves_two_column_header() {
|
||||
let items = vec![
|
||||
make_char("Media", 86.0, 700.0, 10.0, 36.0),
|
||||
make_char("Options", 311.0, 700.0, 10.0, 44.0),
|
||||
make_char("BACnet/IP (Annex J)", 86.0, 680.0, 10.0, 115.0),
|
||||
make_char("Register as Foreign Device", 311.0, 680.0, 10.0, 138.0),
|
||||
];
|
||||
|
||||
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(md.starts_with("|Media|Options|"), "{md}");
|
||||
assert!(
|
||||
md.contains("|BACnet/IP (Annex J)|Register as Foreign Device|"),
|
||||
"{md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_keeps_repeated_spec_sections() {
|
||||
let items = vec![
|
||||
make_char("1.33 DUAL VVT-i", 90.0, 700.0, 9.0, 82.0),
|
||||
make_char("Engine Code", 90.0, 682.0, 9.0, 62.0),
|
||||
make_char("1NR-FE", 406.0, 682.0, 9.0, 42.0),
|
||||
make_char("Type", 90.0, 664.0, 9.0, 24.0),
|
||||
make_char("Four cylinders in-line", 376.0, 664.0, 9.0, 104.0),
|
||||
make_char("1.6 VALVEMATIC", 90.0, 636.0, 9.0, 78.0),
|
||||
make_char("Engine Code", 90.0, 618.0, 9.0, 62.0),
|
||||
make_char("1ZR-FAE", 404.0, 618.0, 9.0, 44.0),
|
||||
];
|
||||
|
||||
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(md.contains("|Section|1.33 DUAL VVT-i|"), "{md}");
|
||||
assert!(md.contains("|Engine Code|1NR-FE|"), "{md}");
|
||||
assert!(md.contains("|Section|1.6 VALVEMATIC|"), "{md}");
|
||||
assert!(md.contains("|Engine Code|1ZR-FAE|"), "{md}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_rejects_split_prose() {
|
||||
let items = vec![
|
||||
make_char(
|
||||
"This paragraph describes an operational process and continues without a field label.",
|
||||
70.0,
|
||||
700.0,
|
||||
10.0,
|
||||
350.0,
|
||||
),
|
||||
make_char(
|
||||
"It was split only because the text wrapped across a wide line.",
|
||||
455.0,
|
||||
700.0,
|
||||
10.0,
|
||||
300.0,
|
||||
),
|
||||
make_char(
|
||||
"Another sentence explains background context rather than a measurable property.",
|
||||
70.0,
|
||||
680.0,
|
||||
10.0,
|
||||
350.0,
|
||||
),
|
||||
make_char(
|
||||
"The neighboring phrase is not a value and should not form a table.",
|
||||
455.0,
|
||||
680.0,
|
||||
10.0,
|
||||
300.0,
|
||||
),
|
||||
make_char(
|
||||
"Finally, this narrative line keeps flowing with normal prose content.",
|
||||
70.0,
|
||||
660.0,
|
||||
10.0,
|
||||
350.0,
|
||||
),
|
||||
make_char(
|
||||
"It has punctuation and complete sentences on both sides of the gap.",
|
||||
455.0,
|
||||
660.0,
|
||||
10.0,
|
||||
300.0,
|
||||
),
|
||||
];
|
||||
|
||||
assert!(try_build_key_value_table_from_rows(&items, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_body_font_table_detected() {
|
||||
let items = vec![
|
||||
|
||||
Reference in New Issue
Block a user