Compare commits

...
4 changed files with 875 additions and 13 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@firecrawl/pdf-inspector",
"version": "1.9.1",
"version": "1.9.3",
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
"main": "index.js",
"types": "index.d.ts",
+94 -8
View File
@@ -823,7 +823,8 @@ pub fn extract_tables_in_regions_mem(
// symmetrically low under font-decode failure — this
// guard breaks that symmetry by comparing against
// bbox area, which is independent of extraction.
if region_text_density_too_low(region_text_chars, region_area)
if source != TableCandidateSource::KeyValue
&& region_text_density_too_low(region_text_chars, region_area)
&& !markdown_table_body_is_dense(&md)
{
return None;
@@ -836,8 +837,10 @@ pub fn extract_tables_in_regions_mem(
Some(TableCandidateIssue::LineRowUndercount)
} else if wide_table_sparse_prefix_undercount(&md) {
Some(TableCandidateIssue::SparseWideUndercount)
} else if source != TableCandidateSource::Line
&& text_cluster_column_undercount(&matched, shape)
} else if !matches!(
source,
TableCandidateSource::Line | TableCandidateSource::KeyValue
) && text_cluster_column_undercount(&matched, shape)
{
Some(TableCandidateIssue::TextColumnUndercount)
} else if prose_grid_fragment_needs_ocr(&md) {
@@ -881,6 +884,16 @@ pub fn extract_tables_in_regions_mem(
{
candidates.push(candidate);
}
if let Some(table) = tables::try_build_table_from_columns(&matched, page_1idx) {
if let Some(candidate) = evaluate(TableCandidateSource::Column, &table) {
candidates.push(candidate);
}
}
if let Some(table) = tables::try_build_key_value_table_from_rows(&matched, page_1idx) {
if let Some(candidate) = evaluate(TableCandidateSource::KeyValue, &table) {
candidates.push(candidate);
}
}
match select_table_candidate(&candidates) {
Some(candidate) => page_results.push(RegionText {
@@ -3752,6 +3765,8 @@ enum TableCandidateSource {
Rect,
Line,
Heuristic,
Column,
KeyValue,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
@@ -3786,8 +3801,12 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
// serving a tidy-looking fragment.
if first.issue == Some(TableCandidateIssue::LineRowUndercount) {
return candidates.iter().find(|candidate| {
candidate.source == TableCandidateSource::Heuristic
&& candidate.issue.is_none()
matches!(
candidate.source,
TableCandidateSource::Heuristic
| TableCandidateSource::Column
| TableCandidateSource::KeyValue
) && candidate.issue.is_none()
&& candidate.shape.cols * 10 >= first.shape.cols * 13
});
}
@@ -3805,14 +3824,31 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
TableCandidateSource::Rect | TableCandidateSource::Line
) {
if let Some(heuristic) = candidates.iter().find(|candidate| {
candidate.source == TableCandidateSource::Heuristic
&& candidate.issue.is_none()
matches!(
candidate.source,
TableCandidateSource::Heuristic
| TableCandidateSource::Column
| TableCandidateSource::KeyValue
) && candidate.issue.is_none()
&& heuristic_substantially_better(candidate.shape, accepted.shape)
}) {
accepted = heuristic;
}
}
if accepted.source == TableCandidateSource::Heuristic {
if let Some(layout_candidate) = candidates.iter().find(|candidate| {
matches!(
candidate.source,
TableCandidateSource::Column | TableCandidateSource::KeyValue
) && candidate.issue.is_none()
&& candidate.shape.cols >= accepted.shape.cols
&& candidate.shape.rows > accepted.shape.rows
}) {
accepted = layout_candidate;
}
}
Some(accepted)
}
@@ -4314,7 +4350,9 @@ fn looks_like_partial_table_ex(markdown: &str, layout_assisted: bool) -> bool {
cell.trim().is_empty() && header_empty_indices.contains(&idx)
})
&& layout_assisted_empty_header_has_dense_body(markdown, n_cols);
if !sparse_row_shares_header_spacer {
let sparse_row_is_section_label = layout_assisted
&& layout_assisted_sparse_section_row_is_ok(data_inner, markdown, n_cols);
if !sparse_row_shares_header_spacer && !sparse_row_is_section_label {
return true;
}
}
@@ -4406,6 +4444,26 @@ fn layout_assisted_empty_header_has_dense_body(markdown: &str, n_cols: usize) ->
&& filled_cells * 100 >= total_cells * 45
}
fn layout_assisted_sparse_section_row_is_ok(row: &[&str], markdown: &str, n_cols: usize) -> bool {
let labels: Vec<&str> = row
.iter()
.map(|cell| cell.trim())
.filter(|cell| !cell.is_empty())
.collect();
if labels.len() != 1 {
return false;
}
let label = labels[0];
if label.len() > 40 || !label.chars().any(|ch| ch.is_alphabetic()) {
return false;
}
if label.ends_with('.') || label.ends_with('!') || label.ends_with('?') || label.ends_with(':')
{
return false;
}
layout_assisted_empty_header_has_dense_body(markdown, n_cols)
}
fn markdown_table_body_is_dense(markdown: &str) -> bool {
let rows = markdown_pipe_rows(markdown);
let data_rows: Vec<&Vec<&str>> = rows
@@ -4787,6 +4845,16 @@ mod table_candidate_selection_tests {
assert_eq!(selected.source, TableCandidateSource::Heuristic);
}
#[test]
fn prefers_clean_column_fallback_when_it_recovers_more_rows() {
let candidates = vec![
candidate(TableCandidateSource::Heuristic, 6, 5, None),
candidate(TableCandidateSource::Column, 7, 5, None),
];
let selected = select_table_candidate(&candidates).unwrap();
assert_eq!(selected.source, TableCandidateSource::Column);
}
#[test]
fn line_candidate_collapsing_captured_y_clusters_is_suspicious() {
let long = "value value value value value value value value value value value value";
@@ -5091,6 +5159,24 @@ mod looks_like_partial_table_tests {
);
}
#[test]
fn sparse_section_row_passes_when_layout_assisted_body_is_dense() {
let md = "|Properties|Conditions|Method|Typical values|Units|\n\
|---|---|---|---|---|\n\
|Rheology|||||\n\
|Melt Flow Rate|230 C/2.16 kg|ASTM D1238|3.0|g/10 min|\n\
|Tensile Stress at Yield|50 mm/min|ASTM D638|31|MPa|\n\
|Elongation at Yield|50 mm/min|ASTM D638|8|%|";
assert!(
looks_like_partial_table(md),
"strict mode rejects the sparse first row"
);
assert!(
!looks_like_partial_table_ex(md, true),
"layout-assisted should allow a short section label above dense table rows"
);
}
#[test]
fn paragraph_still_rejected_when_layout_assisted() {
// Paragraph detection is not relaxed — it's a genuine extraction issue.
+65 -4
View File
@@ -208,6 +208,24 @@ fn looks_like_compact_entry_label(cell: &str) -> bool {
(1..=6).contains(&words)
}
fn looks_like_plain_section_label(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.len() < 4 || trimmed.len() > 40 {
return false;
}
if trimmed.ends_with(['.', ',', ';', ':']) || trimmed.contains(|ch: char| ch.is_ascii_digit()) {
return false;
}
if trimmed.len() <= 4 && trimmed.chars().all(|ch| !ch.is_lowercase()) {
return false;
}
trimmed
.chars()
.all(|ch| ch.is_alphabetic() || ch.is_whitespace() || matches!(ch, '&' | '/' | '-'))
&& starts_with_uppercase_alpha(trimmed)
&& (1..=4).contains(&alpha_word_count(trimmed))
}
fn ends_like_incomplete_phrase(cell: &str) -> bool {
let lower = cell.trim_end().to_ascii_lowercase();
lower.ends_with(" and")
@@ -305,6 +323,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
.and_then(|r| r.first())
.map(|c| c.trim())
.unwrap_or("");
let header_filled = cleaned
.first()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(num_cols);
let looks_like_spanning_first_column_row = first_cell.is_empty()
&& row.len() >= 4
&& non_first_cells.len() == row.len().saturating_sub(1)
@@ -328,6 +350,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
&& non_first_cells
.iter()
.any(|cell| looks_like_compact_entry_label(cell));
let looks_like_section_label_row = !first_cell.is_empty()
&& filled_cells == 1
&& header_filled >= 3
&& looks_like_plain_section_label(first_cell);
// Classic continuation: first cell empty, content in other cells
let is_classic_continuation = first_cell.is_empty()
&& !non_first_cells.is_empty()
@@ -344,10 +370,6 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
.last()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(0);
let header_filled = cleaned
.first()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(num_cols);
// Merge when the row has significantly fewer filled cells than header.
// For wide tables (5+ cols), require ≤50% of header cells.
// For narrow tables (2-4 cols), require fewer than header cells.
@@ -369,6 +391,7 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
&& !looks_like_spanning_first_column_row
&& !looks_like_hierarchical_subrow
&& !looks_like_new_first_column_entry
&& !looks_like_section_label_row
&& !is_short_subheader;
let is_continuation = is_classic_continuation || is_wrapped_continuation;
@@ -514,6 +537,44 @@ mod tests {
assert!(cleaned[1][1].contains("continued text here"));
}
#[test]
fn test_clean_table_cells_first_column_section_label_not_merged() {
let cells = vec![
vec![
"Properties".into(),
"Conditions".into(),
"Method".into(),
"Typical values".into(),
"Units".into(),
],
vec![
"Melt Flow Rate".into(),
"230 C/2.16 kg".into(),
"ASTM D1238".into(),
"3.0".into(),
"g/10 min".into(),
],
vec![
"Mechanical".into(),
"".into(),
"".into(),
"".into(),
"".into(),
],
vec![
"Tensile Stress at Yield".into(),
"50 mm/min".into(),
"ASTM D638".into(),
"31".into(),
"MPa".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 4);
assert_eq!(cleaned[2][0], "Mechanical");
}
#[test]
fn test_clean_table_cells_short_subheader_not_merged() {
let cells = vec![
+715
View File
@@ -460,6 +460,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
item_indices.push(item_idx);
}
}
merge_superscript_marker_rows(&mut row_ys, &mut cells);
// Validate: need reasonable fill rate
let total_cells = row_ys.len() * columns.len();
@@ -547,6 +548,536 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
Some(Table::new(col_xs, row_ys, cells, item_indices))
}
/// Build a region-scoped two-column key/value table from text baselines.
///
/// This intentionally lives outside the full-page heuristic detector. Layout
/// callers already supplied a table-shaped bbox, and some real table regions
/// are plain product/spec forms with only two visual columns. The main column
/// fallback starts at four columns to avoid newspaper/prose false positives;
/// this path keeps tighter key/value-specific guards instead.
pub(crate) fn try_build_key_value_table_from_rows(items: &[TextItem], page: u32) -> Option<Table> {
let page_items: Vec<RowItem> = items
.iter()
.enumerate()
.filter(|(_, item)| item.page == page && !item.text.trim().is_empty())
.map(|(idx, item)| RowItem {
index: idx,
item: item.clone(),
})
.collect();
if page_items.len() < 4 {
return None;
}
let median_font_size = median_f32(page_items.iter().map(|ri| ri.item.font_size).collect())
.unwrap_or(10.0)
.max(1.0);
let y_tol = (median_font_size * 0.75).clamp(4.0, 9.0);
let rows = group_key_value_visual_rows(page_items, y_tol);
if rows.len() < 2 || rows.len() > 80 {
return None;
}
let split_x = infer_key_value_split_x(&rows, median_font_size)?;
let mut kv_rows: Vec<KeyValueRow> = Vec::new();
let mut paired_rows = 0usize;
let mut section_rows = 0usize;
let mut left_label_like = 0usize;
let mut left_starts = Vec::new();
let mut right_starts = Vec::new();
for row in &rows {
let mut left_items = Vec::new();
let mut right_items = Vec::new();
for item in &row.items {
if item.item.x < split_x {
left_items.push(item);
} else {
right_items.push(item);
}
}
let left = join_row_item_text(&left_items);
let right = join_row_item_text(&right_items);
if left.is_empty() && right.is_empty() {
continue;
}
let mut item_indices: Vec<usize> = row.items.iter().map(|ri| ri.index).collect();
item_indices.sort_unstable();
item_indices.dedup();
if !left.is_empty() && !right.is_empty() {
paired_rows += 1;
if looks_like_key_value_label(&left) {
left_label_like += 1;
}
if let Some(x) = left_items.first().map(|ri| ri.item.x) {
left_starts.push(x);
}
if let Some(x) = right_items.first().map(|ri| ri.item.x) {
right_starts.push(x);
}
} else if !left.is_empty() {
section_rows += 1;
}
kv_rows.push(KeyValueRow {
y: row.y,
left,
right,
item_indices,
});
}
if kv_rows.len() < 2 || paired_rows < 2 {
return None;
}
let header_inferred = key_value_first_pair_is_header(&kv_rows);
let data_pairs = if header_inferred {
paired_rows.saturating_sub(1)
} else {
paired_rows
};
if data_pairs < 1 {
return None;
}
if section_rows > paired_rows * 2 + 2 {
return None;
}
let label_rows_for_score = if header_inferred {
paired_rows.saturating_sub(1)
} else {
paired_rows
};
let label_like_for_score = if header_inferred && !kv_rows.is_empty() {
left_label_like.saturating_sub(1)
} else {
left_label_like
};
if label_rows_for_score >= 2 && label_like_for_score * 2 < label_rows_for_score {
return None;
}
let left_x = median_f32(left_starts).unwrap_or_else(|| {
rows.iter()
.flat_map(|row| row.items.iter().map(|ri| ri.item.x))
.fold(f32::INFINITY, f32::min)
});
let right_x = median_f32(right_starts).unwrap_or(split_x);
if !left_x.is_finite() || !right_x.is_finite() || right_x - left_x < 40.0 {
return None;
}
let right_cluster_count = significant_side_x_clusters(&rows, split_x, false);
let marker_rows = marker_matrix_value_rows(&kv_rows);
if (right_cluster_count >= 5 && paired_rows >= 3)
|| (right_cluster_count >= 3 && marker_rows >= 3 && marker_rows * 2 >= paired_rows)
{
return None;
}
if key_value_rows_look_like_prose(&kv_rows, header_inferred) {
return None;
}
let mut table_rows = Vec::new();
let mut cells = Vec::new();
let mut item_indices = Vec::new();
let mut start_idx = 0usize;
if header_inferred {
let header = &kv_rows[0];
table_rows.push(header.y);
cells.push(vec![header.left.clone(), header.right.clone()]);
item_indices.extend(header.item_indices.iter().copied());
start_idx = 1;
} else {
table_rows.push(kv_rows.first().map(|row| row.y + y_tol).unwrap_or(0.0));
cells.push(vec!["Field".to_string(), "Value".to_string()]);
}
for row in kv_rows.iter().skip(start_idx) {
if !row.left.is_empty() && !row.right.is_empty() {
table_rows.push(row.y);
cells.push(vec![row.left.clone(), row.right.clone()]);
item_indices.extend(row.item_indices.iter().copied());
} else if !row.left.is_empty() {
table_rows.push(row.y);
cells.push(vec!["Section".to_string(), row.left.clone()]);
item_indices.extend(row.item_indices.iter().copied());
} else if !row.right.is_empty() {
if let Some(last) = cells.last_mut() {
if let Some(value) = last.get_mut(1) {
if !value.trim().is_empty() {
value.push(' ');
}
value.push_str(&row.right);
item_indices.extend(row.item_indices.iter().copied());
}
}
}
}
if cells.len() < 2 {
return None;
}
item_indices.sort_unstable();
item_indices.dedup();
log::debug!(
"key-value table: {} rows, pairs={}, sections={}, split_x={:.1}",
cells.len(),
paired_rows,
section_rows,
split_x
);
Some(Table::new(
vec![left_x, right_x],
table_rows,
cells,
item_indices,
))
}
#[derive(Debug, Clone)]
struct RowItem {
index: usize,
item: TextItem,
}
#[derive(Debug, Clone)]
struct VisualRow {
y: f32,
items: Vec<RowItem>,
}
#[derive(Debug, Clone)]
struct KeyValueRow {
y: f32,
left: String,
right: String,
item_indices: Vec<usize>,
}
fn group_key_value_visual_rows(mut items: Vec<RowItem>, y_tol: f32) -> Vec<VisualRow> {
items.sort_by(|a, b| {
b.item
.y
.total_cmp(&a.item.y)
.then_with(|| a.item.x.total_cmp(&b.item.x))
});
let mut rows: Vec<VisualRow> = Vec::new();
for row_item in items {
if let Some(row) = rows
.iter_mut()
.find(|row| (row.y - row_item.item.y).abs() <= y_tol)
{
let len = row.items.len() as f32;
row.y = (row.y * len + row_item.item.y) / (len + 1.0);
row.items.push(row_item);
continue;
}
rows.push(VisualRow {
y: row_item.item.y,
items: vec![row_item],
});
}
for row in &mut rows {
row.items.sort_by(|a, b| a.item.x.total_cmp(&b.item.x));
}
rows.sort_by(|a, b| b.y.total_cmp(&a.y));
rows
}
fn infer_key_value_split_x(rows: &[VisualRow], median_font_size: f32) -> Option<f32> {
let min_gap = (median_font_size * 2.0).max(24.0);
let mut splits = Vec::new();
for row in rows {
if row.items.len() < 2 {
continue;
}
let mut best_gap = 0.0f32;
let mut best_split = None;
for pair in row.items.windows(2) {
let left = &pair[0].item;
let right = &pair[1].item;
let left_right = left.x + left.width.max(0.0);
let gap = right.x - left_right;
if gap > best_gap {
best_gap = gap;
best_split = Some(left_right + gap / 2.0);
}
}
if best_gap >= min_gap {
if let Some(split) = best_split {
splits.push(split);
}
}
}
if splits.len() < 2 {
return None;
}
median_f32(splits)
}
fn join_row_item_text(items: &[&RowItem]) -> String {
let mut parts = Vec::new();
for item in items {
let trimmed = item.item.text.trim();
if !trimmed.is_empty() {
parts.push(trimmed);
}
}
normalize_cell_text(&parts.join(" "))
}
fn normalize_cell_text(text: &str) -> String {
text.split_whitespace().collect::<Vec<_>>().join(" ")
}
fn key_value_first_pair_is_header(rows: &[KeyValueRow]) -> bool {
let Some(first) = rows.first() else {
return false;
};
if first.left.is_empty() || first.right.is_empty() {
return false;
}
if !looks_like_key_value_header_cell(&first.left)
|| !looks_like_key_value_header_cell(&first.right)
{
return false;
}
rows.iter()
.skip(1)
.any(|row| !row.left.is_empty() && !row.right.is_empty())
}
fn looks_like_key_value_header_cell(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.len() < 2 || trimmed.len() > 40 {
return false;
}
let words = word_count_simple(trimmed);
if !(1..=4).contains(&words) {
return false;
}
let lower = trimmed.to_ascii_lowercase();
if matches!(
lower.as_str(),
"yes" | "no" | "true" | "false" | "none" | "n/a" | "na"
) {
return false;
}
trimmed.chars().any(|c| c.is_alphabetic())
&& !trimmed.chars().any(|c| c.is_ascii_digit())
&& !trimmed.ends_with(['.', ',', ';', ':'])
}
fn looks_like_key_value_label(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.len() < 2 || trimmed.len() > 90 {
return false;
}
let words = word_count_simple(trimmed);
if words == 0 || words > 10 {
return false;
}
if trimmed.ends_with(['.', ',', ';']) {
return false;
}
trimmed.chars().any(|c| c.is_alphabetic())
}
fn key_value_rows_look_like_prose(rows: &[KeyValueRow], header_inferred: bool) -> bool {
let mut long_sentence_cells = 0usize;
let mut total_cells = 0usize;
let mut total_chars = 0usize;
let mut paired_rows = 0usize;
let mut solo_prose_rows = 0usize;
for row in rows.iter().skip(usize::from(header_inferred)) {
if !row.left.is_empty() && !row.right.is_empty() {
paired_rows += 1;
} else {
let solo = if row.left.is_empty() {
row.right.trim()
} else {
row.left.trim()
};
if solo.chars().count() > 70
|| word_count_simple(solo) > 9
|| (solo.chars().count() > 35 && solo.ends_with(['.', '!', '?']))
{
solo_prose_rows += 1;
}
}
for cell in [&row.left, &row.right] {
let trimmed = cell.trim();
if trimmed.is_empty() {
continue;
}
total_cells += 1;
total_chars += trimmed.chars().count();
if trimmed.chars().count() > 100
|| (trimmed.chars().count() > 55 && trimmed.ends_with(['.', '!', '?']))
{
long_sentence_cells += 1;
}
}
}
if paired_rows < 1 || total_cells == 0 {
return true;
}
if solo_prose_rows >= 3 {
return true;
}
let avg_chars = total_chars as f32 / total_cells as f32;
avg_chars > 75.0 || long_sentence_cells * 2 >= total_cells
}
fn marker_matrix_value_rows(rows: &[KeyValueRow]) -> usize {
rows.iter()
.filter(|row| !row.left.is_empty() && compact_marker_value(&row.right))
.count()
}
fn compact_marker_value(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.is_empty() || trimmed.chars().count() > 80 {
return false;
}
if trimmed.chars().any(|ch| ch.is_alphabetic()) {
return false;
}
trimmed
.chars()
.any(|ch| ch.is_ascii_digit() || matches!(ch, '•' | '●' | '·'))
}
fn significant_side_x_clusters(rows: &[VisualRow], split_x: f32, left_side: bool) -> usize {
let mut xs = Vec::new();
for row in rows {
for item in &row.items {
let is_left = item.item.x < split_x;
if is_left == left_side {
xs.push(item.item.x);
}
}
}
xs.sort_by(|a, b| a.total_cmp(b));
let mut counts = Vec::new();
let mut center = None::<f32>;
let mut count = 0usize;
for x in xs {
match center {
Some(current) if (x - current).abs() <= 8.0 => {
center = Some((current * count as f32 + x) / (count as f32 + 1.0));
count += 1;
}
Some(_) => {
counts.push(count);
center = Some(x);
count = 1;
}
None => {
center = Some(x);
count = 1;
}
}
}
if count > 0 {
counts.push(count);
}
counts.into_iter().filter(|&count| count >= 2).count()
}
fn word_count_simple(cell: &str) -> usize {
cell.split_whitespace()
.filter(|word| word.chars().any(|c| c.is_alphanumeric()))
.count()
}
fn median_f32(mut values: Vec<f32>) -> Option<f32> {
values.retain(|value| value.is_finite());
if values.is_empty() {
return None;
}
values.sort_by(|a, b| a.total_cmp(b));
Some(values[values.len() / 2])
}
fn merge_superscript_marker_rows(row_ys: &mut Vec<f32>, cells: &mut Vec<Vec<String>>) {
let mut row_idx = 0;
while row_idx < cells.len() {
let non_empty: Vec<(usize, String)> = cells[row_idx]
.iter()
.enumerate()
.filter_map(|(col_idx, cell)| {
let trimmed = cell.trim();
(!trimmed.is_empty()).then_some((col_idx, trimmed.to_string()))
})
.collect();
if non_empty.len() != 1 || !is_superscript_marker_cell(&non_empty[0].1) {
row_idx += 1;
continue;
}
let (marker_col, marker) = &non_empty[0];
let prev =
(row_idx > 0).then(|| (row_idx - 1, (row_ys[row_idx - 1] - row_ys[row_idx]).abs()));
let next = (row_idx + 1 < cells.len())
.then(|| (row_idx + 1, (row_ys[row_idx] - row_ys[row_idx + 1]).abs()));
let target = [prev, next]
.into_iter()
.flatten()
.filter(|(_, gap)| *gap <= 10.0)
.min_by(|(_, gap_a), (_, gap_b)| gap_a.total_cmp(gap_b))
.map(|(idx, _)| idx);
let Some(target_idx) = target else {
row_idx += 1;
continue;
};
let target_cell = &mut cells[target_idx][*marker_col];
if target_cell.trim().is_empty() {
*target_cell = marker.to_string();
} else {
target_cell.push_str(marker);
}
cells.remove(row_idx);
row_ys.remove(row_idx);
}
}
fn is_superscript_marker_cell(value: &str) -> bool {
let trimmed = value.trim();
!trimmed.is_empty()
&& trimmed.chars().count() <= 2
&& trimmed
.chars()
.all(|ch| matches!(ch, '*' | '#' | 'o' | 'O' | '°' | 'º' | '†' | '‡'))
}
/// What kind of structure a detected `Table` represents. Classification is
/// computed once at construction so consumers don't have to re-analyze the
/// cells (and stay consistent across detection backends).
@@ -689,6 +1220,190 @@ mod tests {
assert!(md.contains("|Cell 1|"));
}
#[test]
fn test_merge_superscript_marker_rows() {
let mut rows = vec![506.0, 500.0, 480.0];
let mut cells = vec![
vec!["".into(), "".into(), "*".into()],
vec!["Name".into(), "Method".into(), "Typical values".into()],
vec!["Flow".into(), "ASTM D1238".into(), "3.0".into()],
];
merge_superscript_marker_rows(&mut rows, &mut cells);
assert_eq!(rows, vec![500.0, 480.0]);
assert_eq!(cells[0][2], "Typical values*");
}
#[test]
fn test_column_builder_handles_borderless_specs_table() {
let items = vec![
make_char("*", 458.1, 544.2, 8.0, 4.4),
make_char("Properties", 36.0, 538.6, 12.0, 53.1),
make_char("Conditions", 195.8, 538.6, 12.0, 55.0),
make_char("Method", 297.2, 538.6, 12.0, 39.4),
make_char("Typical values", 384.1, 538.6, 12.0, 74.0),
make_char("Units", 510.6, 538.6, 8.0, 17.9),
make_char("Rheology", 36.0, 508.3, 10.0, 40.6),
make_char("o", 209.8, 492.5, 6.5, 3.5),
make_char("Melt Flow Rate", 36.0, 488.0, 10.0, 65.2),
make_char("230 ", 190.4, 488.0, 10.0, 19.4),
make_char("C/2.16 kg", 213.3, 488.0, 10.0, 42.8),
make_char("ASTM D1238", 288.4, 488.0, 10.0, 56.8),
make_char("3.0 ", 416.4, 488.0, 10.0, 16.9),
make_char("g/10 min", 504.1, 488.0, 10.0, 39.5),
make_char("Mechanical", 36.0, 451.5, 10.0, 48.3),
make_char("Tensile Stress at Yield", 36.0, 431.3, 10.0, 96.7),
make_char("50 mm/min", 197.9, 431.3, 10.0, 50.8),
make_char("ASTM D638", 291.2, 431.3, 10.0, 51.3),
make_char("31 ", 417.9, 431.3, 10.0, 13.9),
make_char("MPa", 514.7, 431.3, 10.0, 18.4),
make_char("Elongation at Yield", 36.0, 403.0, 10.0, 82.2),
make_char("50 mm/min", 197.9, 403.0, 10.0, 50.8),
make_char("ASTM D638", 291.2, 403.0, 10.0, 51.3),
make_char("8 ", 420.6, 403.0, 10.0, 8.5),
make_char("%", 519.1, 403.0, 10.0, 9.7),
make_char("Flexural Modulus", 36.0, 374.6, 10.0, 74.0),
make_char("ASTM D790", 291.2, 374.6, 10.0, 51.3),
make_char("1400", 412.4, 374.6, 10.0, 21.8),
make_char("MPa", 514.7, 374.6, 10.0, 18.4),
];
let table = try_build_table_from_columns(&items, 1).unwrap();
let md = table_to_markdown(&table);
assert!(
md.contains("|Properties|Conditions|Method|Typical values*|Units|"),
"{md}"
);
assert!(md.contains("|Mechanical|||||"), "{md}");
assert!(
md.contains("|Flexural Modulus||ASTM D790|1400|MPa|"),
"{md}"
);
}
#[test]
fn test_key_value_builder_recovers_sectioned_specs_table() {
let items = vec![
make_char("Ordering Information", 69.0, 700.0, 9.0, 96.0),
make_char("Package Contents", 69.0, 680.0, 9.0, 82.0),
make_char(
"CCH Adapter Panel with 3 m pigtail; installation guide",
200.0,
680.0,
9.0,
245.0,
),
make_char("Units per Delivery", 69.0, 660.0, 9.0, 78.0),
make_char("1/1", 200.0, 660.0, 9.0, 18.0),
];
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
let md = table_to_markdown(&table);
assert!(md.contains("|Field|Value|"), "{md}");
assert!(md.contains("|Section|Ordering Information|"), "{md}");
assert!(
md.contains(
"|Package Contents|CCH Adapter Panel with 3 m pigtail; installation guide|"
),
"{md}"
);
assert!(md.contains("|Units per Delivery|1/1|"), "{md}");
}
#[test]
fn test_key_value_builder_preserves_two_column_header() {
let items = vec![
make_char("Media", 86.0, 700.0, 10.0, 36.0),
make_char("Options", 311.0, 700.0, 10.0, 44.0),
make_char("BACnet/IP (Annex J)", 86.0, 680.0, 10.0, 115.0),
make_char("Register as Foreign Device", 311.0, 680.0, 10.0, 138.0),
];
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
let md = table_to_markdown(&table);
assert!(md.starts_with("|Media|Options|"), "{md}");
assert!(
md.contains("|BACnet/IP (Annex J)|Register as Foreign Device|"),
"{md}"
);
}
#[test]
fn test_key_value_builder_keeps_repeated_spec_sections() {
let items = vec![
make_char("1.33 DUAL VVT-i", 90.0, 700.0, 9.0, 82.0),
make_char("Engine Code", 90.0, 682.0, 9.0, 62.0),
make_char("1NR-FE", 406.0, 682.0, 9.0, 42.0),
make_char("Type", 90.0, 664.0, 9.0, 24.0),
make_char("Four cylinders in-line", 376.0, 664.0, 9.0, 104.0),
make_char("1.6 VALVEMATIC", 90.0, 636.0, 9.0, 78.0),
make_char("Engine Code", 90.0, 618.0, 9.0, 62.0),
make_char("1ZR-FAE", 404.0, 618.0, 9.0, 44.0),
];
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
let md = table_to_markdown(&table);
assert!(md.contains("|Section|1.33 DUAL VVT-i|"), "{md}");
assert!(md.contains("|Engine Code|1NR-FE|"), "{md}");
assert!(md.contains("|Section|1.6 VALVEMATIC|"), "{md}");
assert!(md.contains("|Engine Code|1ZR-FAE|"), "{md}");
}
#[test]
fn test_key_value_builder_rejects_split_prose() {
let items = vec![
make_char(
"This paragraph describes an operational process and continues without a field label.",
70.0,
700.0,
10.0,
350.0,
),
make_char(
"It was split only because the text wrapped across a wide line.",
455.0,
700.0,
10.0,
300.0,
),
make_char(
"Another sentence explains background context rather than a measurable property.",
70.0,
680.0,
10.0,
350.0,
),
make_char(
"The neighboring phrase is not a value and should not form a table.",
455.0,
680.0,
10.0,
300.0,
),
make_char(
"Finally, this narrative line keeps flowing with normal prose content.",
70.0,
660.0,
10.0,
350.0,
),
make_char(
"It has punctuation and complete sentences on both sides of the gap.",
455.0,
660.0,
10.0,
300.0,
),
];
assert!(try_build_key_value_table_from_rows(&items, 1).is_none());
}
#[test]
fn test_body_font_table_detected() {
let items = vec![