Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3081f94e72 |
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.9.2",
|
||||
"version": "1.9.3",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
+24
-9
@@ -823,7 +823,8 @@ pub fn extract_tables_in_regions_mem(
|
||||
// symmetrically low under font-decode failure — this
|
||||
// guard breaks that symmetry by comparing against
|
||||
// bbox area, which is independent of extraction.
|
||||
if region_text_density_too_low(region_text_chars, region_area)
|
||||
if source != TableCandidateSource::KeyValue
|
||||
&& region_text_density_too_low(region_text_chars, region_area)
|
||||
&& !markdown_table_body_is_dense(&md)
|
||||
{
|
||||
return None;
|
||||
@@ -836,8 +837,10 @@ pub fn extract_tables_in_regions_mem(
|
||||
Some(TableCandidateIssue::LineRowUndercount)
|
||||
} else if wide_table_sparse_prefix_undercount(&md) {
|
||||
Some(TableCandidateIssue::SparseWideUndercount)
|
||||
} else if source != TableCandidateSource::Line
|
||||
&& text_cluster_column_undercount(&matched, shape)
|
||||
} else if !matches!(
|
||||
source,
|
||||
TableCandidateSource::Line | TableCandidateSource::KeyValue
|
||||
) && text_cluster_column_undercount(&matched, shape)
|
||||
{
|
||||
Some(TableCandidateIssue::TextColumnUndercount)
|
||||
} else if prose_grid_fragment_needs_ocr(&md) {
|
||||
@@ -886,6 +889,11 @@ pub fn extract_tables_in_regions_mem(
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
if let Some(table) = tables::try_build_key_value_table_from_rows(&matched, page_1idx) {
|
||||
if let Some(candidate) = evaluate(TableCandidateSource::KeyValue, &table) {
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
match select_table_candidate(&candidates) {
|
||||
Some(candidate) => page_results.push(RegionText {
|
||||
@@ -3758,6 +3766,7 @@ enum TableCandidateSource {
|
||||
Line,
|
||||
Heuristic,
|
||||
Column,
|
||||
KeyValue,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
@@ -3794,7 +3803,9 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
|
||||
return candidates.iter().find(|candidate| {
|
||||
matches!(
|
||||
candidate.source,
|
||||
TableCandidateSource::Heuristic | TableCandidateSource::Column
|
||||
TableCandidateSource::Heuristic
|
||||
| TableCandidateSource::Column
|
||||
| TableCandidateSource::KeyValue
|
||||
) && candidate.issue.is_none()
|
||||
&& candidate.shape.cols * 10 >= first.shape.cols * 13
|
||||
});
|
||||
@@ -3815,7 +3826,9 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
|
||||
if let Some(heuristic) = candidates.iter().find(|candidate| {
|
||||
matches!(
|
||||
candidate.source,
|
||||
TableCandidateSource::Heuristic | TableCandidateSource::Column
|
||||
TableCandidateSource::Heuristic
|
||||
| TableCandidateSource::Column
|
||||
| TableCandidateSource::KeyValue
|
||||
) && candidate.issue.is_none()
|
||||
&& heuristic_substantially_better(candidate.shape, accepted.shape)
|
||||
}) {
|
||||
@@ -3824,13 +3837,15 @@ fn select_table_candidate(candidates: &[TableCandidate]) -> Option<&TableCandida
|
||||
}
|
||||
|
||||
if accepted.source == TableCandidateSource::Heuristic {
|
||||
if let Some(column) = candidates.iter().find(|candidate| {
|
||||
candidate.source == TableCandidateSource::Column
|
||||
&& candidate.issue.is_none()
|
||||
if let Some(layout_candidate) = candidates.iter().find(|candidate| {
|
||||
matches!(
|
||||
candidate.source,
|
||||
TableCandidateSource::Column | TableCandidateSource::KeyValue
|
||||
) && candidate.issue.is_none()
|
||||
&& candidate.shape.cols >= accepted.shape.cols
|
||||
&& candidate.shape.rows > accepted.shape.rows
|
||||
}) {
|
||||
accepted = column;
|
||||
accepted = layout_candidate;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -548,6 +548,482 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
Some(Table::new(col_xs, row_ys, cells, item_indices))
|
||||
}
|
||||
|
||||
/// Build a region-scoped two-column key/value table from text baselines.
|
||||
///
|
||||
/// This intentionally lives outside the full-page heuristic detector. Layout
|
||||
/// callers already supplied a table-shaped bbox, and some real table regions
|
||||
/// are plain product/spec forms with only two visual columns. The main column
|
||||
/// fallback starts at four columns to avoid newspaper/prose false positives;
|
||||
/// this path keeps tighter key/value-specific guards instead.
|
||||
pub(crate) fn try_build_key_value_table_from_rows(items: &[TextItem], page: u32) -> Option<Table> {
|
||||
let page_items: Vec<RowItem> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| item.page == page && !item.text.trim().is_empty())
|
||||
.map(|(idx, item)| RowItem {
|
||||
index: idx,
|
||||
item: item.clone(),
|
||||
})
|
||||
.collect();
|
||||
|
||||
if page_items.len() < 4 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let median_font_size = median_f32(page_items.iter().map(|ri| ri.item.font_size).collect())
|
||||
.unwrap_or(10.0)
|
||||
.max(1.0);
|
||||
let y_tol = (median_font_size * 0.75).clamp(4.0, 9.0);
|
||||
let rows = group_key_value_visual_rows(page_items, y_tol);
|
||||
if rows.len() < 2 || rows.len() > 80 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let split_x = infer_key_value_split_x(&rows, median_font_size)?;
|
||||
let mut kv_rows: Vec<KeyValueRow> = Vec::new();
|
||||
let mut paired_rows = 0usize;
|
||||
let mut section_rows = 0usize;
|
||||
let mut left_label_like = 0usize;
|
||||
let mut left_starts = Vec::new();
|
||||
let mut right_starts = Vec::new();
|
||||
|
||||
for row in &rows {
|
||||
let mut left_items = Vec::new();
|
||||
let mut right_items = Vec::new();
|
||||
for item in &row.items {
|
||||
if item.item.x < split_x {
|
||||
left_items.push(item);
|
||||
} else {
|
||||
right_items.push(item);
|
||||
}
|
||||
}
|
||||
|
||||
let left = join_row_item_text(&left_items);
|
||||
let right = join_row_item_text(&right_items);
|
||||
if left.is_empty() && right.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut item_indices: Vec<usize> = row.items.iter().map(|ri| ri.index).collect();
|
||||
item_indices.sort_unstable();
|
||||
item_indices.dedup();
|
||||
|
||||
if !left.is_empty() && !right.is_empty() {
|
||||
paired_rows += 1;
|
||||
if looks_like_key_value_label(&left) {
|
||||
left_label_like += 1;
|
||||
}
|
||||
if let Some(x) = left_items.first().map(|ri| ri.item.x) {
|
||||
left_starts.push(x);
|
||||
}
|
||||
if let Some(x) = right_items.first().map(|ri| ri.item.x) {
|
||||
right_starts.push(x);
|
||||
}
|
||||
} else if !left.is_empty() {
|
||||
section_rows += 1;
|
||||
}
|
||||
|
||||
kv_rows.push(KeyValueRow {
|
||||
y: row.y,
|
||||
left,
|
||||
right,
|
||||
item_indices,
|
||||
});
|
||||
}
|
||||
|
||||
if kv_rows.len() < 2 || paired_rows < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let header_inferred = key_value_first_pair_is_header(&kv_rows);
|
||||
let data_pairs = if header_inferred {
|
||||
paired_rows.saturating_sub(1)
|
||||
} else {
|
||||
paired_rows
|
||||
};
|
||||
if data_pairs < 1 {
|
||||
return None;
|
||||
}
|
||||
|
||||
if section_rows > paired_rows * 2 + 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let label_rows_for_score = if header_inferred {
|
||||
paired_rows.saturating_sub(1)
|
||||
} else {
|
||||
paired_rows
|
||||
};
|
||||
let label_like_for_score = if header_inferred && !kv_rows.is_empty() {
|
||||
left_label_like.saturating_sub(1)
|
||||
} else {
|
||||
left_label_like
|
||||
};
|
||||
if label_rows_for_score >= 2 && label_like_for_score * 2 < label_rows_for_score {
|
||||
return None;
|
||||
}
|
||||
|
||||
let left_x = median_f32(left_starts).unwrap_or_else(|| {
|
||||
rows.iter()
|
||||
.flat_map(|row| row.items.iter().map(|ri| ri.item.x))
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
});
|
||||
let right_x = median_f32(right_starts).unwrap_or(split_x);
|
||||
if !left_x.is_finite() || !right_x.is_finite() || right_x - left_x < 40.0 {
|
||||
return None;
|
||||
}
|
||||
let right_cluster_count = significant_side_x_clusters(&rows, split_x, false);
|
||||
let marker_rows = marker_matrix_value_rows(&kv_rows);
|
||||
if (right_cluster_count >= 5 && paired_rows >= 3)
|
||||
|| (right_cluster_count >= 3 && marker_rows >= 3 && marker_rows * 2 >= paired_rows)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
|
||||
if key_value_rows_look_like_prose(&kv_rows, header_inferred) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut table_rows = Vec::new();
|
||||
let mut cells = Vec::new();
|
||||
let mut item_indices = Vec::new();
|
||||
|
||||
let mut start_idx = 0usize;
|
||||
if header_inferred {
|
||||
let header = &kv_rows[0];
|
||||
table_rows.push(header.y);
|
||||
cells.push(vec![header.left.clone(), header.right.clone()]);
|
||||
item_indices.extend(header.item_indices.iter().copied());
|
||||
start_idx = 1;
|
||||
} else {
|
||||
table_rows.push(kv_rows.first().map(|row| row.y + y_tol).unwrap_or(0.0));
|
||||
cells.push(vec!["Field".to_string(), "Value".to_string()]);
|
||||
}
|
||||
|
||||
for row in kv_rows.iter().skip(start_idx) {
|
||||
if !row.left.is_empty() && !row.right.is_empty() {
|
||||
table_rows.push(row.y);
|
||||
cells.push(vec![row.left.clone(), row.right.clone()]);
|
||||
item_indices.extend(row.item_indices.iter().copied());
|
||||
} else if !row.left.is_empty() {
|
||||
table_rows.push(row.y);
|
||||
cells.push(vec!["Section".to_string(), row.left.clone()]);
|
||||
item_indices.extend(row.item_indices.iter().copied());
|
||||
} else if !row.right.is_empty() {
|
||||
if let Some(last) = cells.last_mut() {
|
||||
if let Some(value) = last.get_mut(1) {
|
||||
if !value.trim().is_empty() {
|
||||
value.push(' ');
|
||||
}
|
||||
value.push_str(&row.right);
|
||||
item_indices.extend(row.item_indices.iter().copied());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if cells.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
item_indices.sort_unstable();
|
||||
item_indices.dedup();
|
||||
|
||||
log::debug!(
|
||||
"key-value table: {} rows, pairs={}, sections={}, split_x={:.1}",
|
||||
cells.len(),
|
||||
paired_rows,
|
||||
section_rows,
|
||||
split_x
|
||||
);
|
||||
|
||||
Some(Table::new(
|
||||
vec![left_x, right_x],
|
||||
table_rows,
|
||||
cells,
|
||||
item_indices,
|
||||
))
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct RowItem {
|
||||
index: usize,
|
||||
item: TextItem,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct VisualRow {
|
||||
y: f32,
|
||||
items: Vec<RowItem>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct KeyValueRow {
|
||||
y: f32,
|
||||
left: String,
|
||||
right: String,
|
||||
item_indices: Vec<usize>,
|
||||
}
|
||||
|
||||
fn group_key_value_visual_rows(mut items: Vec<RowItem>, y_tol: f32) -> Vec<VisualRow> {
|
||||
items.sort_by(|a, b| {
|
||||
b.item
|
||||
.y
|
||||
.total_cmp(&a.item.y)
|
||||
.then_with(|| a.item.x.total_cmp(&b.item.x))
|
||||
});
|
||||
|
||||
let mut rows: Vec<VisualRow> = Vec::new();
|
||||
for row_item in items {
|
||||
if let Some(row) = rows
|
||||
.iter_mut()
|
||||
.find(|row| (row.y - row_item.item.y).abs() <= y_tol)
|
||||
{
|
||||
let len = row.items.len() as f32;
|
||||
row.y = (row.y * len + row_item.item.y) / (len + 1.0);
|
||||
row.items.push(row_item);
|
||||
continue;
|
||||
}
|
||||
|
||||
rows.push(VisualRow {
|
||||
y: row_item.item.y,
|
||||
items: vec![row_item],
|
||||
});
|
||||
}
|
||||
|
||||
for row in &mut rows {
|
||||
row.items.sort_by(|a, b| a.item.x.total_cmp(&b.item.x));
|
||||
}
|
||||
rows.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
rows
|
||||
}
|
||||
|
||||
fn infer_key_value_split_x(rows: &[VisualRow], median_font_size: f32) -> Option<f32> {
|
||||
let min_gap = (median_font_size * 2.0).max(24.0);
|
||||
let mut splits = Vec::new();
|
||||
|
||||
for row in rows {
|
||||
if row.items.len() < 2 {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut best_gap = 0.0f32;
|
||||
let mut best_split = None;
|
||||
for pair in row.items.windows(2) {
|
||||
let left = &pair[0].item;
|
||||
let right = &pair[1].item;
|
||||
let left_right = left.x + left.width.max(0.0);
|
||||
let gap = right.x - left_right;
|
||||
if gap > best_gap {
|
||||
best_gap = gap;
|
||||
best_split = Some(left_right + gap / 2.0);
|
||||
}
|
||||
}
|
||||
|
||||
if best_gap >= min_gap {
|
||||
if let Some(split) = best_split {
|
||||
splits.push(split);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if splits.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
median_f32(splits)
|
||||
}
|
||||
|
||||
fn join_row_item_text(items: &[&RowItem]) -> String {
|
||||
let mut parts = Vec::new();
|
||||
for item in items {
|
||||
let trimmed = item.item.text.trim();
|
||||
if !trimmed.is_empty() {
|
||||
parts.push(trimmed);
|
||||
}
|
||||
}
|
||||
normalize_cell_text(&parts.join(" "))
|
||||
}
|
||||
|
||||
fn normalize_cell_text(text: &str) -> String {
|
||||
text.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
|
||||
fn key_value_first_pair_is_header(rows: &[KeyValueRow]) -> bool {
|
||||
let Some(first) = rows.first() else {
|
||||
return false;
|
||||
};
|
||||
if first.left.is_empty() || first.right.is_empty() {
|
||||
return false;
|
||||
}
|
||||
if !looks_like_key_value_header_cell(&first.left)
|
||||
|| !looks_like_key_value_header_cell(&first.right)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
rows.iter()
|
||||
.skip(1)
|
||||
.any(|row| !row.left.is_empty() && !row.right.is_empty())
|
||||
}
|
||||
|
||||
fn looks_like_key_value_header_cell(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 2 || trimmed.len() > 40 {
|
||||
return false;
|
||||
}
|
||||
let words = word_count_simple(trimmed);
|
||||
if !(1..=4).contains(&words) {
|
||||
return false;
|
||||
}
|
||||
let lower = trimmed.to_ascii_lowercase();
|
||||
if matches!(
|
||||
lower.as_str(),
|
||||
"yes" | "no" | "true" | "false" | "none" | "n/a" | "na"
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
trimmed.chars().any(|c| c.is_alphabetic())
|
||||
&& !trimmed.chars().any(|c| c.is_ascii_digit())
|
||||
&& !trimmed.ends_with(['.', ',', ';', ':'])
|
||||
}
|
||||
|
||||
fn looks_like_key_value_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 2 || trimmed.len() > 90 {
|
||||
return false;
|
||||
}
|
||||
let words = word_count_simple(trimmed);
|
||||
if words == 0 || words > 10 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.ends_with(['.', ',', ';']) {
|
||||
return false;
|
||||
}
|
||||
trimmed.chars().any(|c| c.is_alphabetic())
|
||||
}
|
||||
|
||||
fn key_value_rows_look_like_prose(rows: &[KeyValueRow], header_inferred: bool) -> bool {
|
||||
let mut long_sentence_cells = 0usize;
|
||||
let mut total_cells = 0usize;
|
||||
let mut total_chars = 0usize;
|
||||
let mut paired_rows = 0usize;
|
||||
let mut solo_prose_rows = 0usize;
|
||||
|
||||
for row in rows.iter().skip(usize::from(header_inferred)) {
|
||||
if !row.left.is_empty() && !row.right.is_empty() {
|
||||
paired_rows += 1;
|
||||
} else {
|
||||
let solo = if row.left.is_empty() {
|
||||
row.right.trim()
|
||||
} else {
|
||||
row.left.trim()
|
||||
};
|
||||
if solo.chars().count() > 70
|
||||
|| word_count_simple(solo) > 9
|
||||
|| (solo.chars().count() > 35 && solo.ends_with(['.', '!', '?']))
|
||||
{
|
||||
solo_prose_rows += 1;
|
||||
}
|
||||
}
|
||||
for cell in [&row.left, &row.right] {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.is_empty() {
|
||||
continue;
|
||||
}
|
||||
total_cells += 1;
|
||||
total_chars += trimmed.chars().count();
|
||||
if trimmed.chars().count() > 100
|
||||
|| (trimmed.chars().count() > 55 && trimmed.ends_with(['.', '!', '?']))
|
||||
{
|
||||
long_sentence_cells += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if paired_rows < 1 || total_cells == 0 {
|
||||
return true;
|
||||
}
|
||||
if solo_prose_rows >= 3 {
|
||||
return true;
|
||||
}
|
||||
|
||||
let avg_chars = total_chars as f32 / total_cells as f32;
|
||||
avg_chars > 75.0 || long_sentence_cells * 2 >= total_cells
|
||||
}
|
||||
|
||||
fn marker_matrix_value_rows(rows: &[KeyValueRow]) -> usize {
|
||||
rows.iter()
|
||||
.filter(|row| !row.left.is_empty() && compact_marker_value(&row.right))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn compact_marker_value(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.is_empty() || trimmed.chars().count() > 80 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.chars().any(|ch| ch.is_alphabetic()) {
|
||||
return false;
|
||||
}
|
||||
trimmed
|
||||
.chars()
|
||||
.any(|ch| ch.is_ascii_digit() || matches!(ch, '•' | '●' | '·'))
|
||||
}
|
||||
|
||||
fn significant_side_x_clusters(rows: &[VisualRow], split_x: f32, left_side: bool) -> usize {
|
||||
let mut xs = Vec::new();
|
||||
for row in rows {
|
||||
for item in &row.items {
|
||||
let is_left = item.item.x < split_x;
|
||||
if is_left == left_side {
|
||||
xs.push(item.item.x);
|
||||
}
|
||||
}
|
||||
}
|
||||
xs.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
let mut counts = Vec::new();
|
||||
let mut center = None::<f32>;
|
||||
let mut count = 0usize;
|
||||
for x in xs {
|
||||
match center {
|
||||
Some(current) if (x - current).abs() <= 8.0 => {
|
||||
center = Some((current * count as f32 + x) / (count as f32 + 1.0));
|
||||
count += 1;
|
||||
}
|
||||
Some(_) => {
|
||||
counts.push(count);
|
||||
center = Some(x);
|
||||
count = 1;
|
||||
}
|
||||
None => {
|
||||
center = Some(x);
|
||||
count = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
if count > 0 {
|
||||
counts.push(count);
|
||||
}
|
||||
|
||||
counts.into_iter().filter(|&count| count >= 2).count()
|
||||
}
|
||||
|
||||
fn word_count_simple(cell: &str) -> usize {
|
||||
cell.split_whitespace()
|
||||
.filter(|word| word.chars().any(|c| c.is_alphanumeric()))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn median_f32(mut values: Vec<f32>) -> Option<f32> {
|
||||
values.retain(|value| value.is_finite());
|
||||
if values.is_empty() {
|
||||
return None;
|
||||
}
|
||||
values.sort_by(|a, b| a.total_cmp(b));
|
||||
Some(values[values.len() / 2])
|
||||
}
|
||||
|
||||
fn merge_superscript_marker_rows(row_ys: &mut Vec<f32>, cells: &mut Vec<Vec<String>>) {
|
||||
let mut row_idx = 0;
|
||||
while row_idx < cells.len() {
|
||||
@@ -807,6 +1283,127 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_recovers_sectioned_specs_table() {
|
||||
let items = vec![
|
||||
make_char("Ordering Information", 69.0, 700.0, 9.0, 96.0),
|
||||
make_char("Package Contents", 69.0, 680.0, 9.0, 82.0),
|
||||
make_char(
|
||||
"CCH Adapter Panel with 3 m pigtail; installation guide",
|
||||
200.0,
|
||||
680.0,
|
||||
9.0,
|
||||
245.0,
|
||||
),
|
||||
make_char("Units per Delivery", 69.0, 660.0, 9.0, 78.0),
|
||||
make_char("1/1", 200.0, 660.0, 9.0, 18.0),
|
||||
];
|
||||
|
||||
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(md.contains("|Field|Value|"), "{md}");
|
||||
assert!(md.contains("|Section|Ordering Information|"), "{md}");
|
||||
assert!(
|
||||
md.contains(
|
||||
"|Package Contents|CCH Adapter Panel with 3 m pigtail; installation guide|"
|
||||
),
|
||||
"{md}"
|
||||
);
|
||||
assert!(md.contains("|Units per Delivery|1/1|"), "{md}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_preserves_two_column_header() {
|
||||
let items = vec![
|
||||
make_char("Media", 86.0, 700.0, 10.0, 36.0),
|
||||
make_char("Options", 311.0, 700.0, 10.0, 44.0),
|
||||
make_char("BACnet/IP (Annex J)", 86.0, 680.0, 10.0, 115.0),
|
||||
make_char("Register as Foreign Device", 311.0, 680.0, 10.0, 138.0),
|
||||
];
|
||||
|
||||
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(md.starts_with("|Media|Options|"), "{md}");
|
||||
assert!(
|
||||
md.contains("|BACnet/IP (Annex J)|Register as Foreign Device|"),
|
||||
"{md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_keeps_repeated_spec_sections() {
|
||||
let items = vec![
|
||||
make_char("1.33 DUAL VVT-i", 90.0, 700.0, 9.0, 82.0),
|
||||
make_char("Engine Code", 90.0, 682.0, 9.0, 62.0),
|
||||
make_char("1NR-FE", 406.0, 682.0, 9.0, 42.0),
|
||||
make_char("Type", 90.0, 664.0, 9.0, 24.0),
|
||||
make_char("Four cylinders in-line", 376.0, 664.0, 9.0, 104.0),
|
||||
make_char("1.6 VALVEMATIC", 90.0, 636.0, 9.0, 78.0),
|
||||
make_char("Engine Code", 90.0, 618.0, 9.0, 62.0),
|
||||
make_char("1ZR-FAE", 404.0, 618.0, 9.0, 44.0),
|
||||
];
|
||||
|
||||
let table = try_build_key_value_table_from_rows(&items, 1).unwrap();
|
||||
let md = table_to_markdown(&table);
|
||||
|
||||
assert!(md.contains("|Section|1.33 DUAL VVT-i|"), "{md}");
|
||||
assert!(md.contains("|Engine Code|1NR-FE|"), "{md}");
|
||||
assert!(md.contains("|Section|1.6 VALVEMATIC|"), "{md}");
|
||||
assert!(md.contains("|Engine Code|1ZR-FAE|"), "{md}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_key_value_builder_rejects_split_prose() {
|
||||
let items = vec![
|
||||
make_char(
|
||||
"This paragraph describes an operational process and continues without a field label.",
|
||||
70.0,
|
||||
700.0,
|
||||
10.0,
|
||||
350.0,
|
||||
),
|
||||
make_char(
|
||||
"It was split only because the text wrapped across a wide line.",
|
||||
455.0,
|
||||
700.0,
|
||||
10.0,
|
||||
300.0,
|
||||
),
|
||||
make_char(
|
||||
"Another sentence explains background context rather than a measurable property.",
|
||||
70.0,
|
||||
680.0,
|
||||
10.0,
|
||||
350.0,
|
||||
),
|
||||
make_char(
|
||||
"The neighboring phrase is not a value and should not form a table.",
|
||||
455.0,
|
||||
680.0,
|
||||
10.0,
|
||||
300.0,
|
||||
),
|
||||
make_char(
|
||||
"Finally, this narrative line keeps flowing with normal prose content.",
|
||||
70.0,
|
||||
660.0,
|
||||
10.0,
|
||||
350.0,
|
||||
),
|
||||
make_char(
|
||||
"It has punctuation and complete sentences on both sides of the gap.",
|
||||
455.0,
|
||||
660.0,
|
||||
10.0,
|
||||
300.0,
|
||||
),
|
||||
];
|
||||
|
||||
assert!(try_build_key_value_table_from_rows(&items, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_body_font_table_detected() {
|
||||
let items = vec![
|
||||
|
||||
Reference in New Issue
Block a user