Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ed3ab81f8e | ||
|
|
e165206fef |
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "firecrawl-pdf-inspector",
|
"name": "firecrawl-pdf-inspector",
|
||||||
"version": "0.7.0",
|
"version": "0.7.1",
|
||||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"types": "index.d.ts",
|
"types": "index.d.ts",
|
||||||
|
|||||||
@@ -1144,6 +1144,33 @@ pub(crate) fn find_first_table_row(
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Skip rows that have duplicate non-empty cells. These are spanning
|
||||||
|
// super-headers (e.g., "First Degree | First Degree | Higher Degree")
|
||||||
|
// that sit above the real column header row. Using them as the markdown
|
||||||
|
// header produces duplicate column names that downstream validation
|
||||||
|
// rejects. Only skip if a subsequent row looks like a better header
|
||||||
|
// (denser fill or has data).
|
||||||
|
if filled_count >= 2 && !has_data {
|
||||||
|
let mut text_counts: std::collections::HashMap<&str, usize> =
|
||||||
|
std::collections::HashMap::new();
|
||||||
|
for cell in &filled_cells {
|
||||||
|
*text_counts.entry(cell.trim()).or_insert(0) += 1;
|
||||||
|
}
|
||||||
|
let has_duplicates = text_counts.values().any(|&count| count >= 2);
|
||||||
|
if has_duplicates {
|
||||||
|
// Check if a later row is a better header candidate
|
||||||
|
let has_better_below = cells.iter().skip(row_idx + 1).take(3).any(|r| {
|
||||||
|
let next_filled = r.iter().filter(|c| !c.trim().is_empty()).count();
|
||||||
|
let next_fill = next_filled as f32 / total_cols as f32;
|
||||||
|
let next_numeric = r.iter().filter(|c| looks_like_number(c.trim())).count();
|
||||||
|
next_fill >= 0.4 || next_numeric >= 2
|
||||||
|
});
|
||||||
|
if has_better_below {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Data rows are definitely table content
|
// Data rows are definitely table content
|
||||||
if has_data {
|
if has_data {
|
||||||
first_table_row = row_idx;
|
first_table_row = row_idx;
|
||||||
|
|||||||
+132
-12
@@ -82,33 +82,42 @@ pub(crate) fn find_column_boundaries(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut columns = Vec::new();
|
// Track cluster membership: for each cluster, store the list of x positions
|
||||||
let mut cluster_items: Vec<f32> = vec![x_positions[0]];
|
let mut cluster_xs: Vec<Vec<f32>> = vec![vec![x_positions[0]]];
|
||||||
|
|
||||||
for &x in &x_positions[1..] {
|
for &x in &x_positions[1..] {
|
||||||
|
let last_cluster = cluster_xs.last().unwrap();
|
||||||
// For dense columns (gap-histogram triggered), use edge-based clustering:
|
// For dense columns (gap-histogram triggered), use edge-based clustering:
|
||||||
// compare with the last item to avoid center-drift that merges adjacent
|
// compare with the last item to avoid center-drift that merges adjacent
|
||||||
// narrow columns. For normal tables, use center-based (original behavior).
|
// narrow columns. For normal tables, use center-based (original behavior).
|
||||||
let reference = if use_edge_clustering {
|
let reference = if use_edge_clustering {
|
||||||
*cluster_items.last().unwrap()
|
*last_cluster.last().unwrap()
|
||||||
} else {
|
} else {
|
||||||
cluster_items.iter().sum::<f32>() / cluster_items.len() as f32
|
last_cluster.iter().sum::<f32>() / last_cluster.len() as f32
|
||||||
};
|
};
|
||||||
|
|
||||||
if x - reference > cluster_threshold {
|
if x - reference > cluster_threshold {
|
||||||
let cluster_center = cluster_items.iter().sum::<f32>() / cluster_items.len() as f32;
|
cluster_xs.push(vec![x]);
|
||||||
columns.push(cluster_center);
|
|
||||||
cluster_items = vec![x];
|
|
||||||
} else {
|
} else {
|
||||||
cluster_items.push(x);
|
cluster_xs.last_mut().unwrap().push(x);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Don't forget last cluster
|
// Numeric column merge pass: when a sparse cluster (few items, typically
|
||||||
if !cluster_items.is_empty() {
|
// header text) is adjacent to a dense numeric cluster and within 1.5×
|
||||||
columns.push(cluster_items.iter().sum::<f32>() / cluster_items.len() as f32);
|
// threshold, merge them. This fixes tables where multi-line wrapped
|
||||||
|
// headers have slightly different X positions than the data columns,
|
||||||
|
// causing the header and data to split into separate clusters.
|
||||||
|
let columns_before_merge = cluster_xs.len();
|
||||||
|
if columns_before_merge >= 3 {
|
||||||
|
cluster_xs = merge_numeric_adjacent_clusters(cluster_xs, items, cluster_threshold);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let columns: Vec<f32> = cluster_xs
|
||||||
|
.iter()
|
||||||
|
.map(|xs| xs.iter().sum::<f32>() / xs.len() as f32)
|
||||||
|
.collect();
|
||||||
|
|
||||||
// Filter columns - each should have multiple items
|
// Filter columns - each should have multiple items
|
||||||
let min_items_per_col = (items.len() / columns.len().max(1) / 4).max(2);
|
let min_items_per_col = (items.len() / columns.len().max(1) / 4).max(2);
|
||||||
let columns: Vec<f32> = columns
|
let columns: Vec<f32> = columns
|
||||||
@@ -123,8 +132,9 @@ pub(crate) fn find_column_boundaries(
|
|||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
log::debug!(
|
log::debug!(
|
||||||
" find_column_boundaries: {} columns before filter, threshold={:.1}, {} items",
|
" find_column_boundaries: {} columns (merged from {}), threshold={:.1}, {} items",
|
||||||
columns.len(),
|
columns.len(),
|
||||||
|
columns_before_merge,
|
||||||
cluster_threshold,
|
cluster_threshold,
|
||||||
items.len()
|
items.len()
|
||||||
);
|
);
|
||||||
@@ -148,6 +158,116 @@ pub(crate) fn find_column_boundaries(
|
|||||||
columns
|
columns
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Check if a text string looks like a number (digits, decimals, sign, comma).
|
||||||
|
fn is_numeric_text(s: &str) -> bool {
|
||||||
|
let s = s.trim();
|
||||||
|
if s.is_empty() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// Match patterns like: 8.23, -1.05, 9.99, 7.12, 100, 3,456.78, +5%, ---
|
||||||
|
// But NOT: BIO, Department, Core Courses
|
||||||
|
s.chars()
|
||||||
|
.all(|c| c.is_ascii_digit() || c == '.' || c == ',' || c == '-' || c == '+' || c == '%')
|
||||||
|
&& s.chars().any(|c| c.is_ascii_digit())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Merge adjacent X-position clusters when one is a sparse header cluster
|
||||||
|
/// and the other is a dense numeric data cluster. This prevents multi-line
|
||||||
|
/// wrapped headers from splitting a logical column into two clusters.
|
||||||
|
fn merge_numeric_adjacent_clusters(
|
||||||
|
mut clusters: Vec<Vec<f32>>,
|
||||||
|
items: &[(usize, &TextItem)],
|
||||||
|
threshold: f32,
|
||||||
|
) -> Vec<Vec<f32>> {
|
||||||
|
// For each cluster, compute: center, item count, numeric fraction
|
||||||
|
struct ClusterInfo {
|
||||||
|
center: f32,
|
||||||
|
count: usize,
|
||||||
|
numeric_frac: f32,
|
||||||
|
}
|
||||||
|
|
||||||
|
let compute_info = |xs: &[f32]| -> ClusterInfo {
|
||||||
|
let center = xs.iter().sum::<f32>() / xs.len() as f32;
|
||||||
|
// Count items and numeric fraction for items near this cluster center
|
||||||
|
let mut total = 0;
|
||||||
|
let mut numeric = 0;
|
||||||
|
for (_, item) in items {
|
||||||
|
if (item.x - center).abs() < threshold {
|
||||||
|
total += 1;
|
||||||
|
if is_numeric_text(&item.text) {
|
||||||
|
numeric += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ClusterInfo {
|
||||||
|
center,
|
||||||
|
count: total,
|
||||||
|
numeric_frac: if total > 0 {
|
||||||
|
numeric as f32 / total as f32
|
||||||
|
} else {
|
||||||
|
0.0
|
||||||
|
},
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Merge distance: allow merging clusters that are slightly beyond the
|
||||||
|
// original threshold. Use 1.5× threshold to catch header-vs-data splits.
|
||||||
|
let merge_dist = threshold * 1.5;
|
||||||
|
|
||||||
|
// Iterate and merge adjacent pairs. Use a simple left-to-right scan.
|
||||||
|
let mut merged = true;
|
||||||
|
while merged {
|
||||||
|
merged = false;
|
||||||
|
let mut i = 0;
|
||||||
|
while i + 1 < clusters.len() {
|
||||||
|
let info_a = compute_info(&clusters[i]);
|
||||||
|
let info_b = compute_info(&clusters[i + 1]);
|
||||||
|
let dist = (info_b.center - info_a.center).abs();
|
||||||
|
|
||||||
|
if dist > merge_dist {
|
||||||
|
i += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Determine if one cluster is sparse (header) and the other
|
||||||
|
// is dense and numeric (data). A cluster is "sparse" if it has
|
||||||
|
// significantly fewer items than the other.
|
||||||
|
let (sparse, dense) = if info_a.count < info_b.count {
|
||||||
|
(&info_a, &info_b)
|
||||||
|
} else {
|
||||||
|
(&info_b, &info_a)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Merge if the dense cluster is predominantly numeric (>50%)
|
||||||
|
// and the sparse cluster has at most 1/3 the items of the dense one.
|
||||||
|
let should_merge =
|
||||||
|
dense.numeric_frac > 0.50 && sparse.count <= dense.count / 2 && sparse.count <= 5;
|
||||||
|
|
||||||
|
if should_merge {
|
||||||
|
log::debug!(
|
||||||
|
" merging column clusters: center {:.1} ({} items, {:.0}% numeric) + {:.1} ({} items, {:.0}% numeric), dist={:.1}",
|
||||||
|
info_a.center,
|
||||||
|
info_a.count,
|
||||||
|
info_a.numeric_frac * 100.0,
|
||||||
|
info_b.center,
|
||||||
|
info_b.count,
|
||||||
|
info_b.numeric_frac * 100.0,
|
||||||
|
dist,
|
||||||
|
);
|
||||||
|
// Merge cluster i+1 into cluster i
|
||||||
|
let next = clusters.remove(i + 1);
|
||||||
|
clusters[i].extend(next);
|
||||||
|
merged = true;
|
||||||
|
// Don't increment i — check if the merged cluster can merge further
|
||||||
|
} else {
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
clusters
|
||||||
|
}
|
||||||
|
|
||||||
/// Find row boundaries by clustering Y positions
|
/// Find row boundaries by clustering Y positions
|
||||||
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||||
|
|||||||
BIN
Binary file not shown.
@@ -1438,6 +1438,42 @@ fn test_extract_tables_in_regions_nonexistent_page() {
|
|||||||
assert!(region.text.is_empty());
|
assert!(region.text.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_bits_pilani_page4_table_detection() {
|
||||||
|
// Page 4 (0-indexed 3) has a table with multi-line wrapped headers and
|
||||||
|
// numeric data columns. The heuristic detector previously failed because:
|
||||||
|
// 1. Header items at different X positions than data created extra column
|
||||||
|
// clusters (6 cols instead of 4)
|
||||||
|
// 2. Spanning super-header row ("First Degree | First Degree") produced
|
||||||
|
// duplicate header cells that looks_like_partial_table_ex rejected
|
||||||
|
let buf = std::fs::read("tests/fixtures/bits_pilani_feedback.pdf").unwrap();
|
||||||
|
let results =
|
||||||
|
extract_tables_in_regions_mem(&buf, &[(3, vec![[0.0, 0.0, 612.0, 792.0]])]).unwrap();
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
assert!(
|
||||||
|
!region.needs_ocr,
|
||||||
|
"Page 4 table should be detected, got needs_ocr=true"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
region.text.contains("BIO"),
|
||||||
|
"Should contain department name BIO"
|
||||||
|
);
|
||||||
|
assert!(region.text.contains("8.23"), "Should contain numeric data");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_bits_pilani_page8_table_detection() {
|
||||||
|
// Page 8 (0-indexed 7) has a numbered-row table that already worked.
|
||||||
|
// Verify it still works after changes.
|
||||||
|
let buf = std::fs::read("tests/fixtures/bits_pilani_feedback.pdf").unwrap();
|
||||||
|
let results =
|
||||||
|
extract_tables_in_regions_mem(&buf, &[(7, vec![[0.0, 0.0, 612.0, 792.0]])]).unwrap();
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
assert!(!region.needs_ocr, "Page 8 table should still be detected");
|
||||||
|
}
|
||||||
|
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
// extract_pages_markdown_mem tests
|
// extract_pages_markdown_mem tests
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
|
|||||||
Reference in New Issue
Block a user