Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
780efdb955 | ||
|
|
d0dd067e70 | ||
|
|
8282c2f8ee | ||
|
|
8e8ab4a19d | ||
|
|
8e3084183c |
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "firecrawl-pdf-inspector",
|
"name": "firecrawl-pdf-inspector",
|
||||||
"version": "0.3.6",
|
"version": "0.4.2",
|
||||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"types": "index.d.ts",
|
"types": "index.d.ts",
|
||||||
|
|||||||
+41
-9
@@ -255,7 +255,42 @@ pub fn extract_text_in_regions(
|
|||||||
page_regions: Vec<PageRegions>,
|
page_regions: Vec<PageRegions>,
|
||||||
) -> Result<Vec<PageRegionTexts>> {
|
) -> Result<Vec<PageRegionTexts>> {
|
||||||
let bytes: Vec<u8> = buffer.to_vec();
|
let bytes: Vec<u8> = buffer.to_vec();
|
||||||
let regions: Vec<(u32, Vec<[f32; 4]>)> = page_regions
|
let regions = parse_page_regions(&page_regions);
|
||||||
|
|
||||||
|
catch_panic("extract_text_in_regions", move || {
|
||||||
|
let results = pdf_inspector::extract_text_in_regions_mem(&bytes, ®ions)
|
||||||
|
.map_err(|e| to_napi_err(e, "extract_text_in_regions"))?;
|
||||||
|
Ok(to_page_region_texts(results))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Extract markdown tables within bounding-box regions from a PDF.
|
||||||
|
///
|
||||||
|
/// Like `extractTextInRegions` but runs table detection on items within each
|
||||||
|
/// region and returns markdown pipe-tables instead of flat text.
|
||||||
|
///
|
||||||
|
/// When table structure is detected, `text` contains a markdown pipe-table and
|
||||||
|
/// `needsOcr` is `false`. When no table is found, `text` is empty and
|
||||||
|
/// `needsOcr` is `true` so the caller can fall back to GPU OCR.
|
||||||
|
///
|
||||||
|
/// Coordinates are PDF points with top-left origin.
|
||||||
|
#[napi]
|
||||||
|
pub fn extract_tables_in_regions(
|
||||||
|
buffer: Buffer,
|
||||||
|
page_regions: Vec<PageRegions>,
|
||||||
|
) -> Result<Vec<PageRegionTexts>> {
|
||||||
|
let bytes: Vec<u8> = buffer.to_vec();
|
||||||
|
let regions = parse_page_regions(&page_regions);
|
||||||
|
|
||||||
|
catch_panic("extract_tables_in_regions", move || {
|
||||||
|
let results = pdf_inspector::extract_tables_in_regions_mem(&bytes, ®ions)
|
||||||
|
.map_err(|e| to_napi_err(e, "extract_tables_in_regions"))?;
|
||||||
|
Ok(to_page_region_texts(results))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_page_regions(page_regions: &[PageRegions]) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||||
|
page_regions
|
||||||
.iter()
|
.iter()
|
||||||
.map(|pr| {
|
.map(|pr| {
|
||||||
let bboxes: Vec<[f32; 4]> = pr
|
let bboxes: Vec<[f32; 4]> = pr
|
||||||
@@ -271,13 +306,11 @@ pub fn extract_text_in_regions(
|
|||||||
.collect();
|
.collect();
|
||||||
(pr.page, bboxes)
|
(pr.page, bboxes)
|
||||||
})
|
})
|
||||||
.collect();
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
catch_panic("extract_text_in_regions", move || {
|
fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<PageRegionTexts> {
|
||||||
let results = pdf_inspector::extract_text_in_regions_mem(&bytes, ®ions)
|
results
|
||||||
.map_err(|e| to_napi_err(e, "extract_text_in_regions"))?;
|
|
||||||
|
|
||||||
Ok(results
|
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.map(|page_result| PageRegionTexts {
|
.map(|page_result| PageRegionTexts {
|
||||||
page: page_result.page,
|
page: page_result.page,
|
||||||
@@ -290,6 +323,5 @@ pub fn extract_text_in_regions(
|
|||||||
})
|
})
|
||||||
.collect(),
|
.collect(),
|
||||||
})
|
})
|
||||||
.collect())
|
.collect()
|
||||||
})
|
|
||||||
}
|
}
|
||||||
|
|||||||
+416
@@ -444,6 +444,165 @@ pub fn extract_text_in_regions_mem(
|
|||||||
Ok(results)
|
Ok(results)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Extract tables within bounding-box regions from a PDF in memory.
|
||||||
|
///
|
||||||
|
/// Similar to [`extract_text_in_regions_mem`] but runs table detection on items
|
||||||
|
/// within each region and returns markdown pipe-tables instead of flat text.
|
||||||
|
///
|
||||||
|
/// When table structure is detected, `text` contains a markdown pipe-table and
|
||||||
|
/// `needs_ocr` is `false`. When no table is found (too few items, poor alignment,
|
||||||
|
/// GID fonts, etc.), `text` is empty and `needs_ocr` is `true` so the caller can
|
||||||
|
/// fall back to GPU OCR.
|
||||||
|
pub fn extract_tables_in_regions_mem(
|
||||||
|
buffer: &[u8],
|
||||||
|
page_regions: &[(u32, Vec<[f32; 4]>)],
|
||||||
|
) -> Result<Vec<PageRegionResult>, PdfError> {
|
||||||
|
validate_pdf_bytes(buffer)?;
|
||||||
|
let (doc, _page_count) = load_document_from_mem(buffer)?;
|
||||||
|
let pages = doc.get_pages();
|
||||||
|
|
||||||
|
let needed_pages: HashSet<u32> = page_regions.iter().map(|(p, _)| p + 1).collect();
|
||||||
|
let font_cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed_pages));
|
||||||
|
|
||||||
|
let mut items_by_page: HashMap<u32, Vec<TextItem>> = HashMap::new();
|
||||||
|
let mut page_heights: HashMap<u32, f32> = HashMap::new();
|
||||||
|
let mut gid_pages: HashSet<u32> = HashSet::new();
|
||||||
|
let mut page_thresholds: HashMap<u32, f32> = HashMap::new();
|
||||||
|
let mut rotated_pages: HashSet<u32> = HashSet::new();
|
||||||
|
|
||||||
|
for (page_num, &page_id) in pages.iter() {
|
||||||
|
if !needed_pages.contains(page_num) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let height = get_page_height(&doc, page_id).unwrap_or(792.0);
|
||||||
|
page_heights.insert(*page_num, height);
|
||||||
|
|
||||||
|
let ((mut items, _rects, _lines), has_gid, coords_rotated) =
|
||||||
|
extractor::content_stream::extract_page_text_items(
|
||||||
|
&doc,
|
||||||
|
page_id,
|
||||||
|
*page_num,
|
||||||
|
&font_cmaps,
|
||||||
|
false,
|
||||||
|
)?;
|
||||||
|
let threshold = text_utils::fix_letterspaced_items(&mut items);
|
||||||
|
if threshold > 0.10 {
|
||||||
|
page_thresholds.insert(*page_num, threshold);
|
||||||
|
}
|
||||||
|
if has_gid {
|
||||||
|
gid_pages.insert(*page_num);
|
||||||
|
}
|
||||||
|
if coords_rotated {
|
||||||
|
rotated_pages.insert(*page_num);
|
||||||
|
}
|
||||||
|
items_by_page.insert(*page_num, items);
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut results = Vec::with_capacity(page_regions.len());
|
||||||
|
|
||||||
|
for (page_0idx, regions) in page_regions {
|
||||||
|
let page_1idx = page_0idx + 1;
|
||||||
|
let items = items_by_page.get(&page_1idx);
|
||||||
|
let page_h = page_heights.get(&page_1idx).copied().unwrap_or(792.0);
|
||||||
|
let page_has_gid = gid_pages.contains(&page_1idx);
|
||||||
|
let coords = if rotated_pages.contains(&page_1idx) {
|
||||||
|
RegionCoordSpace::Rotated90Ccw
|
||||||
|
} else {
|
||||||
|
RegionCoordSpace::Standard
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut page_results = Vec::with_capacity(regions.len());
|
||||||
|
|
||||||
|
for rect in regions {
|
||||||
|
let [rx1, ry1, rx2, ry2] = *rect;
|
||||||
|
|
||||||
|
// If page has GID font issues, bail early
|
||||||
|
if page_has_gid {
|
||||||
|
page_results.push(RegionText {
|
||||||
|
text: String::new(),
|
||||||
|
needs_ocr: true,
|
||||||
|
});
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let matched: Vec<TextItem> = match items {
|
||||||
|
Some(items) => {
|
||||||
|
let bounds = region_bounds(rx1, ry1, rx2, ry2, page_h, coords);
|
||||||
|
items
|
||||||
|
.iter()
|
||||||
|
.filter(|item| region_overlaps_item(item, bounds))
|
||||||
|
.cloned()
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
None => Vec::new(),
|
||||||
|
};
|
||||||
|
|
||||||
|
if matched.is_empty() {
|
||||||
|
page_results.push(RegionText {
|
||||||
|
text: String::new(),
|
||||||
|
needs_ocr: true,
|
||||||
|
});
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Compute base_font_size as most common font size in the region
|
||||||
|
let base_font_size = {
|
||||||
|
let mut freq: HashMap<i32, usize> = HashMap::new();
|
||||||
|
for item in &matched {
|
||||||
|
*freq.entry((item.font_size * 10.0) as i32).or_default() += 1;
|
||||||
|
}
|
||||||
|
freq.into_iter()
|
||||||
|
.max_by_key(|(_, count)| *count)
|
||||||
|
.map(|(size, _)| size as f32 / 10.0)
|
||||||
|
.unwrap_or(12.0)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Run heuristic table detection; skip_body_font = false since
|
||||||
|
// the layout model already identified this region as a table.
|
||||||
|
let detected = tables::detect_tables(&matched, base_font_size, false);
|
||||||
|
|
||||||
|
if let Some(table) = detected.into_iter().next() {
|
||||||
|
let md = tables::table_to_markdown(&table);
|
||||||
|
if md.trim().is_empty() {
|
||||||
|
page_results.push(RegionText {
|
||||||
|
text: String::new(),
|
||||||
|
needs_ocr: true,
|
||||||
|
});
|
||||||
|
} else {
|
||||||
|
// needs_ocr fires on any of:
|
||||||
|
// - garbage text (non-alphanumeric heavy)
|
||||||
|
// - CID/Latin-1 mojibake
|
||||||
|
// - encoding issues (U+FFFD, dollar-as-space)
|
||||||
|
// - structural giveaways that the table is partial /
|
||||||
|
// mis-detected (numeric "header", empty header cells,
|
||||||
|
// duplicate header cells). Caught GLM-OCR-as-baseline
|
||||||
|
// scoring 0 TEDS on real prod tables in eval.
|
||||||
|
let needs_ocr = is_garbage_text(&md)
|
||||||
|
|| is_cid_garbage(&md)
|
||||||
|
|| detect_encoding_issues(&md)
|
||||||
|
|| looks_like_partial_table(&md);
|
||||||
|
page_results.push(RegionText {
|
||||||
|
text: if needs_ocr { String::new() } else { md },
|
||||||
|
needs_ocr,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
page_results.push(RegionText {
|
||||||
|
text: String::new(),
|
||||||
|
needs_ocr: true,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
results.push(PageRegionResult {
|
||||||
|
page: *page_0idx,
|
||||||
|
regions: page_results,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(results)
|
||||||
|
}
|
||||||
|
|
||||||
/// Get page height in points from MediaBox.
|
/// Get page height in points from MediaBox.
|
||||||
fn get_page_height(doc: &Document, page_id: lopdf::ObjectId) -> Option<f32> {
|
fn get_page_height(doc: &Document, page_id: lopdf::ObjectId) -> Option<f32> {
|
||||||
let page_dict = doc.get_dictionary(page_id).ok()?;
|
let page_dict = doc.get_dictionary(page_id).ok()?;
|
||||||
@@ -1041,6 +1200,263 @@ fn is_cid_garbage(text: &str) -> bool {
|
|||||||
high_latin * 5 >= total * 2 && ascii_letters * 3 < total
|
high_latin * 5 >= total * 2 && ascii_letters * 3 < total
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Detect markdown tables with suspicious structure that suggest the heuristic
|
||||||
|
/// missed/mangled rows or columns. Returns true when the caller should treat
|
||||||
|
/// the result as `needs_ocr` and fall back to GPU OCR.
|
||||||
|
///
|
||||||
|
/// Catches three failure modes observed in production:
|
||||||
|
///
|
||||||
|
/// 1. **Header row looks like a data row** — first row starts with a numeric
|
||||||
|
/// value (e.g. `|2|...`), suggesting we missed the actual header above it.
|
||||||
|
/// Real headers almost never start with a bare number.
|
||||||
|
///
|
||||||
|
/// 2. **Header has empty cells in a multi-column table** — e.g.
|
||||||
|
/// `|Position||Administration|Administration|` (3+ cols, ≥1 empty cell).
|
||||||
|
/// Indicates poor column boundary detection.
|
||||||
|
///
|
||||||
|
/// 3. **Header has duplicate non-empty cells** in a multi-column table —
|
||||||
|
/// e.g. `Administration|Administration` appearing as adjacent cells means
|
||||||
|
/// we collapsed multi-line headers wrong.
|
||||||
|
///
|
||||||
|
/// Conservative by design: a few false positives (perfectly fine tables flagged)
|
||||||
|
/// just mean we run GPU OCR which is the existing safe path.
|
||||||
|
fn looks_like_partial_table(markdown: &str) -> bool {
|
||||||
|
let lines: Vec<&str> = markdown.lines().filter(|l| l.starts_with('|')).collect();
|
||||||
|
if lines.len() < 2 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// Header is the first pipe-line; separator is the second
|
||||||
|
let header_line = lines[0];
|
||||||
|
let separator_line = lines.get(1).copied().unwrap_or("");
|
||||||
|
let is_separator = |l: &str| l.chars().all(|c| matches!(c, '|' | '-' | ' '));
|
||||||
|
if !is_separator(separator_line) {
|
||||||
|
// No separator after the first line — not a well-formed pipe-table.
|
||||||
|
// table_to_markdown always emits one when it returns content, so this
|
||||||
|
// shouldn't happen in practice. If it does, fall through to OCR.
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse header cells: split on '|', drop the leading/trailing empty pieces
|
||||||
|
let cells: Vec<&str> = header_line.split('|').map(|s| s.trim()).collect::<Vec<_>>();
|
||||||
|
// The first and last items are always empty (string starts and ends with '|')
|
||||||
|
if cells.len() < 3 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let header_cells: Vec<&str> = cells[1..cells.len() - 1].to_vec();
|
||||||
|
let n_cols = header_cells.len();
|
||||||
|
if n_cols < 2 {
|
||||||
|
// Single-column tables are usually lists/keys, not tables. Keep them
|
||||||
|
// (caller can decide), but multi-column header checks below don't
|
||||||
|
// apply.
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 1: header starts with a bare number (likely we missed
|
||||||
|
// the real header row above)
|
||||||
|
if let Some(first) = header_cells.first() {
|
||||||
|
let trimmed = first.trim();
|
||||||
|
if !trimmed.is_empty() && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 2: header has empty cells in a multi-column table
|
||||||
|
let empty_count = header_cells.iter().filter(|c| c.is_empty()).count();
|
||||||
|
if n_cols >= 3 && empty_count >= 1 {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 3: header has duplicate non-empty cells
|
||||||
|
let mut seen: std::collections::HashSet<&str> = std::collections::HashSet::new();
|
||||||
|
for cell in &header_cells {
|
||||||
|
if cell.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if !seen.insert(cell) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 4: first data row has many empty cells in a multi-column
|
||||||
|
// table. Real tables rarely have a leading row with most cells blank;
|
||||||
|
// when this happens it usually means the heuristic split a multi-row
|
||||||
|
// header (e.g. "Position\nAdministration (1986-1992) | Administration
|
||||||
|
// (1992-1998)") into a single-row header + a sparse data row.
|
||||||
|
if let Some(first_data_line) = lines.get(2) {
|
||||||
|
let data_cells: Vec<&str> = first_data_line
|
||||||
|
.split('|')
|
||||||
|
.map(|s| s.trim())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
if data_cells.len() >= 3 {
|
||||||
|
let data_inner = &data_cells[1..data_cells.len() - 1];
|
||||||
|
let empty_data = data_inner.iter().filter(|c| c.is_empty()).count();
|
||||||
|
// ≥3 cols, and a third or more of cells in the first data row
|
||||||
|
// are empty → very likely we mis-split a multi-row header.
|
||||||
|
if n_cols >= 3 && empty_data * 3 >= n_cols {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 5: cells flow as continuation paragraph (text wrapping
|
||||||
|
// mistaken for column structure). When a paragraph of prose gets mis-
|
||||||
|
// detected as a multi-column table, cells in the same column tend to
|
||||||
|
// start with lowercase letters or punctuation (continuation), not
|
||||||
|
// capital letters / digits (new entries). Real tables almost never
|
||||||
|
// have most data cells starting lowercase.
|
||||||
|
//
|
||||||
|
// Signal: ≥2 cols, ≥4 data rows, and ≥60% of non-empty data cells
|
||||||
|
// start with a lowercase letter or continuation punctuation.
|
||||||
|
let data_rows: Vec<Vec<&str>> = lines
|
||||||
|
.iter()
|
||||||
|
.skip(2) // header + separator
|
||||||
|
.map(|l| {
|
||||||
|
let parts: Vec<&str> = l.split('|').map(|s| s.trim()).collect();
|
||||||
|
if parts.len() >= 3 {
|
||||||
|
parts[1..parts.len() - 1].to_vec()
|
||||||
|
} else {
|
||||||
|
Vec::new()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.filter(|cells| !cells.is_empty())
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
if n_cols >= 2 && data_rows.len() >= 4 {
|
||||||
|
let mut continuation = 0;
|
||||||
|
let mut total = 0;
|
||||||
|
for row in &data_rows {
|
||||||
|
for cell in row {
|
||||||
|
let trimmed = cell.trim();
|
||||||
|
if trimmed.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
total += 1;
|
||||||
|
let first = trimmed.chars().next().unwrap();
|
||||||
|
// Continuation indicators: lowercase letter, common
|
||||||
|
// mid-sentence punctuation, closing quote
|
||||||
|
if first.is_lowercase()
|
||||||
|
|| matches!(first, ',' | '.' | ';' | ')' | '"' | '\'' | '”' | '’')
|
||||||
|
{
|
||||||
|
continuation += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if total > 0 && continuation * 5 >= total * 3 {
|
||||||
|
// ≥60% of cells look like sentence continuations → paragraph
|
||||||
|
// misread as table.
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod looks_like_partial_table_tests {
|
||||||
|
use super::looks_like_partial_table;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn good_table_passes() {
|
||||||
|
let md = "|Name|Year|Country|\n|---|---|---|\n|Alice|2020|US|\n|Bob|2021|UK|";
|
||||||
|
assert!(
|
||||||
|
!looks_like_partial_table(md),
|
||||||
|
"should not flag well-formed table"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn header_starting_with_number_is_partial() {
|
||||||
|
// Heuristic missed the actual header row above
|
||||||
|
let md = "|2|Cambodian Women for Peace|9,835|\n|---|---|---|\n|3|Association|711|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn header_with_empty_cells_in_3col_is_partial() {
|
||||||
|
// Empty cell in 3+ column header → bad column detection
|
||||||
|
let md =
|
||||||
|
"|Position||Administration|Administration|\n|---|---|---|---|\n|Senate|24|8.3|16.7|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn header_with_duplicate_cells_is_partial() {
|
||||||
|
// Duplicate "Administration" → collapsed multi-line header wrong
|
||||||
|
let md =
|
||||||
|
"|Position|Administration|Administration|Notes|\n|---|---|---|---|\n|Senate|24|16|x|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn two_column_with_one_empty_cell_passes() {
|
||||||
|
// Many real two-column tables have key-only rows; don't penalise.
|
||||||
|
let md = "|Key||\n|---|---|\n|Alice|123|\n|Bob|456|";
|
||||||
|
// Header "Key|" has one empty cell but only 2 cols total — keep it.
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn single_column_table_is_kept() {
|
||||||
|
// Single-column "tables" are common (lists). Caller can decide; we
|
||||||
|
// don't second-guess based on column count alone.
|
||||||
|
let md = "|Item|\n|---|\n|First|\n|Second|";
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn no_table_at_all_returns_true() {
|
||||||
|
// table_to_markdown should never produce this, but defensive — if
|
||||||
|
// there's no separator, treat as not-a-table.
|
||||||
|
let md = "Just some text\nWith multiple lines";
|
||||||
|
// No lines start with '|' so we return false (no header to inspect).
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn first_data_row_with_many_empty_cells_is_partial() {
|
||||||
|
// Multi-row header collapsed to single-row → first "data row" has
|
||||||
|
// most cells empty (the actual sub-header values).
|
||||||
|
let md = "|Government|No. of Seats|Aquino|Ramos|\n|---|---|---|---|\n|Position|||(1986-1992)|\n|Senate|24|8.3|16.7|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn first_data_row_with_one_empty_cell_in_4col_passes() {
|
||||||
|
// Real data rows can have one empty cell (e.g. missing value);
|
||||||
|
// only flag when ≥1/3 of cells are empty.
|
||||||
|
let md = "|A|B|C|D|\n|---|---|---|---|\n|x|y||z|\n|p|q|r|s|";
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn paragraph_misread_as_two_column_table_is_partial() {
|
||||||
|
// Real production failure: text-wrapped paragraph mis-detected as
|
||||||
|
// 2-col table. Each cell continues the previous one as prose.
|
||||||
|
let md = "|Approval is needed from the|Acquisitions of|\n\
|
||||||
|
|---|---|\n\
|
||||||
|
|Treasurer if the acquisition|residential and|\n\
|
||||||
|
|constitutes a \"significant|agricultural|\n\
|
||||||
|
|action,\" including acquiring an|land by foreign|\n\
|
||||||
|
|interest in different types of|persons must be|\n\
|
||||||
|
|land where the monetary|reported to the|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn real_multi_word_table_is_kept() {
|
||||||
|
// Real table with multi-word entries — cells start with capital
|
||||||
|
// letters / proper nouns, NOT lowercase continuations.
|
||||||
|
let md = "|Country|Capital|Notes|\n\
|
||||||
|
|---|---|---|\n\
|
||||||
|
|United States|Washington DC|Federal capital|\n\
|
||||||
|
|United Kingdom|London|City of London is a separate|\n\
|
||||||
|
|France|Paris|Île-de-France region|\n\
|
||||||
|
|Germany|Berlin|Reunified 1990|\n\
|
||||||
|
|Spain|Madrid|Largest city in Spain|";
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Analyse extracted items and rects for layout complexity.
|
/// Analyse extracted items and rects for layout complexity.
|
||||||
fn compute_layout_complexity(
|
fn compute_layout_complexity(
|
||||||
items: &[types::TextItem],
|
items: &[types::TextItem],
|
||||||
|
|||||||
+133
-1
@@ -14,6 +14,57 @@ use super::postprocess::clean_markdown;
|
|||||||
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
||||||
use super::MarkdownOptions;
|
use super::MarkdownOptions;
|
||||||
|
|
||||||
|
/// Pre-scan struct heading tags to find levels that are overused — i.e., tagged on
|
||||||
|
/// so many lines that they clearly represent body text, not real headings.
|
||||||
|
/// Returns the set of heading levels (1–6) that should be suppressed.
|
||||||
|
///
|
||||||
|
/// Some PDFs (e.g. British Academy grant guidance) tag every numbered paragraph
|
||||||
|
/// line as H2, producing hundreds of false headings. We detect this by checking
|
||||||
|
/// if any heading level accounts for >25% of tagged lines.
|
||||||
|
fn detect_overused_struct_heading_levels(
|
||||||
|
lines: &[TextLine],
|
||||||
|
struct_roles: Option<
|
||||||
|
&std::collections::HashMap<u32, std::collections::HashMap<i64, StructRole>>,
|
||||||
|
>,
|
||||||
|
) -> HashSet<usize> {
|
||||||
|
let mut overused = HashSet::new();
|
||||||
|
let Some(roles) = struct_roles else {
|
||||||
|
return overused;
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut level_counts: HashMap<usize, usize> = HashMap::new();
|
||||||
|
let mut total = 0usize;
|
||||||
|
|
||||||
|
for line in lines {
|
||||||
|
if let Some(role) = resolve_line_struct_role(line, roles) {
|
||||||
|
total += 1;
|
||||||
|
if let Some(level) = struct_role_heading_level(&role) {
|
||||||
|
*level_counts.entry(level).or_insert(0) += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if total < 20 {
|
||||||
|
return overused;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (&level, &count) in &level_counts {
|
||||||
|
let ratio = count as f32 / total as f32;
|
||||||
|
if ratio > 0.15 {
|
||||||
|
log::debug!(
|
||||||
|
"struct heading H{} overused: {}/{} lines ({:.0}%), suppressing",
|
||||||
|
level,
|
||||||
|
count,
|
||||||
|
total,
|
||||||
|
ratio * 100.0
|
||||||
|
);
|
||||||
|
overused.insert(level);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
overused
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-scan lines to find "isolated" ones: short lines with paragraph breaks both
|
/// Pre-scan lines to find "isolated" ones: short lines with paragraph breaks both
|
||||||
/// before and after. These are heading candidates even at body font size — common
|
/// before and after. These are heading candidates even at body font size — common
|
||||||
/// in academic papers ("Acknowledgements", "B.3 Prompt Engineering").
|
/// in academic papers ("Acknowledgements", "B.3 Prompt Engineering").
|
||||||
@@ -345,6 +396,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
|||||||
// lookahead in HeadingProcessor (prevNode/nextNode context).
|
// lookahead in HeadingProcessor (prevNode/nextNode context).
|
||||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||||
|
|
||||||
|
// Detect struct heading levels that are overused (body text mistagged as headings)
|
||||||
|
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
|
||||||
|
|
||||||
let mut output = String::new();
|
let mut output = String::new();
|
||||||
let mut current_page = 0u32;
|
let mut current_page = 0u32;
|
||||||
let mut prev_y = f32::MAX;
|
let mut prev_y = f32::MAX;
|
||||||
@@ -526,7 +580,10 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
|||||||
// Structure roles ADD headings (e.g. same-size text tagged H2) but do NOT
|
// Structure roles ADD headings (e.g. same-size text tagged H2) but do NOT
|
||||||
// suppress headings that the font heuristic would detect (some tagged PDFs
|
// suppress headings that the font heuristic would detect (some tagged PDFs
|
||||||
// mark obvious headings as P or Span).
|
// mark obvious headings as P or Span).
|
||||||
let struct_heading = struct_role.as_ref().and_then(struct_role_heading_level);
|
let struct_heading = struct_role
|
||||||
|
.as_ref()
|
||||||
|
.and_then(struct_role_heading_level)
|
||||||
|
.filter(|level| !overused_heading_levels.contains(level));
|
||||||
let heuristic_heading = if options.detect_headers
|
let heuristic_heading = if options.detect_headers
|
||||||
&& plain_trimmed.len() > 3
|
&& plain_trimmed.len() > 3
|
||||||
&& plain_trimmed.split_whitespace().count() <= 15
|
&& plain_trimmed.split_whitespace().count() <= 15
|
||||||
@@ -1260,4 +1317,79 @@ mod tests {
|
|||||||
"Should not have adjacent close/open fences: {md}"
|
"Should not have adjacent close/open fences: {md}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_overused_struct_heading_suppressed() {
|
||||||
|
// Simulate a PDF where H2 is mistagged on body text lines.
|
||||||
|
// 30 lines total: 5 tagged H1 (real headings), 20 tagged H2 (mistagged body),
|
||||||
|
// 5 tagged P.
|
||||||
|
let mut lines = Vec::new();
|
||||||
|
let mut page_roles = HashMap::new();
|
||||||
|
let mut mcid = 0i64;
|
||||||
|
|
||||||
|
for i in 0..30 {
|
||||||
|
let mut item = make_item(&format!("Line {i}"), 1, Some(mcid));
|
||||||
|
item.y = 700.0 - (i as f32 * 15.0);
|
||||||
|
lines.push(make_line(vec![item]));
|
||||||
|
|
||||||
|
let role = if i < 5 {
|
||||||
|
StructRole::H1
|
||||||
|
} else if i < 25 {
|
||||||
|
StructRole::H2
|
||||||
|
} else {
|
||||||
|
StructRole::P
|
||||||
|
};
|
||||||
|
page_roles.insert(mcid, role);
|
||||||
|
mcid += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut roles = HashMap::new();
|
||||||
|
roles.insert(1u32, page_roles);
|
||||||
|
|
||||||
|
let overused = detect_overused_struct_heading_levels(&lines, Some(&roles));
|
||||||
|
// H2 is on 20/30 = 67% of lines — should be suppressed
|
||||||
|
assert!(
|
||||||
|
overused.contains(&2),
|
||||||
|
"H2 should be detected as overused: {:?}",
|
||||||
|
overused
|
||||||
|
);
|
||||||
|
// H1 is on 5/30 = 17% — should also be suppressed at >15% threshold
|
||||||
|
assert!(
|
||||||
|
overused.contains(&1),
|
||||||
|
"H1 at 17% should also be suppressed: {:?}",
|
||||||
|
overused
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_normal_struct_headings_not_suppressed() {
|
||||||
|
// Normal document: a few headings, mostly body text
|
||||||
|
let mut lines = Vec::new();
|
||||||
|
let mut page_roles = HashMap::new();
|
||||||
|
let mut mcid = 0i64;
|
||||||
|
|
||||||
|
for i in 0..50 {
|
||||||
|
let mut item = make_item(&format!("Line {i}"), 1, Some(mcid));
|
||||||
|
item.y = 700.0 - (i as f32 * 14.0);
|
||||||
|
lines.push(make_line(vec![item]));
|
||||||
|
|
||||||
|
let role = if i % 10 == 0 {
|
||||||
|
StructRole::H1 // 5 headings out of 50 = 10%
|
||||||
|
} else {
|
||||||
|
StructRole::P
|
||||||
|
};
|
||||||
|
page_roles.insert(mcid, role);
|
||||||
|
mcid += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut roles = HashMap::new();
|
||||||
|
roles.insert(1u32, page_roles);
|
||||||
|
|
||||||
|
let overused = detect_overused_struct_heading_levels(&lines, Some(&roles));
|
||||||
|
assert!(
|
||||||
|
overused.is_empty(),
|
||||||
|
"No heading level should be overused: {:?}",
|
||||||
|
overused
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,9 +4,9 @@ use pdf_inspector::detector::{DetectionConfig, ScanStrategy};
|
|||||||
use pdf_inspector::extractor::group_into_lines;
|
use pdf_inspector::extractor::group_into_lines;
|
||||||
use pdf_inspector::types::TextLine;
|
use pdf_inspector::types::TextLine;
|
||||||
use pdf_inspector::{
|
use pdf_inspector::{
|
||||||
detect_pdf_type, extract_text, extract_text_in_regions_mem, extract_text_with_positions,
|
detect_pdf_type, extract_tables_in_regions_mem, extract_text, extract_text_in_regions_mem,
|
||||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
extract_text_with_positions, process_pdf_mem, process_pdf_with_options, to_markdown,
|
||||||
PdfType, TextItem,
|
MarkdownOptions, PdfError, PdfOptions, PdfType, TextItem,
|
||||||
};
|
};
|
||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
|
|
||||||
@@ -1344,3 +1344,95 @@ fn test_extract_regions_fast_vs_normal_comparison() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// extract_tables_in_regions_mem tests
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_in_regions_table_pdf() {
|
||||||
|
// tnagriculture has a clear table with district names and spice columns
|
||||||
|
let buf = std::fs::read("tests/fixtures/tnagriculture_06_12.pdf").unwrap();
|
||||||
|
let results =
|
||||||
|
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
assert_eq!(results[0].regions.len(), 1);
|
||||||
|
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
// Should detect a table with pipe-delimited markdown
|
||||||
|
if !region.needs_ocr {
|
||||||
|
assert!(
|
||||||
|
region.text.contains('|'),
|
||||||
|
"Table output should contain pipe delimiters"
|
||||||
|
);
|
||||||
|
// Should have separator row
|
||||||
|
assert!(
|
||||||
|
region.text.lines().any(|l| l.contains("---")),
|
||||||
|
"Table output should contain separator row"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_in_regions_non_table_region() {
|
||||||
|
// Use a small region that likely won't contain enough items for a table
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results =
|
||||||
|
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 50.0, 50.0]])]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
assert_eq!(results[0].regions.len(), 1);
|
||||||
|
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
// Small region with few items should fall back to needs_ocr
|
||||||
|
assert!(
|
||||||
|
region.needs_ocr,
|
||||||
|
"Non-table region should set needs_ocr = true"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
region.text.is_empty(),
|
||||||
|
"Non-table region should have empty text"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_in_regions_empty_region() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results = extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 0.0, 0.0]])]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
assert!(region.needs_ocr);
|
||||||
|
assert!(region.text.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_in_regions_identity_h_needs_ocr() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||||
|
let results =
|
||||||
|
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
assert!(region.needs_ocr, "Identity-H font should trigger needs_ocr");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_in_regions_not_a_pdf() {
|
||||||
|
let result =
|
||||||
|
extract_tables_in_regions_mem(b"not a pdf", &[(0, vec![[0.0, 0.0, 100.0, 100.0]])]);
|
||||||
|
assert!(result.is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_in_regions_nonexistent_page() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results =
|
||||||
|
extract_tables_in_regions_mem(&buf, &[(9999, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
let region = &results[0].regions[0];
|
||||||
|
assert!(region.needs_ocr);
|
||||||
|
assert!(region.text.is_empty());
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user