Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
abb0b925fb | ||
|
|
00c5c18e2a | ||
|
|
5159abe9c2 | ||
|
|
843a745460 | ||
|
|
780efdb955 | ||
|
|
d0dd067e70 |
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"name": "firecrawl-pdf-inspector",
|
"name": "firecrawl-pdf-inspector",
|
||||||
"version": "0.4.0",
|
"version": "0.6.0",
|
||||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||||
"main": "index.js",
|
"main": "index.js",
|
||||||
"types": "index.d.ts",
|
"types": "index.d.ts",
|
||||||
|
|||||||
+80
-17
@@ -5,6 +5,28 @@ use napi_derive::napi;
|
|||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
use std::panic;
|
use std::panic;
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Enums
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// PDF document type classification.
|
||||||
|
#[napi(string_enum)]
|
||||||
|
pub enum PdfType {
|
||||||
|
TextBased,
|
||||||
|
Scanned,
|
||||||
|
ImageBased,
|
||||||
|
Mixed,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Type of a positioned text item.
|
||||||
|
#[napi(string_enum)]
|
||||||
|
pub enum ItemType {
|
||||||
|
Text,
|
||||||
|
Image,
|
||||||
|
Link,
|
||||||
|
FormField,
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Result types
|
// Result types
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -12,7 +34,7 @@ use std::panic;
|
|||||||
/// Full PDF processing result with markdown and metadata.
|
/// Full PDF processing result with markdown and metadata.
|
||||||
#[napi(object)]
|
#[napi(object)]
|
||||||
pub struct PdfResult {
|
pub struct PdfResult {
|
||||||
pub pdf_type: String,
|
pub pdf_type: PdfType,
|
||||||
pub markdown: Option<String>,
|
pub markdown: Option<String>,
|
||||||
pub page_count: u32,
|
pub page_count: u32,
|
||||||
pub processing_time_ms: u32,
|
pub processing_time_ms: u32,
|
||||||
@@ -29,7 +51,7 @@ pub struct PdfResult {
|
|||||||
/// Lightweight PDF classification result.
|
/// Lightweight PDF classification result.
|
||||||
#[napi(object)]
|
#[napi(object)]
|
||||||
pub struct PdfClassification {
|
pub struct PdfClassification {
|
||||||
pub pdf_type: String,
|
pub pdf_type: PdfType,
|
||||||
pub page_count: u32,
|
pub page_count: u32,
|
||||||
/// 0-indexed page numbers that need OCR.
|
/// 0-indexed page numbers that need OCR.
|
||||||
pub pages_needing_ocr: Vec<u32>,
|
pub pages_needing_ocr: Vec<u32>,
|
||||||
@@ -49,7 +71,9 @@ pub struct TextItem {
|
|||||||
pub page: u32,
|
pub page: u32,
|
||||||
pub is_bold: bool,
|
pub is_bold: bool,
|
||||||
pub is_italic: bool,
|
pub is_italic: bool,
|
||||||
pub item_type: String,
|
pub item_type: ItemType,
|
||||||
|
/// URL for link items, `None` for other types.
|
||||||
|
pub link_url: Option<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
||||||
@@ -79,18 +103,18 @@ pub struct PageRegionTexts {
|
|||||||
// Helpers
|
// Helpers
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
fn pdf_type_string(t: pdf_inspector::PdfType) -> String {
|
fn convert_pdf_type(t: pdf_inspector::PdfType) -> PdfType {
|
||||||
match t {
|
match t {
|
||||||
pdf_inspector::PdfType::TextBased => "TextBased".to_string(),
|
pdf_inspector::PdfType::TextBased => PdfType::TextBased,
|
||||||
pdf_inspector::PdfType::Scanned => "Scanned".to_string(),
|
pdf_inspector::PdfType::Scanned => PdfType::Scanned,
|
||||||
pdf_inspector::PdfType::ImageBased => "ImageBased".to_string(),
|
pdf_inspector::PdfType::ImageBased => PdfType::ImageBased,
|
||||||
pdf_inspector::PdfType::Mixed => "Mixed".to_string(),
|
pdf_inspector::PdfType::Mixed => PdfType::Mixed,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||||
PdfResult {
|
PdfResult {
|
||||||
pdf_type: pdf_type_string(r.pdf_type),
|
pdf_type: convert_pdf_type(r.pdf_type),
|
||||||
markdown: r.markdown,
|
markdown: r.markdown,
|
||||||
page_count: r.page_count,
|
page_count: r.page_count,
|
||||||
processing_time_ms: r.processing_time_ms as u32,
|
processing_time_ms: r.processing_time_ms as u32,
|
||||||
@@ -104,12 +128,12 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn item_type_string(t: &pdf_inspector::types::ItemType) -> String {
|
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||||
match t {
|
match t {
|
||||||
pdf_inspector::types::ItemType::Text => "text".into(),
|
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||||
pdf_inspector::types::ItemType::Image => "image".into(),
|
pdf_inspector::types::ItemType::Image => (ItemType::Image, None),
|
||||||
pdf_inspector::types::ItemType::Link(url) => format!("link:{url}"),
|
pdf_inspector::types::ItemType::Link(url) => (ItemType::Link, Some(url.clone())),
|
||||||
pdf_inspector::types::ItemType::FormField => "form_field".into(),
|
pdf_inspector::types::ItemType::FormField => (ItemType::FormField, None),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -181,7 +205,7 @@ pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
|||||||
let result =
|
let result =
|
||||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||||
Ok(PdfClassification {
|
Ok(PdfClassification {
|
||||||
pdf_type: pdf_type_string(result.pdf_type),
|
pdf_type: convert_pdf_type(result.pdf_type),
|
||||||
page_count: result.page_count,
|
page_count: result.page_count,
|
||||||
pages_needing_ocr: result.pages_needing_ocr,
|
pages_needing_ocr: result.pages_needing_ocr,
|
||||||
confidence: result.confidence as f64,
|
confidence: result.confidence as f64,
|
||||||
@@ -222,7 +246,9 @@ pub fn extract_text_with_positions(
|
|||||||
|
|
||||||
Ok(items
|
Ok(items
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.map(|item| TextItem {
|
.map(|item| {
|
||||||
|
let (item_type, link_url) = convert_item_type(&item.item_type);
|
||||||
|
TextItem {
|
||||||
text: item.text,
|
text: item.text,
|
||||||
x: item.x as f64,
|
x: item.x as f64,
|
||||||
y: item.y as f64,
|
y: item.y as f64,
|
||||||
@@ -233,7 +259,9 @@ pub fn extract_text_with_positions(
|
|||||||
page: item.page,
|
page: item.page,
|
||||||
is_bold: item.is_bold,
|
is_bold: item.is_bold,
|
||||||
is_italic: item.is_italic,
|
is_italic: item.is_italic,
|
||||||
item_type: item_type_string(&item.item_type),
|
item_type,
|
||||||
|
link_url,
|
||||||
|
}
|
||||||
})
|
})
|
||||||
.collect())
|
.collect())
|
||||||
})
|
})
|
||||||
@@ -289,6 +317,41 @@ pub fn extract_tables_in_regions(
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Per-page markdown extraction result.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct PageMarkdownResult {
|
||||||
|
/// 0-indexed page number.
|
||||||
|
pub page: u32,
|
||||||
|
/// Formatted markdown for this page.
|
||||||
|
pub markdown: String,
|
||||||
|
/// `true` when text on this page is unreliable.
|
||||||
|
pub needs_ocr: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Extract formatted markdown for specific pages of a PDF.
|
||||||
|
///
|
||||||
|
/// Returns per-page markdown so callers can mix direct extraction
|
||||||
|
/// (for simple text pages) with GPU OCR (for complex/scanned pages).
|
||||||
|
///
|
||||||
|
/// Font statistics are computed from the full document so header
|
||||||
|
/// detection is consistent across pages.
|
||||||
|
#[napi]
|
||||||
|
pub fn extract_pages_markdown(buffer: Buffer, pages: Vec<u32>) -> Result<Vec<PageMarkdownResult>> {
|
||||||
|
let bytes: Vec<u8> = buffer.to_vec();
|
||||||
|
catch_panic("extract_pages_markdown", move || {
|
||||||
|
let results = pdf_inspector::extract_pages_markdown_mem(&bytes, &pages)
|
||||||
|
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||||
|
Ok(results
|
||||||
|
.into_iter()
|
||||||
|
.map(|r| PageMarkdownResult {
|
||||||
|
page: r.page,
|
||||||
|
markdown: r.markdown,
|
||||||
|
needs_ocr: r.needs_ocr,
|
||||||
|
})
|
||||||
|
.collect())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_page_regions(page_regions: &[PageRegions]) -> Vec<(u32, Vec<[f32; 4]>)> {
|
fn parse_page_regions(page_regions: &[PageRegions]) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||||
page_regions
|
page_regions
|
||||||
.iter()
|
.iter()
|
||||||
|
|||||||
+510
-14
@@ -299,6 +299,115 @@ pub fn classify_pdf_mem(buffer: &[u8]) -> Result<PdfClassification, PdfError> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Per-page markdown extraction
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
/// Per-page markdown extraction result.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct PageMarkdown {
|
||||||
|
/// 0-indexed page number.
|
||||||
|
pub page: u32,
|
||||||
|
/// Formatted markdown for this page.
|
||||||
|
pub markdown: String,
|
||||||
|
/// `true` when text on this page is unreliable (GID-encoded fonts,
|
||||||
|
/// encoding issues, garbage text, or empty extraction).
|
||||||
|
pub needs_ocr: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Extract formatted markdown for specific pages of a PDF.
|
||||||
|
///
|
||||||
|
/// Unlike [`process_pdf_mem`] which returns one concatenated markdown string,
|
||||||
|
/// this returns per-page markdown so callers can mix direct extraction
|
||||||
|
/// (for simple text pages) with GPU OCR (for complex/scanned pages).
|
||||||
|
///
|
||||||
|
/// Font statistics are computed from the full document so header
|
||||||
|
/// detection thresholds are consistent regardless of which pages are
|
||||||
|
/// requested. Per-page `needs_ocr` is set when the page has GID-encoded
|
||||||
|
/// fonts, encoding issues, or garbage text.
|
||||||
|
pub fn extract_pages_markdown_mem(
|
||||||
|
buffer: &[u8],
|
||||||
|
pages: &[u32],
|
||||||
|
) -> Result<Vec<PageMarkdown>, PdfError> {
|
||||||
|
validate_pdf_bytes(buffer)?;
|
||||||
|
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||||
|
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||||
|
|
||||||
|
// Extract ALL pages to get accurate, document-wide font stats.
|
||||||
|
let ((all_items, all_rects, _all_lines), page_thresholds, gid_pages) =
|
||||||
|
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||||
|
|
||||||
|
// Compute font stats from full document (cross-page consistency).
|
||||||
|
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||||
|
|
||||||
|
let mut results = Vec::with_capacity(pages.len());
|
||||||
|
|
||||||
|
for &page_0idx in pages {
|
||||||
|
// Out-of-range pages → empty + needs_ocr
|
||||||
|
if page_0idx >= page_count {
|
||||||
|
results.push(PageMarkdown {
|
||||||
|
page: page_0idx,
|
||||||
|
markdown: String::new(),
|
||||||
|
needs_ocr: true,
|
||||||
|
});
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let page_1idx = page_0idx + 1;
|
||||||
|
|
||||||
|
// Filter items/rects for this page only
|
||||||
|
let page_items: Vec<TextItem> = all_items
|
||||||
|
.iter()
|
||||||
|
.filter(|i| i.page == page_1idx)
|
||||||
|
.cloned()
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let page_rects: Vec<PdfRect> = all_rects
|
||||||
|
.iter()
|
||||||
|
.filter(|r| r.page == page_1idx)
|
||||||
|
.cloned()
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let has_gid = gid_pages.contains(&page_1idx);
|
||||||
|
|
||||||
|
// Build markdown with document-wide font stats
|
||||||
|
let options = MarkdownOptions {
|
||||||
|
base_font_size: Some(font_stats.most_common_size),
|
||||||
|
include_page_numbers: false,
|
||||||
|
strip_headers_footers: false,
|
||||||
|
..MarkdownOptions::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let md = markdown::to_markdown_from_items_with_rects_and_lines(
|
||||||
|
page_items,
|
||||||
|
options,
|
||||||
|
&page_rects,
|
||||||
|
&[],
|
||||||
|
&page_thresholds,
|
||||||
|
None,
|
||||||
|
&[],
|
||||||
|
);
|
||||||
|
|
||||||
|
let needs_ocr = md.trim().is_empty()
|
||||||
|
|| has_gid
|
||||||
|
|| is_garbage_text(&md)
|
||||||
|
|| is_cid_garbage(&md)
|
||||||
|
|| detect_encoding_issues(&md);
|
||||||
|
|
||||||
|
results.push(PageMarkdown {
|
||||||
|
page: page_0idx,
|
||||||
|
markdown: if needs_ocr { String::new() } else { md },
|
||||||
|
needs_ocr,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(results)
|
||||||
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Region-based text extraction (for hybrid OCR pipelines)
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
/// Result for a single region's text extraction.
|
/// Result for a single region's text extraction.
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
pub struct RegionText {
|
pub struct RegionText {
|
||||||
@@ -399,7 +508,7 @@ pub fn extract_text_in_regions_mem(
|
|||||||
let page_1idx = page_0idx + 1;
|
let page_1idx = page_0idx + 1;
|
||||||
let items = items_by_page.get(&page_1idx);
|
let items = items_by_page.get(&page_1idx);
|
||||||
let page_h = page_heights.get(&page_1idx).copied().unwrap_or(792.0);
|
let page_h = page_heights.get(&page_1idx).copied().unwrap_or(792.0);
|
||||||
let page_has_gid = gid_pages.contains(&page_1idx);
|
let _page_has_gid = gid_pages.contains(&page_1idx);
|
||||||
let adaptive_threshold = page_thresholds.get(&page_1idx).copied().unwrap_or(0.10);
|
let adaptive_threshold = page_thresholds.get(&page_1idx).copied().unwrap_or(0.10);
|
||||||
let coords = if rotated_pages.contains(&page_1idx) {
|
let coords = if rotated_pages.contains(&page_1idx) {
|
||||||
RegionCoordSpace::Rotated90Ccw
|
RegionCoordSpace::Rotated90Ccw
|
||||||
@@ -426,8 +535,10 @@ pub fn extract_text_in_regions_mem(
|
|||||||
None => String::new(),
|
None => String::new(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Check per-region text quality instead of blanket page-level
|
||||||
|
// GID rejection. A GID font in a logo elsewhere on the page
|
||||||
|
// shouldn't force GPU OCR for clean text regions.
|
||||||
let needs_ocr = text.trim().is_empty()
|
let needs_ocr = text.trim().is_empty()
|
||||||
|| page_has_gid
|
|
||||||
|| is_garbage_text(&text)
|
|| is_garbage_text(&text)
|
||||||
|| is_cid_garbage(&text)
|
|| is_cid_garbage(&text)
|
||||||
|| detect_encoding_issues(&text);
|
|| detect_encoding_issues(&text);
|
||||||
@@ -504,7 +615,7 @@ pub fn extract_tables_in_regions_mem(
|
|||||||
let page_1idx = page_0idx + 1;
|
let page_1idx = page_0idx + 1;
|
||||||
let items = items_by_page.get(&page_1idx);
|
let items = items_by_page.get(&page_1idx);
|
||||||
let page_h = page_heights.get(&page_1idx).copied().unwrap_or(792.0);
|
let page_h = page_heights.get(&page_1idx).copied().unwrap_or(792.0);
|
||||||
let page_has_gid = gid_pages.contains(&page_1idx);
|
let _page_has_gid = gid_pages.contains(&page_1idx);
|
||||||
let coords = if rotated_pages.contains(&page_1idx) {
|
let coords = if rotated_pages.contains(&page_1idx) {
|
||||||
RegionCoordSpace::Rotated90Ccw
|
RegionCoordSpace::Rotated90Ccw
|
||||||
} else {
|
} else {
|
||||||
@@ -516,14 +627,14 @@ pub fn extract_tables_in_regions_mem(
|
|||||||
for rect in regions {
|
for rect in regions {
|
||||||
let [rx1, ry1, rx2, ry2] = *rect;
|
let [rx1, ry1, rx2, ry2] = *rect;
|
||||||
|
|
||||||
// If page has GID font issues, bail early
|
// Note: we intentionally DO NOT bail on page_has_gid here.
|
||||||
if page_has_gid {
|
// The GID flag means some font on the page uses unresolvable
|
||||||
page_results.push(RegionText {
|
// glyph IDs, but that font may only appear in a logo or
|
||||||
text: String::new(),
|
// header — not in the table region. Instead we let the
|
||||||
needs_ocr: true,
|
// per-region text quality checks (is_garbage_text, is_cid_garbage,
|
||||||
});
|
// detect_encoding_issues) reject based on the actual extracted
|
||||||
continue;
|
// content. This avoids rejecting clean tables just because an
|
||||||
}
|
// unrelated decorative font on the same page is GID-encoded.
|
||||||
|
|
||||||
let matched: Vec<TextItem> = match items {
|
let matched: Vec<TextItem> = match items {
|
||||||
Some(items) => {
|
Some(items) => {
|
||||||
@@ -569,10 +680,22 @@ pub fn extract_tables_in_regions_mem(
|
|||||||
needs_ocr: true,
|
needs_ocr: true,
|
||||||
});
|
});
|
||||||
} else {
|
} else {
|
||||||
let needs_ocr =
|
// needs_ocr fires on any of:
|
||||||
is_garbage_text(&md) || is_cid_garbage(&md) || detect_encoding_issues(&md);
|
// - garbage text (non-alphanumeric heavy)
|
||||||
|
// - CID/Latin-1 mojibake
|
||||||
|
// - encoding issues (U+FFFD, dollar-as-space)
|
||||||
|
// - structural giveaways that the table is partial /
|
||||||
|
// mis-detected (numeric "header", empty header cells,
|
||||||
|
// duplicate header cells). Caught GLM-OCR-as-baseline
|
||||||
|
// scoring 0 TEDS on real prod tables in eval.
|
||||||
|
// Layout model already identified this region as a table,
|
||||||
|
// so use relaxed partial-table checks (layout_assisted=true).
|
||||||
|
let needs_ocr = is_garbage_text(&md)
|
||||||
|
|| is_cid_garbage(&md)
|
||||||
|
|| detect_encoding_issues(&md)
|
||||||
|
|| looks_like_partial_table_ex(&md, true);
|
||||||
page_results.push(RegionText {
|
page_results.push(RegionText {
|
||||||
text: md,
|
text: if needs_ocr { String::new() } else { md },
|
||||||
needs_ocr,
|
needs_ocr,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -1190,6 +1313,379 @@ fn is_cid_garbage(text: &str) -> bool {
|
|||||||
high_latin * 5 >= total * 2 && ascii_letters * 3 < total
|
high_latin * 5 >= total * 2 && ascii_letters * 3 < total
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Detect markdown tables with suspicious structure that suggest the heuristic
|
||||||
|
/// missed/mangled rows or columns. Returns true when the caller should treat
|
||||||
|
/// the result as `needs_ocr` and fall back to GPU OCR.
|
||||||
|
///
|
||||||
|
/// Catches three failure modes observed in production:
|
||||||
|
///
|
||||||
|
/// 1. **Header row looks like a data row** — first row starts with a numeric
|
||||||
|
/// value (e.g. `|2|...`), suggesting we missed the actual header above it.
|
||||||
|
/// Real headers almost never start with a bare number.
|
||||||
|
///
|
||||||
|
/// 2. **Header has empty cells in a multi-column table** — e.g.
|
||||||
|
/// `|Position||Administration|Administration|` (3+ cols, ≥1 empty cell).
|
||||||
|
/// Indicates poor column boundary detection.
|
||||||
|
///
|
||||||
|
/// 3. **Header has duplicate non-empty cells** in a multi-column table —
|
||||||
|
/// e.g. `Administration|Administration` appearing as adjacent cells means
|
||||||
|
/// we collapsed multi-line headers wrong.
|
||||||
|
///
|
||||||
|
/// Conservative by design: a few false positives (perfectly fine tables flagged)
|
||||||
|
/// just mean we run GPU OCR which is the existing safe path.
|
||||||
|
/// When `layout_assisted` is true (the layout model identified this region
|
||||||
|
/// as a table), we relax boundary-detection heuristics (numeric header,
|
||||||
|
/// empty header cells, sparse first data row) because the layout model
|
||||||
|
/// already gave us the table bbox — we're not guessing "is this a table?"
|
||||||
|
/// anymore, only "can we extract it correctly?". Paragraph and duplicate-
|
||||||
|
/// header checks stay, since those indicate genuine extraction quality
|
||||||
|
/// issues regardless of how the region was identified.
|
||||||
|
fn looks_like_partial_table_ex(markdown: &str, layout_assisted: bool) -> bool {
|
||||||
|
let lines: Vec<&str> = markdown.lines().filter(|l| l.starts_with('|')).collect();
|
||||||
|
if lines.len() < 2 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// Header is the first pipe-line; separator is the second
|
||||||
|
let header_line = lines[0];
|
||||||
|
let separator_line = lines.get(1).copied().unwrap_or("");
|
||||||
|
let is_separator = |l: &str| l.chars().all(|c| matches!(c, '|' | '-' | ' '));
|
||||||
|
if !is_separator(separator_line) {
|
||||||
|
// No separator after the first line — not a well-formed pipe-table.
|
||||||
|
// table_to_markdown always emits one when it returns content, so this
|
||||||
|
// shouldn't happen in practice. If it does, fall through to OCR.
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse header cells: split on '|', drop the leading/trailing empty pieces
|
||||||
|
let cells: Vec<&str> = header_line.split('|').map(|s| s.trim()).collect::<Vec<_>>();
|
||||||
|
// The first and last items are always empty (string starts and ends with '|')
|
||||||
|
if cells.len() < 3 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let header_cells: Vec<&str> = cells[1..cells.len() - 1].to_vec();
|
||||||
|
let n_cols = header_cells.len();
|
||||||
|
if n_cols < 2 {
|
||||||
|
// Single-column tables are usually lists/keys, not tables. Keep them
|
||||||
|
// (caller can decide), but multi-column header checks below don't
|
||||||
|
// apply.
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 1: header starts with a bare number (likely we missed
|
||||||
|
// the real header row above). Skip when layout-assisted — the layout
|
||||||
|
// model's bbox includes the real header; a numeric first cell (e.g.,
|
||||||
|
// a year "2024") is legitimate.
|
||||||
|
if !layout_assisted {
|
||||||
|
if let Some(first) = header_cells.first() {
|
||||||
|
let trimmed = first.trim();
|
||||||
|
if !trimmed.is_empty() && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 2: header has empty cells in a multi-column table.
|
||||||
|
// When layout-assisted, allow up to 1 empty header cell (common in
|
||||||
|
// tables with merged/spanning header cells that we can't represent).
|
||||||
|
let empty_count = header_cells.iter().filter(|c| c.is_empty()).count();
|
||||||
|
if layout_assisted {
|
||||||
|
// Reject only if >1 empty header cell (2+ means serious boundary issue)
|
||||||
|
if n_cols >= 3 && empty_count >= 2 {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
} else if n_cols >= 3 && empty_count >= 1 {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 3: header has duplicate non-empty cells
|
||||||
|
let mut seen: std::collections::HashSet<&str> = std::collections::HashSet::new();
|
||||||
|
for cell in &header_cells {
|
||||||
|
if cell.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if !seen.insert(cell) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 4: first data row has many empty cells in a multi-column
|
||||||
|
// table. Real tables rarely have a leading row with most cells blank;
|
||||||
|
// when this happens it usually means the heuristic split a multi-row
|
||||||
|
// header (e.g. "Position\nAdministration (1986-1992) | Administration
|
||||||
|
// (1992-1998)") into a single-row header + a sparse data row.
|
||||||
|
if let Some(first_data_line) = lines.get(2) {
|
||||||
|
let data_cells: Vec<&str> = first_data_line
|
||||||
|
.split('|')
|
||||||
|
.map(|s| s.trim())
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
if data_cells.len() >= 3 {
|
||||||
|
let data_inner = &data_cells[1..data_cells.len() - 1];
|
||||||
|
let empty_data = data_inner.iter().filter(|c| c.is_empty()).count();
|
||||||
|
// ≥3 cols, and significant portion of cells in the first data
|
||||||
|
// row are empty → likely we mis-split a multi-row header.
|
||||||
|
// When layout-assisted, relax from 33% to 50% — the bbox is
|
||||||
|
// more reliable, and real tables with one sparse first row
|
||||||
|
// (totals, subtotals) are common.
|
||||||
|
let threshold = if layout_assisted { 2 } else { 3 };
|
||||||
|
if n_cols >= 3 && empty_data * threshold >= n_cols {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Failure mode 5: cells flow as continuation paragraph (text wrapping
|
||||||
|
// mistaken for column structure). When a paragraph of prose gets mis-
|
||||||
|
// detected as a multi-column table, cells in the same column tend to
|
||||||
|
// start with lowercase letters or punctuation (continuation), not
|
||||||
|
// capital letters / digits (new entries). Real tables almost never
|
||||||
|
// have most data cells starting lowercase.
|
||||||
|
//
|
||||||
|
// Signal: ≥2 cols, ≥4 data rows, and ≥60% of non-empty data cells
|
||||||
|
// start with a lowercase letter or continuation punctuation.
|
||||||
|
let data_rows: Vec<Vec<&str>> = lines
|
||||||
|
.iter()
|
||||||
|
.skip(2) // header + separator
|
||||||
|
.map(|l| {
|
||||||
|
let parts: Vec<&str> = l.split('|').map(|s| s.trim()).collect();
|
||||||
|
if parts.len() >= 3 {
|
||||||
|
parts[1..parts.len() - 1].to_vec()
|
||||||
|
} else {
|
||||||
|
Vec::new()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.filter(|cells| !cells.is_empty())
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
if n_cols >= 2 && data_rows.len() >= 4 {
|
||||||
|
let mut continuation = 0;
|
||||||
|
let mut total = 0;
|
||||||
|
for row in &data_rows {
|
||||||
|
for cell in row {
|
||||||
|
let trimmed = cell.trim();
|
||||||
|
if trimmed.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
total += 1;
|
||||||
|
let first = trimmed.chars().next().unwrap();
|
||||||
|
// Continuation indicators: lowercase letter, common
|
||||||
|
// mid-sentence punctuation, closing quote
|
||||||
|
if first.is_lowercase()
|
||||||
|
|| matches!(first, ',' | '.' | ';' | ')' | '"' | '\'' | '”' | '’')
|
||||||
|
{
|
||||||
|
continuation += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if total > 0 && continuation * 5 >= total * 3 {
|
||||||
|
// ≥60% of cells look like sentence continuations → paragraph
|
||||||
|
// misread as table.
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Original strict validation (no layout assistance). Used by tests and
|
||||||
|
/// full-page extraction paths that don't have layout model assistance.
|
||||||
|
#[cfg(test)]
|
||||||
|
fn looks_like_partial_table(markdown: &str) -> bool {
|
||||||
|
looks_like_partial_table_ex(markdown, false)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod looks_like_partial_table_tests {
|
||||||
|
use super::{looks_like_partial_table, looks_like_partial_table_ex};
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn good_table_passes() {
|
||||||
|
let md = "|Name|Year|Country|\n|---|---|---|\n|Alice|2020|US|\n|Bob|2021|UK|";
|
||||||
|
assert!(
|
||||||
|
!looks_like_partial_table(md),
|
||||||
|
"should not flag well-formed table"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn header_starting_with_number_is_partial() {
|
||||||
|
// Heuristic missed the actual header row above
|
||||||
|
let md = "|2|Cambodian Women for Peace|9,835|\n|---|---|---|\n|3|Association|711|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn header_with_empty_cells_in_3col_is_partial() {
|
||||||
|
// Empty cell in 3+ column header → bad column detection
|
||||||
|
let md =
|
||||||
|
"|Position||Administration|Administration|\n|---|---|---|---|\n|Senate|24|8.3|16.7|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn header_with_duplicate_cells_is_partial() {
|
||||||
|
// Duplicate "Administration" → collapsed multi-line header wrong
|
||||||
|
let md =
|
||||||
|
"|Position|Administration|Administration|Notes|\n|---|---|---|---|\n|Senate|24|16|x|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn two_column_with_one_empty_cell_passes() {
|
||||||
|
// Many real two-column tables have key-only rows; don't penalise.
|
||||||
|
let md = "|Key||\n|---|---|\n|Alice|123|\n|Bob|456|";
|
||||||
|
// Header "Key|" has one empty cell but only 2 cols total — keep it.
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn single_column_table_is_kept() {
|
||||||
|
// Single-column "tables" are common (lists). Caller can decide; we
|
||||||
|
// don't second-guess based on column count alone.
|
||||||
|
let md = "|Item|\n|---|\n|First|\n|Second|";
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn no_table_at_all_returns_true() {
|
||||||
|
// table_to_markdown should never produce this, but defensive — if
|
||||||
|
// there's no separator, treat as not-a-table.
|
||||||
|
let md = "Just some text\nWith multiple lines";
|
||||||
|
// No lines start with '|' so we return false (no header to inspect).
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn first_data_row_with_many_empty_cells_is_partial() {
|
||||||
|
// Multi-row header collapsed to single-row → first "data row" has
|
||||||
|
// most cells empty (the actual sub-header values).
|
||||||
|
let md = "|Government|No. of Seats|Aquino|Ramos|\n|---|---|---|---|\n|Position|||(1986-1992)|\n|Senate|24|8.3|16.7|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn first_data_row_with_one_empty_cell_in_4col_passes() {
|
||||||
|
// Real data rows can have one empty cell (e.g. missing value);
|
||||||
|
// only flag when ≥1/3 of cells are empty.
|
||||||
|
let md = "|A|B|C|D|\n|---|---|---|---|\n|x|y||z|\n|p|q|r|s|";
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn paragraph_misread_as_two_column_table_is_partial() {
|
||||||
|
// Real production failure: text-wrapped paragraph mis-detected as
|
||||||
|
// 2-col table. Each cell continues the previous one as prose.
|
||||||
|
let md = "|Approval is needed from the|Acquisitions of|\n\
|
||||||
|
|---|---|\n\
|
||||||
|
|Treasurer if the acquisition|residential and|\n\
|
||||||
|
|constitutes a \"significant|agricultural|\n\
|
||||||
|
|action,\" including acquiring an|land by foreign|\n\
|
||||||
|
|interest in different types of|persons must be|\n\
|
||||||
|
|land where the monetary|reported to the|";
|
||||||
|
assert!(looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn real_multi_word_table_is_kept() {
|
||||||
|
// Real table with multi-word entries — cells start with capital
|
||||||
|
// letters / proper nouns, NOT lowercase continuations.
|
||||||
|
let md = "|Country|Capital|Notes|\n\
|
||||||
|
|---|---|---|\n\
|
||||||
|
|United States|Washington DC|Federal capital|\n\
|
||||||
|
|United Kingdom|London|City of London is a separate|\n\
|
||||||
|
|France|Paris|Île-de-France region|\n\
|
||||||
|
|Germany|Berlin|Reunified 1990|\n\
|
||||||
|
|Spain|Madrid|Largest city in Spain|";
|
||||||
|
assert!(!looks_like_partial_table(md));
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- layout_assisted relaxation tests ---
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn numeric_header_accepted_when_layout_assisted() {
|
||||||
|
// Year as first header cell is valid when layout model gave us the bbox.
|
||||||
|
let md = "|2024|Revenue|Growth|\n|---|---|---|\n|Q1|1.2M|5%|\n|Q2|1.4M|8%|";
|
||||||
|
assert!(
|
||||||
|
looks_like_partial_table(md),
|
||||||
|
"strict mode rejects numeric header"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!looks_like_partial_table_ex(md, true),
|
||||||
|
"layout-assisted should accept"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn one_empty_header_accepted_when_layout_assisted() {
|
||||||
|
// Common in merged-header tables: one spanning cell leaves a gap.
|
||||||
|
let md = "|Position||Senate|House|\n|---|---|---|---|\n|Chair|1|2|3|\n|Vice|4|5|6|";
|
||||||
|
assert!(
|
||||||
|
looks_like_partial_table(md),
|
||||||
|
"strict rejects 1 empty header"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!looks_like_partial_table_ex(md, true),
|
||||||
|
"layout-assisted allows 1 empty"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn two_empty_headers_still_rejected_when_layout_assisted() {
|
||||||
|
// 2+ empty headers is still bad even with layout assistance.
|
||||||
|
let md = "|A|||D|\n|---|---|---|---|\n|x|y|z|w|";
|
||||||
|
assert!(
|
||||||
|
looks_like_partial_table_ex(md, true),
|
||||||
|
"2 empty headers rejected even layout-assisted"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn sparse_first_row_relaxed_when_layout_assisted() {
|
||||||
|
// 1/4 empty = 25%, below strict 33% threshold but accepted by layout-assisted 50%.
|
||||||
|
let md = "|A|B|C|D|\n|---|---|---|---|\n|x||y|z|\n|p|q|r|s|";
|
||||||
|
assert!(!looks_like_partial_table(md), "strict: 25% empty is OK");
|
||||||
|
// 2/4 = 50%, strict would flag (2*3>=4), relaxed threshold (2*2>=4) would also flag.
|
||||||
|
let md2 = "|A|B|C|D|\n|---|---|---|---|\n|||y|z|\n|p|q|r|s|";
|
||||||
|
assert!(looks_like_partial_table(md2), "strict: 50% empty flagged");
|
||||||
|
assert!(
|
||||||
|
looks_like_partial_table_ex(md2, true),
|
||||||
|
"layout-assisted: 50% also flagged"
|
||||||
|
);
|
||||||
|
// 2/6 = 33%, strict flags (2*3>=6), relaxed does not (2*2<6)
|
||||||
|
let md3 = "|A|B|C|D|E|F|\n|---|---|---|---|---|---|\n|x|||y|z|w|\n|a|b|c|d|e|f|";
|
||||||
|
assert!(looks_like_partial_table(md3), "strict: 33% flagged");
|
||||||
|
assert!(
|
||||||
|
!looks_like_partial_table_ex(md3, true),
|
||||||
|
"layout-assisted: 33% accepted"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn paragraph_still_rejected_when_layout_assisted() {
|
||||||
|
// Paragraph detection is not relaxed — it's a genuine extraction issue.
|
||||||
|
let md = "|Approval is needed from the|Acquisitions of|\n\
|
||||||
|
|---|---|\n\
|
||||||
|
|Treasurer if the acquisition|residential and|\n\
|
||||||
|
|constitutes a \"significant|agricultural|\n\
|
||||||
|
|action,\" including acquiring an|land by foreign|\n\
|
||||||
|
|interest in different types of|persons must be|\n\
|
||||||
|
|land where the monetary|reported to the|";
|
||||||
|
assert!(
|
||||||
|
looks_like_partial_table_ex(md, true),
|
||||||
|
"paragraph rejection stays strict"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn duplicate_headers_still_rejected_when_layout_assisted() {
|
||||||
|
let md =
|
||||||
|
"|Position|Administration|Administration|Notes|\n|---|---|---|---|\n|Senate|24|16|x|";
|
||||||
|
assert!(
|
||||||
|
looks_like_partial_table_ex(md, true),
|
||||||
|
"duplicate headers rejected even layout-assisted"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Analyse extracted items and rects for layout complexity.
|
/// Analyse extracted items and rects for layout complexity.
|
||||||
fn compute_layout_complexity(
|
fn compute_layout_complexity(
|
||||||
items: &[types::TextItem],
|
items: &[types::TextItem],
|
||||||
|
|||||||
+114
-3
@@ -4,9 +4,10 @@ use pdf_inspector::detector::{DetectionConfig, ScanStrategy};
|
|||||||
use pdf_inspector::extractor::group_into_lines;
|
use pdf_inspector::extractor::group_into_lines;
|
||||||
use pdf_inspector::types::TextLine;
|
use pdf_inspector::types::TextLine;
|
||||||
use pdf_inspector::{
|
use pdf_inspector::{
|
||||||
detect_pdf_type, extract_tables_in_regions_mem, extract_text, extract_text_in_regions_mem,
|
detect_pdf_type, extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||||
extract_text_with_positions, process_pdf_mem, process_pdf_with_options, to_markdown,
|
extract_text_in_regions_mem, extract_text_with_positions, process_pdf_mem,
|
||||||
MarkdownOptions, PdfError, PdfOptions, PdfType, TextItem,
|
process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions, PdfType,
|
||||||
|
TextItem,
|
||||||
};
|
};
|
||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
|
|
||||||
@@ -1436,3 +1437,113 @@ fn test_extract_tables_in_regions_nonexistent_page() {
|
|||||||
assert!(region.needs_ocr);
|
assert!(region.needs_ocr);
|
||||||
assert!(region.text.is_empty());
|
assert!(region.text.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// extract_pages_markdown_mem tests
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_basic() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results = extract_pages_markdown_mem(&buf, &[0, 1]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 2);
|
||||||
|
assert_eq!(results[0].page, 0);
|
||||||
|
assert_eq!(results[1].page, 1);
|
||||||
|
// Text-based PDF should produce non-empty markdown
|
||||||
|
assert!(!results[0].markdown.is_empty());
|
||||||
|
assert!(!results[0].needs_ocr);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_page_ordering() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
// Request pages in non-sequential order
|
||||||
|
let results = extract_pages_markdown_mem(&buf, &[1, 0]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 2);
|
||||||
|
// Results should match input order, not document order
|
||||||
|
assert_eq!(results[0].page, 1);
|
||||||
|
assert_eq!(results[1].page, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_out_of_range() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results = extract_pages_markdown_mem(&buf, &[9999]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
assert_eq!(results[0].page, 9999);
|
||||||
|
assert!(results[0].markdown.is_empty());
|
||||||
|
assert!(results[0].needs_ocr);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_empty_pages_list() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results = extract_pages_markdown_mem(&buf, &[]).unwrap();
|
||||||
|
assert!(results.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_single_page() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
let results = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
assert_eq!(results[0].page, 0);
|
||||||
|
assert!(!results[0].markdown.is_empty());
|
||||||
|
assert!(!results[0].needs_ocr);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_invalid_buffer() {
|
||||||
|
let result = extract_pages_markdown_mem(b"not a pdf", &[0]);
|
||||||
|
assert!(result.is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_gid_pages_need_ocr() {
|
||||||
|
// shinagawa_identity_h.pdf has GID-encoded fonts
|
||||||
|
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||||
|
let results = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(results.len(), 1);
|
||||||
|
assert!(results[0].needs_ocr);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_pages_markdown_consistency_with_process_pdf() {
|
||||||
|
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||||
|
|
||||||
|
// Get full process_pdf output
|
||||||
|
let full = process_pdf_mem(&buf).unwrap();
|
||||||
|
let full_md = full.markdown.unwrap_or_default();
|
||||||
|
|
||||||
|
// Get per-page output for all pages
|
||||||
|
let page_count = full.page_count;
|
||||||
|
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||||
|
let per_page = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||||
|
|
||||||
|
// Concatenated per-page markdown should contain substantial overlap with
|
||||||
|
// the full output (exact match not expected due to header/footer stripping
|
||||||
|
// and cross-page paragraph merging differences)
|
||||||
|
let concat: String = per_page
|
||||||
|
.iter()
|
||||||
|
.map(|p| p.markdown.as_str())
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join("\n");
|
||||||
|
|
||||||
|
// Both should be non-empty for a text-based PDF
|
||||||
|
assert!(!full_md.is_empty());
|
||||||
|
assert!(!concat.is_empty());
|
||||||
|
|
||||||
|
// The per-page version should contain at least 50% of the full content's
|
||||||
|
// length (accounting for header/footer stripping differences)
|
||||||
|
assert!(
|
||||||
|
concat.len() * 2 >= full_md.len(),
|
||||||
|
"per-page concat ({} chars) is too short vs full ({} chars)",
|
||||||
|
concat.len(),
|
||||||
|
full_md.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user