Compare commits

..
Author SHA1 Message Date
Cursor AgentandAbimael Martell 92e2d6ec8e fix(napi): copy async task input on the JS thread for soundness
Review feedback on #337: holding the napi Buffer and reading it from
the libuv worker was unsound. Buffer derefs straight to the JS-side
allocation, so a caller mutating it before the promise settled would
race the worker's reads — undefined behavior, not a recoverable error,
and the documented don't-mutate contract was unenforceable. Deferring
the copy to compute() would not help: any off-thread read races the
same way. The JS thread is the only race-free place to take the copy,
because JS is single-threaded and nothing can mutate the buffer during
the synchronous part of the call.

Revert to an owned Vec<u8> copied at call time. The cost is one memcpy,
negligible next to the parse the async variants exist to unblock. Docs
now state the buffer may be reused or mutated immediately, and a test
locks in the copy semantics by mutating the input while a parse is in
flight.

Co-authored-by: Abimael Martell <abimaelmartell@users.noreply.github.com>
2026-08-10 19:06:24 +00:00
Cursor AgentandAbimael Martell 74a633dfd0 fix(napi): read async task buffers in place instead of copying
Review feedback on #337: buffer.to_vec() copied the whole PDF on the
event loop before the task was queued, so large inputs still stalled
the loop and doubled peak memory. The tasks now hold the napi Buffer
itself — its ref pins the JS allocation for the task's lifetime and
the backing store is stable, so compute() reads it directly from the
worker thread. Callers must not mutate the buffer until the promise
settles (same contract as Node's async fs APIs); documented on each
export and in the README.

The suggested removal of ts_return_type was checked and rejected:
without it napi-rs generates Promise<unknown> for AsyncTask returns.
A comment now records that finding.

Co-authored-by: Abimael Martell <abimaelmartell@users.noreply.github.com>
2026-08-10 17:40:05 +00:00
Cursor AgentandAbimael Martell c1957bf750 feat(napi): add processPdfAsync, classifyPdfAsync, extractPagesMarkdownAsync
The Node bindings are synchronous, so every call parses on the event
loop thread — up to hundreds of milliseconds of dead loop per document
in a server. Add additive AsyncTask-based variants that run the same
shared implementations on the libuv thread pool and return promises.

The existing synchronous exports keep their names, signatures, and
behaviour; each sync/async pair shares one implementation. Panics in
compute() are caught and surfaced as rejections, matching the sync
error contract.

Closes #336

Co-authored-by: Abimael Martell <abimaelmartell@users.noreply.github.com>
2026-08-10 08:31:23 +00:00
17 changed files with 76 additions and 1605 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector"
version = "0.1.8"
version = "0.1.7"
edition = "2021"
autobins = false
authors = ["Firecrawl Team"]
+2 -2
View File
@@ -851,7 +851,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
[[package]]
name = "pdf-inspector"
version = "0.1.8"
version = "0.1.7"
dependencies = [
"env_logger",
"include_dir",
@@ -867,7 +867,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-napi"
version = "0.2.3"
version = "0.2.2"
dependencies = [
"napi",
"napi-build",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-napi"
version = "0.2.3"
version = "0.2.2"
edition = "2021"
[lib]
+6 -6
View File
@@ -8,12 +8,12 @@
"@napi-rs/cli": "^3.4.1",
},
"optionalDependencies": {
"@firecrawl/pdf-inspector-darwin-arm64": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.13.0",
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.13.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.13.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
},
},
},
+7 -7
View File
@@ -1,6 +1,6 @@
{
"name": "@firecrawl/pdf-inspector",
"version": "1.13.0",
"version": "1.12.0",
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
"main": "index.js",
"types": "index.d.ts",
@@ -52,11 +52,11 @@
"@napi-rs/cli": "^3.4.1"
},
"optionalDependencies": {
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.13.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.13.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.13.0"
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
}
}
-54
View File
@@ -89,11 +89,6 @@ pub struct TextItem {
pub item_type: ItemType,
/// URL for link items, `None` for other types.
pub link_url: Option<String>,
/// Marked Content ID from the content stream's BDC/BMC operator, `None`
/// when the text is not part of marked content. Join with the
/// `page`/`mcid` pairs from [`extractStructureElements`] to attach
/// structure-tree roles (headings, paragraphs, …) in tagged PDFs.
pub mcid: Option<i64>,
}
/// A page's regions for text extraction: (page_index_0based, bboxes).
@@ -311,61 +306,12 @@ pub fn extract_text_with_positions(
is_strikeout: item.is_strikeout,
item_type,
link_url,
mcid: item.mcid,
}
})
.collect())
})
}
/// One structure-tree element reference from a tagged PDF.
#[napi(object)]
pub struct StructureElementJs {
/// 1-indexed page number (matches `TextItem.page`).
pub page: u32,
/// Marked Content ID from the page's content stream (matches
/// `TextItem.mcid`).
pub mcid: i64,
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
/// Custom tags are resolved through the document's role map; tags with
/// no standard mapping are returned verbatim.
pub role: String,
}
/// Extract structure-tree element references from a tagged PDF.
///
/// Parses the document's structure tree (when present) and returns one
/// entry per marked-content reference, resolved to its 1-indexed page,
/// MCID, and structure type name. Returns an empty array when the PDF is
/// not tagged.
///
/// Join `(page, mcid)` against the `page`/`mcid` fields from
/// [`extractTextWithPositions`] to attach heading levels (H1..H6) and other
/// semantic roles to extracted text.
///
/// Pass 1-indexed page numbers (matching `TextItem.page`) to restrict
/// output; omit `pages` for the whole document. Entries are sorted by
/// `(page, mcid)`.
#[napi]
pub fn extract_structure_elements(
buffer: Buffer,
pages: Option<Vec<u32>>,
) -> Result<Vec<StructureElementJs>> {
let bytes: Vec<u8> = buffer.to_vec();
catch_panic("extract_structure_elements", move || {
let elements = pdf_inspector::extract_structure_elements_mem(&bytes, pages.as_deref())
.map_err(|e| to_napi_err(e, "extract_structure_elements"))?;
Ok(elements
.into_iter()
.map(|e| StructureElementJs {
page: e.page,
mcid: e.mcid,
role: e.role,
})
.collect())
})
}
/// Extract text within bounding-box regions from a PDF.
///
/// For hybrid OCR: layout model detects regions in rendered images,
-42
View File
@@ -8,7 +8,6 @@ import {
classifyPdfAsync,
extractText,
extractTextWithPositions,
extractStructureElements,
extractTextInRegions,
detectVectorGridInRegion,
extractPagesMarkdown,
@@ -16,7 +15,6 @@ import {
} from './index.js';
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
const taggedFixture = readFileSync('../tests/fixtures/firecrawl_docs_tagged.pdf');
// --- processPdf ---
console.log('Testing processPdf...');
@@ -84,46 +82,6 @@ assert.ok(page1Items.length > 0);
assert.ok(page1Items.every(i => i.page === 1));
console.log(' extractTextWithPositions with pages: OK');
// mcid: undefined on untagged PDFs, numeric on tagged marked content
assert.ok(items.every(i => i.mcid === undefined || typeof i.mcid === 'number'));
const taggedItems = extractTextWithPositions(taggedFixture);
assert.ok(
taggedItems.some(i => typeof i.mcid === 'number'),
'tagged PDF text items should carry Marked Content IDs',
);
console.log(' extractTextWithPositions mcid: OK');
// --- extractStructureElements ---
console.log('Testing extractStructureElements...');
const structureElements = extractStructureElements(taggedFixture);
assert.ok(structureElements.length > 0);
assert.ok(structureElements.every(e => typeof e.page === 'number'));
assert.ok(structureElements.every(e => typeof e.mcid === 'number'));
assert.ok(structureElements.every(e => typeof e.role === 'string' && e.role.length > 0));
assert.ok(
structureElements.some(e => e.role === 'H1'),
'tagged fixture should surface H1 heading roles',
);
// (page, mcid) joins against extractTextWithPositions to recover heading text
const h1Refs = new Set(
structureElements.filter(e => e.role === 'H1').map(e => `${e.page}:${e.mcid}`),
);
const h1Text = taggedItems
.filter(i => typeof i.mcid === 'number' && h1Refs.has(`${i.page}:${i.mcid}`))
.map(i => i.text)
.join('');
assert.ok(h1Text.trim().length > 0, 'H1 join should recover heading text');
// pages filter is 1-indexed, matching TextItem.page
const page1Elements = extractStructureElements(taggedFixture, [1]);
assert.ok(page1Elements.length > 0);
assert.ok(page1Elements.every(e => e.page === 1));
// untagged PDFs yield an empty array
assert.deepEqual(extractStructureElements(fixture), []);
console.log(' extractStructureElements: OK');
// --- extractTextInRegions ---
console.log('Testing extractTextInRegions...');
const regionResults = extractTextInRegions(fixture, [
-35
View File
@@ -51,20 +51,6 @@ class TextItem:
is_underline: bool
is_strikeout: bool
item_type: str
mcid: Optional[int]
"""Marked Content ID from the content stream's BDC/BMC operator, None when
the text is not part of marked content. Join with the (page, mcid) pairs
from extract_structure_elements to attach structure-tree roles in tagged
PDFs."""
class StructureElement:
"""One structure-tree element reference from a tagged PDF."""
page: int
"""1-indexed page number (matches TextItem.page)."""
mcid: int
"""Marked Content ID from the page's content stream (matches TextItem.mcid)."""
role: str
"""Standard structure type name ("H1".."H6", "P", "Table", "TD", ...)."""
class RegionText:
"""Extracted text for a single region."""
@@ -146,27 +132,6 @@ def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] =
"""Extract text with position information from bytes."""
...
def extract_structure_elements(path: str, pages: Optional[list[int]] = None) -> list[StructureElement]:
"""Extract structure-tree element references from a tagged PDF file.
Returns one entry per marked-content reference, resolved to its 1-indexed
page, MCID, and structure type name ("H1".."H6", "P", "Table", ...), sorted
by (page, mcid). Returns an empty list when the PDF is not tagged.
Args:
path: Path to the PDF file.
pages: Optional list of 1-indexed pages (matching ``TextItem.page``).
When ``None`` (default), the whole document is returned.
"""
...
def extract_structure_elements_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[StructureElement]:
"""Extract structure-tree element references from tagged PDF bytes.
See :func:`extract_structure_elements` for details.
"""
...
def extract_text_in_regions(
path: str,
page_regions: list[tuple[int, list[list[float]]]],
+1 -1
View File
@@ -6,7 +6,7 @@ build-backend = "maturin"
name = "pdf-inspector"
# Bump this to publish to PyPI — CI publishes automatically when the version
# changes on main (same flow as napi/package.json for npm).
version = "0.2.7"
version = "0.2.6"
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
readme = "docs/python.md"
license = { text = "MIT" }
+17 -327
View File
@@ -41,110 +41,15 @@ pub(crate) fn detect_columns(
}
debug!("page {}: detect_columns: {} items", page, page_items.len());
// The width of one ordinary page, used three ways below: as the largest
// credible width for a single text run, as the size of empty gap that marks
// content as detached, and as the span past which those checks run at all.
// This is a heuristic, not a format rule: PDF 2.0 sets no page-size limit,
// and since PDF 1.6 `UserUnit` scales a page's physical size independently
// of its coordinates. 14_400 units (200in at the default 1/72in unit) is
// the traditional Acrobat architectural limit, which makes it a reasonable
// "wider than any ordinary page" mark in coordinate space.
const MAX_PAGE_EXTENT: f32 = 14_400.0;
// A detached cluster is only dropped if it also holds a small minority of
// the items, so a genuine two-part layout keeps its full bounds even when
// the halves are far apart.
const MAX_TRIM_FRACTION: f32 = 0.10;
// Position and width of each item, skipping only non-finite geometry.
let finite_span = |i: &&TextItem| -> Option<(f32, f32)> {
let (left, width) = (i.x, effective_width(i));
(left.is_finite() && (left + width).is_finite()).then_some((left, width))
};
let (min_left, max_right, total) = page_items.iter().filter_map(finite_span).fold(
(f32::INFINITY, f32::NEG_INFINITY, 0usize),
|(lo, hi, n), (left, width)| (lo.min(left), hi.max(left + width), n + 1),
);
// No item had usable geometry, so there is no layout to report.
if total == 0 {
return vec![];
}
// Every threshold below (gutter margins, spanning-item width, the XY-cut
// margin) is a fraction of the page width, so a far item can set the scale
// for the whole page and shrink the effective detection window to a
// rounding error — real gutters then fall inside the margin band and a
// genuine multi-column page collapses to one region.
//
// Anything inside one page extent is ordinary, so the common case keeps the
// plain bounds and skips the work below entirely.
let (x_min, x_max) = if max_right - min_left <= MAX_PAGE_EXTENT {
(min_left, max_right)
} else {
// Discarding content needs positive evidence that it is not part of the
// layout, because a count-based rule alone cannot tell a stray from a
// sparse far sidebar. The evidence is geometric: positions are grouped
// into clusters separated by more than a whole page of continuous
// emptiness. Real content, however sparse, does not leave a void that
// large; a malformed coordinate sits alone beyond one.
let mut spans: Vec<(f32, f32)> = page_items.iter().filter_map(finite_span).collect();
spans.sort_by(|a, b| a.0.total_cmp(&b.0));
let mut core: Option<std::ops::Range<usize>> = None;
let mut start = 0usize;
for i in 1..=spans.len() {
if i < spans.len() && spans[i].0 - spans[i - 1].0 <= MAX_PAGE_EXTENT {
continue;
}
if core.as_ref().is_none_or(|best| i - start > best.len()) {
core = Some(start..i);
}
start = i;
}
let mut core = core.unwrap_or(0..spans.len());
// Only drop the detached clusters when they are a small minority, so a
// genuine two-part layout keeps its full bounds.
let dropped = spans.len() - core.len();
if dropped as f32 > spans.len() as f32 * MAX_TRIM_FRACTION {
core = 0..spans.len();
}
let core = &spans[core];
// Positions cannot be inflated by a bogus width, so the spread of the
// content is a sound scale for judging one. A run much wider than the
// page's own content is a malformed width — the test is relative, so a
// genuinely large page keeps its genuinely long runs.
let (lo, widest_left) = (core[0].0, core[core.len() - 1].0);
let max_run_width = (widest_left - lo) + MAX_PAGE_EXTENT;
let hi = core
.iter()
.filter(|&&(_, width)| width <= max_run_width)
.map(|&(left, width)| left + width)
.fold(widest_left, f32::max);
if lo != min_left || hi != max_right {
debug!(
"page {page}: bounds {min_left}..{max_right} exceed one page; \
dropped {dropped}/{} detached item(s), using {lo}..{hi}",
spans.len()
);
}
(lo, hi)
};
// Hard ceiling on the histogram size, independent of the trimming above:
// the bounds are attacker-influenced, so an unclamped
// `page_width / BIN_WIDTH` lets a crafted PDF force an arbitrarily large
// `vec![0u32; num_bins]` allocation. 65_536 bins covers ~128k points at
// BIN_WIDTH 2.0 — roughly 9x the largest legal page — so this never binds
// on a real layout. Kept as a bound that does not depend on the outlier
// heuristic staying correct.
const MAX_BINS: usize = 65_536;
// Find page bounds
let x_min = page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
let x_max = page_items
.iter()
.map(|i| i.x + effective_width(i))
.fold(f32::NEG_INFINITY, f32::max);
let page_width = x_max - x_min;
if !page_width.is_finite() || page_width < 200.0 {
if page_width < 200.0 {
return vec![ColumnRegion { x_min, x_max }];
}
@@ -152,20 +57,13 @@ pub(crate) fn detect_columns(
return vec![ColumnRegion { x_min, x_max }];
}
// Widen the bins rather than dropping the tail of the page. Clamping the
// count alone would leave anything past MAX_BINS * BIN_WIDTH outside the
// histogram, folded into the last bin, which places gutters at the wrong
// coordinates. Scaling keeps full coverage under the same allocation
// ceiling; only the resolution degrades, and only beyond ~131k points.
let bin_width = BIN_WIDTH.max(page_width / MAX_BINS as f32);
// Build occupancy histogram.
// Exclude items wider than 60% of page width — these are spanning items
// (titles, full-width paragraphs) that would fill the gutter and prevent
// detection of partial-page column layouts (e.g. two-column abstracts on
// a page that also has single-column introduction text).
let wide_threshold = page_width * 0.6;
let num_bins = ((page_width / bin_width).ceil() as usize).clamp(1, MAX_BINS);
let num_bins = ((page_width / BIN_WIDTH).ceil() as usize).max(1);
let mut histogram = vec![0u32; num_bins];
for item in &page_items {
@@ -173,8 +71,8 @@ pub(crate) fn detect_columns(
if w > wide_threshold {
continue;
}
let left = ((item.x - x_min) / bin_width).floor() as usize;
let right = (((item.x + w) - x_min) / bin_width).ceil() as usize;
let left = ((item.x - x_min) / BIN_WIDTH).floor() as usize;
let right = (((item.x + w) - x_min) / BIN_WIDTH).ceil() as usize;
let left = left.min(num_bins);
let right = right.min(num_bins);
for count in histogram.iter_mut().take(right).skip(left) {
@@ -211,12 +109,12 @@ pub(crate) fn detect_columns(
let valleys: Vec<(usize, usize)> = valleys
.into_iter()
.filter(|&(start, end)| {
let width_pts = (end - start) as f32 * bin_width;
let width_pts = (end - start) as f32 * BIN_WIDTH;
if width_pts < MIN_GUTTER_WIDTH {
return false;
}
// Valley center must not be within 5% of page edges
let center_pts = ((start + end) as f32 / 2.0) * bin_width;
let center_pts = ((start + end) as f32 / 2.0) * BIN_WIDTH;
center_pts > margin_threshold && center_pts < (page_width - margin_threshold)
})
.collect();
@@ -234,7 +132,7 @@ pub(crate) fn detect_columns(
&histogram,
num_bins,
x_min,
bin_width,
BIN_WIDTH,
page_width,
margin_threshold,
);
@@ -243,7 +141,7 @@ pub(crate) fn detect_columns(
&rel_valleys,
&page_items,
x_min,
bin_width,
BIN_WIDTH,
x_max,
MIN_ITEMS_PER_COLUMN,
MIN_VERTICAL_SPAN_RATIO,
@@ -284,7 +182,7 @@ pub(crate) fn detect_columns(
&valleys,
&page_items,
x_min,
bin_width,
BIN_WIDTH,
x_max,
MIN_ITEMS_PER_COLUMN,
MIN_VERTICAL_SPAN_RATIO,
@@ -298,7 +196,7 @@ pub(crate) fn detect_columns(
&valleys,
&page_items,
x_min,
bin_width,
BIN_WIDTH,
x_max,
MIN_ITEMS_PER_COLUMN,
MIN_VERTICAL_SPAN_RATIO,
@@ -1929,7 +1827,7 @@ fn split_column_stragglers(lines: Vec<TextLine>) -> (Vec<TextLine>, Vec<TextLine
.unwrap();
let (cs, ce) = segments[core_seg];
let mut core = Vec::with_capacity(ce.saturating_sub(cs));
let mut core = Vec::with_capacity(ce - cs);
let mut stragglers = Vec::new();
for (i, line) in lines.into_iter().enumerate() {
if i >= cs && i < ce {
@@ -2635,214 +2533,6 @@ mod tests {
);
}
#[test]
fn extreme_far_coordinate_does_not_allocate_unboundedly() {
// A crafted PDF can place a text run at an arbitrary coordinate via the
// text matrix. The derived page width must not drive an unbounded
// histogram allocation (previously `page_width / BIN_WIDTH` bins with no
// upper bound would try to reserve terabytes and abort the process).
let mut items = Vec::new();
for i in 0..24 {
items.push(make_item(1, i as f32 * 10.0, 700.0 - i as f32 * 5.0, "A"));
}
// Item placed 1e12 points away — 5e11 bins if left unclamped.
items.push(make_item(1, 1e12, 700.0, "Z"));
// Must return without aborting; content is preserved as a single region.
let cols = detect_columns(&items, 1, false);
assert!(!cols.is_empty());
}
#[test]
fn non_finite_coordinates_never_leak_into_region_bounds() {
// An inf/NaN coordinate must not escape as a column boundary: callers
// treat these as page/column edges.
for bad_x in [f32::INFINITY, f32::NEG_INFINITY, f32::NAN] {
let mut items = Vec::new();
for i in 0..24 {
items.push(make_item(1, i as f32 * 10.0, 700.0 - i as f32 * 5.0, "A"));
}
items.push(make_item(1, bad_x, 700.0, "Z"));
for col in detect_columns(&items, 1, false) {
assert!(
col.x_min.is_finite() && col.x_max.is_finite(),
"bad_x {bad_x} leaked bounds {}..{}",
col.x_min,
col.x_max
);
}
}
}
#[test]
fn all_non_finite_coordinates_yield_no_columns() {
let items: Vec<TextItem> = (0..24)
.map(|i| make_item(1, f32::NAN, 700.0 - i as f32 * 5.0, "A"))
.collect();
assert!(detect_columns(&items, 1, false).is_empty());
}
#[test]
fn one_bad_item_does_not_disable_column_detection() {
// A single stray item should not collapse a clean two-column page to
// one region. Every gutter threshold is a fraction of the page width,
// so an untrimmed outlier pushes real gutters inside the rejected
// margin band. A malformed *width* at an ordinary position poisons the
// bounds just as a malformed position does.
for (label, bad_x, bad_width) in [
("nan position", f32::NAN, 0.0),
("inf position", f32::INFINITY, 0.0),
("far position", 50_000.0, 0.0),
("very far position", 1e12, 0.0),
("huge width", 100.0, 1e12),
("inf width", 100.0, f32::INFINITY),
] {
let mut items = Vec::new();
items.extend(fill_zone(1, 30.0, 280.0, 750.0, 50.0));
items.extend(fill_zone(1, 320.0, 570.0, 750.0, 50.0));
let mut bad = make_item(1, bad_x, 400.0, "Z");
bad.width = bad_width;
items.push(bad);
let cols = detect_columns(&items, 1, false);
assert_eq!(
cols.len(),
2,
"{label}: expected 2 columns, got {}",
cols.len()
);
for col in &cols {
assert!(
col.x_max - col.x_min <= MAX_PAGE_EXTENT_FOR_TEST,
"{label}: region {}..{} exceeds one page",
col.x_min,
col.x_max
);
}
}
}
/// Mirrors `MAX_PAGE_EXTENT` in `detect_columns`.
const MAX_PAGE_EXTENT_FOR_TEST: f32 = 14_400.0;
#[test]
fn very_wide_page_keeps_full_histogram_coverage() {
// Beyond MAX_BINS * BIN_WIDTH (~131k points) the bins must widen rather
// than stop covering the page. Three zones: the first gutter is inside
// the old coverage limit, the second is past it. Because the first
// gutter is found, the XY-cut fallback never runs, so a truncated
// histogram silently reports two columns instead of three.
let mut items = Vec::new();
items.extend(fill_zone(1, 0.0, 60_000.0, 750.0, 700.0));
items.extend(fill_zone(1, 70_000.0, 140_000.0, 750.0, 700.0));
items.extend(fill_zone(1, 160_000.0, 200_000.0, 750.0, 700.0));
let cols = detect_columns(&items, 1, false);
assert_eq!(
cols.len(),
3,
"Expected 3 columns across a 200k-wide page, got {}",
cols.len()
);
assert!(
(140_000.0..=160_000.0).contains(&cols[1].x_max),
"second gutter at {}, expected inside the real 140k..160k gap",
cols[1].x_max
);
}
#[test]
fn large_page_with_legitimately_long_runs_is_kept() {
// On a very large page, individual runs can exceed one ordinary page's
// width. They are real content, so they must not be judged malformed:
// the page keeps its columns and its full right edge.
let mut items = Vec::new();
for row in 0..30 {
let y = 750.0 - row as f32 * 14.0;
let mut left = make_item(1, 0.0, y, "Left run");
left.width = 20_000.0;
let mut right = make_item(1, 25_000.0, y, "Right run");
right.width = 20_000.0;
items.extend([left, right]);
}
let cols = detect_columns(&items, 1, false);
assert!(
!cols.is_empty(),
"a page of long-but-valid runs must still report a layout"
);
let right_edge = cols
.iter()
.map(|c| c.x_max)
.fold(f32::NEG_INFINITY, f32::max);
assert!(
right_edge > 44_000.0,
"long runs were treated as malformed: right edge {right_edge}, expected ~45_000"
);
}
#[test]
fn sparse_far_sidebar_on_a_large_page_is_kept() {
// A large-format page with a thin, sparsely-populated sidebar far from
// the main block. The sidebar is a small minority of the items, so an
// item-count rule alone would discard it — but nothing about its
// geometry says it is invalid, so its bounds must survive.
let mut items = Vec::new();
items.extend(fill_zone(1, 0.0, 12_000.0, 750.0, 500.0));
for i in 0..12 {
items.push(make_item(1, 24_000.0, 750.0 - i as f32 * 14.0, "Sidebar"));
}
let cols = detect_columns(&items, 1, false);
let right_edge = cols
.iter()
.map(|c| c.x_max)
.fold(f32::NEG_INFINITY, f32::max);
assert!(
right_edge > 24_000.0,
"sidebar was trimmed away: right edge {right_edge}, expected >24_000"
);
}
#[test]
fn genuinely_wide_layout_keeps_its_true_bounds() {
// A large-format page whose content really is spread beyond one
// ordinary page must not be trimmed to the median cluster: its far
// items are the majority, not strays.
let mut items = Vec::new();
items.extend(fill_zone(1, 100.0, 20_000.0, 750.0, 600.0));
items.extend(fill_zone(1, 22_000.0, 40_000.0, 750.0, 600.0));
let cols = detect_columns(&items, 1, false);
let widest = cols
.iter()
.map(|c| c.x_max)
.fold(f32::NEG_INFINITY, f32::max);
assert!(
widest > 35_000.0,
"wide layout was trimmed: right edge {widest}, expected ~40_000"
);
}
#[test]
fn oversized_but_legal_page_is_not_trimmed() {
// A wide-format page well inside the 14_400pt spec limit must keep its
// real bounds — outlier trimming is only for spans beyond a legal page.
let mut items = Vec::new();
items.extend(fill_zone(1, 100.0, 4_000.0, 750.0, 400.0));
items.extend(fill_zone(1, 4_400.0, 8_000.0, 750.0, 400.0));
let cols = detect_columns(&items, 1, false);
assert_eq!(cols.len(), 2, "Expected 2 columns, got {}", cols.len());
assert!(
cols[1].x_max > 7_000.0,
"right column should keep its true extent, got {}",
cols[1].x_max
);
}
#[test]
fn two_column_regression_guard() {
// Standard 2-column layout with clear gutter at center
-74
View File
@@ -657,80 +657,6 @@ pub fn extract_pages_markdown<P: AsRef<Path>>(
extract_pages_markdown_mem(&buffer, pages)
}
// =========================================================================
// Structure-tree element extraction (tagged PDFs)
// =========================================================================
/// One structure-tree element reference from a tagged PDF, resolved to a
/// page and Marked Content ID.
///
/// Join `(page, mcid)` against [`TextItem::page`] / [`TextItem::mcid`] from
/// [`extract_text_with_positions`] to attach semantic roles (heading levels,
/// paragraphs, table cells, …) to extracted text.
#[derive(Debug, Clone)]
pub struct StructureElement {
/// 1-indexed page number (matches [`TextItem::page`]).
pub page: u32,
/// Marked Content ID from the page's content stream (matches
/// [`TextItem::mcid`]).
pub mcid: i64,
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
/// Custom tags are resolved through the document's `/RoleMap`; tags
/// with no standard mapping are returned verbatim.
pub role: String,
}
/// Extract structure-tree element references from a tagged PDF in memory.
///
/// Parses `/StructTreeRoot` (when present) and returns one entry per
/// marked-content reference, resolved to its 1-indexed page, MCID, and
/// structure type name. Returns an empty list when the PDF is not tagged.
///
/// Pass `Some(&[...])` with 1-indexed page numbers (matching
/// [`TextItem::page`]) to restrict output to those pages; pass `None` for
/// the whole document. Entries are sorted by `(page, mcid)`.
pub fn extract_structure_elements_mem(
buffer: &[u8],
pages: Option<&[u32]>,
) -> Result<Vec<StructureElement>, PdfError> {
validate_pdf_bytes(buffer)?;
let (doc, _page_count) = load_document_from_mem(buffer)?;
let Some(tree) = structure_tree::StructTree::from_doc(&doc) else {
return Ok(Vec::new());
};
let page_ids = doc.get_pages();
let roles = tree.mcid_to_roles(&page_ids);
let page_filter: Option<HashSet<u32>> = pages.map(|p| p.iter().copied().collect());
let mut elements: Vec<StructureElement> = roles
.into_iter()
.filter(|(page, _)| page_filter.as_ref().is_none_or(|f| f.contains(page)))
.flat_map(|(page, mcids)| {
mcids.into_iter().map(move |(mcid, role)| StructureElement {
page,
mcid,
role: role.name().to_string(),
})
})
.collect();
elements.sort_unstable_by_key(|e| (e.page, e.mcid));
Ok(elements)
}
/// Path-based wrapper for [`extract_structure_elements_mem`].
///
/// Reads the PDF from disk and extracts structure-tree element references.
/// Pass `None` for `pages` to return the whole document, or `Some(&[...])`
/// to restrict to specific 1-indexed pages.
pub fn extract_structure_elements<P: AsRef<Path>>(
path: P,
pages: Option<&[u32]>,
) -> Result<Vec<StructureElement>, PdfError> {
validate_pdf_file(&path)?;
let buffer = std::fs::read(path.as_ref())?;
extract_structure_elements_mem(&buffer, pages)
}
// =========================================================================
// Region-based text extraction (for hybrid OCR pipelines)
// =========================================================================
-89
View File
@@ -271,12 +271,6 @@ pub struct PyTextItem {
pub is_strikeout: bool,
#[pyo3(get)]
pub item_type: String,
/// Marked Content ID from the content stream's BDC/BMC operator, None
/// when the text is not part of marked content. Join with the
/// (page, mcid) pairs from extract_structure_elements to attach
/// structure-tree roles (headings, paragraphs, ...) in tagged PDFs.
#[pyo3(get)]
pub mcid: Option<i64>,
}
#[pymethods]
@@ -292,32 +286,6 @@ impl PyTextItem {
}
}
/// One structure-tree element reference from a tagged PDF.
#[pyclass(name = "StructureElement")]
#[derive(Clone)]
pub struct PyStructureElement {
/// 1-indexed page number (matches TextItem.page).
#[pyo3(get)]
pub page: u32,
/// Marked Content ID from the page's content stream (matches
/// TextItem.mcid).
#[pyo3(get)]
pub mcid: i64,
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", ...).
#[pyo3(get)]
pub role: String,
}
#[pymethods]
impl PyStructureElement {
fn __repr__(&self) -> String {
format!(
"StructureElement(page={}, mcid={}, role='{}')",
self.page, self.mcid, self.role
)
}
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
@@ -388,18 +356,6 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
is_underline: item.is_underline,
is_strikeout: item.is_strikeout,
item_type: item_type_str(&item.item_type),
mcid: item.mcid,
})
.collect()
}
fn convert_structure_elements(elements: Vec<crate::StructureElement>) -> Vec<PyStructureElement> {
elements
.into_iter()
.map(|e| PyStructureElement {
page: e.page,
mcid: e.mcid,
role: e.role,
})
.collect()
}
@@ -657,48 +613,6 @@ fn extract_pages_markdown_bytes(
Ok(to_py_pages_result(result))
}
/// Extract structure-tree element references from a tagged PDF file.
///
/// Parses the document's structure tree (when present) and returns one
/// entry per marked-content reference, resolved to its 1-indexed page,
/// MCID, and structure type name ("H1".."H6", "P", "Table", ...). Returns
/// an empty list when the PDF is not tagged.
///
/// Join (page, mcid) against the page/mcid attributes from
/// [`extract_text_with_positions`] to attach heading levels and other
/// semantic roles to extracted text.
///
/// Args:
/// path: Path to the PDF file.
/// pages: Optional list of 1-indexed pages (matching TextItem.page).
/// When None (default), the whole document is returned.
///
/// Returns:
/// List of StructureElement sorted by (page, mcid).
#[pyfunction]
#[pyo3(signature = (path, pages=None))]
fn extract_structure_elements(
path: &str,
pages: Option<Vec<u32>>,
) -> PyResult<Vec<PyStructureElement>> {
let elements = crate::extract_structure_elements(path, pages.as_deref()).map_err(to_py_err)?;
Ok(convert_structure_elements(elements))
}
/// Extract structure-tree element references from tagged PDF bytes.
///
/// See [`extract_structure_elements`] for details.
#[pyfunction]
#[pyo3(signature = (data, pages=None))]
fn extract_structure_elements_bytes(
data: &[u8],
pages: Option<Vec<u32>>,
) -> PyResult<Vec<PyStructureElement>> {
let elements =
crate::extract_structure_elements_mem(data, pages.as_deref()).map_err(to_py_err)?;
Ok(convert_structure_elements(elements))
}
/// Python module definition.
#[pymodule]
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
@@ -706,7 +620,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<PyPageOcrReasons>()?;
m.add_class::<PyPdfClassification>()?;
m.add_class::<PyTextItem>()?;
m.add_class::<PyStructureElement>()?;
m.add_class::<PyRegionText>()?;
m.add_class::<PyPageRegionTexts>()?;
m.add_class::<PyPageMarkdown>()?;
@@ -721,8 +634,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_function(wrap_pyfunction!(extract_text_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_with_positions, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_structure_elements, m)?)?;
m.add_function(wrap_pyfunction!(extract_structure_elements_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
+38 -819
View File
File diff suppressed because it is too large Load Diff
-71
View File
@@ -1429,77 +1429,6 @@ fn test_firecrawl_tagged_pdf_struct_tree() {
assert_eq!(fence_count % 2, 0, "Code fences should be balanced");
}
#[test]
fn test_tagged_pdf_text_items_carry_mcid() {
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
assert!(
items.iter().any(|i| i.mcid.is_some()),
"Tagged PDF text items should carry Marked Content IDs"
);
}
#[test]
fn test_extract_structure_elements_tagged_pdf() {
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
assert!(!elements.is_empty(), "Tagged PDF should yield elements");
assert!(
elements.iter().any(|e| e.role == "H1"),
"Should surface H1 heading roles"
);
assert!(
elements.iter().all(|e| !e.role.is_empty()),
"Every element should carry a role name"
);
// Sorted by (page, mcid) for deterministic output
assert!(
elements
.windows(2)
.all(|w| (w[0].page, w[0].mcid) <= (w[1].page, w[1].mcid)),
"Elements should be sorted by (page, mcid)"
);
// The advertised join: (page, mcid) pairs must line up with the
// mcid-carrying TextItems from positioned extraction, and joining the
// H1 entries must recover non-empty heading text.
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
let h1_refs: std::collections::HashSet<(u32, i64)> = elements
.iter()
.filter(|e| e.role == "H1")
.map(|e| (e.page, e.mcid))
.collect();
let h1_text: String = items
.iter()
.filter(|i| i.mcid.is_some_and(|mcid| h1_refs.contains(&(i.page, mcid))))
.map(|i| i.text.as_str())
.collect();
assert!(
!h1_text.trim().is_empty(),
"Joining H1 structure elements to text items should recover heading text"
);
// Page filter is 1-indexed (matching TextItem.page) and equals the
// corresponding subset of the full document result.
let page1 = pdf_inspector::extract_structure_elements_mem(&buf, Some(&[1])).unwrap();
assert!(!page1.is_empty(), "Page 1 should have elements");
assert!(page1.iter().all(|e| e.page == 1));
let full_page1_count = elements.iter().filter(|e| e.page == 1).count();
assert_eq!(page1.len(), full_page1_count);
}
#[test]
fn test_extract_structure_elements_untagged_pdf_empty() {
let buf = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
assert!(
elements.is_empty(),
"Untagged PDF should yield no structure elements, got {:?}",
elements
);
}
#[test]
fn test_identity_h_no_tounicode_suppresses_garbage() {
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
-73
View File
@@ -203,79 +203,6 @@ class TestExtractTextWithPositions:
assert len(items) > 0
assert all(item.page == 1 for item in items)
def test_mcid(self):
# Untagged fixture: mcid is None or int, never anything else
items = pdf_inspector.extract_text_with_positions(
fixture_path("thermo-freon12.pdf")
)
assert all(item.mcid is None or isinstance(item.mcid, int) for item in items)
# Tagged fixture: marked content carries MCIDs
tagged = pdf_inspector.extract_text_with_positions(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert any(item.mcid is not None for item in tagged)
# ---------------------------------------------------------------------------
# extract_structure_elements / extract_structure_elements_bytes
# ---------------------------------------------------------------------------
class TestExtractStructureElements:
def test_tagged_file(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert len(elements) > 0
assert all(isinstance(e.page, int) for e in elements)
assert all(isinstance(e.mcid, int) for e in elements)
assert all(isinstance(e.role, str) and len(e.role) > 0 for e in elements)
assert any(e.role == "H1" for e in elements)
def test_join_with_text_items(self):
# (page, mcid) joins against extract_text_with_positions to recover
# heading text
path = fixture_path("firecrawl_docs_tagged.pdf")
elements = pdf_inspector.extract_structure_elements(path)
items = pdf_inspector.extract_text_with_positions(path)
h1_refs = {(e.page, e.mcid) for e in elements if e.role == "H1"}
h1_text = "".join(
item.text
for item in items
if item.mcid is not None and (item.page, item.mcid) in h1_refs
)
assert len(h1_text.strip()) > 0
def test_with_pages(self):
# pages filter is 1-indexed, matching TextItem.page
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf"), pages=[1]
)
assert len(elements) > 0
assert all(e.page == 1 for e in elements)
def test_bytes(self):
data = fixture_bytes("firecrawl_docs_tagged.pdf")
elements = pdf_inspector.extract_structure_elements_bytes(data)
assert len(elements) > 0
assert any(e.role == "H1" for e in elements)
def test_untagged_returns_empty(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("thermo-freon12.pdf")
)
assert elements == []
def test_repr(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert "StructureElement" in repr(elements[0])
def test_not_a_pdf(self):
with pytest.raises(ValueError):
pdf_inspector.extract_structure_elements_bytes(b"not a pdf")
# ---------------------------------------------------------------------------
# extract_text_in_regions / extract_text_in_regions_bytes
+2 -2
View File
@@ -724,7 +724,7 @@ checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
[[package]]
name = "pdf-inspector"
version = "0.1.8"
version = "0.1.7"
dependencies = [
"env_logger",
"include_dir",
@@ -740,7 +740,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-wasm"
version = "0.1.4"
version = "0.1.3"
dependencies = [
"console_error_panic_hook",
"js-sys",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-wasm"
version = "0.1.4"
version = "0.1.3"
edition = "2021"
authors = ["Firecrawl Team"]
description = "Browser WebAssembly bindings for pdf-inspector"