Compare commits

..
Author SHA1 Message Date
Abimael Martell 399e83caf2 fix(site): refresh benchmark results 2026-08-06 12:44:34 -07:00
25 changed files with 140 additions and 2748 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector"
version = "0.1.8"
version = "0.1.7"
edition = "2021"
autobins = false
authors = ["Firecrawl Team"]
+3 -4
View File
@@ -5,15 +5,14 @@
If you believe you've found a security vulnerability in pdf-inspector, please
report it privately so we can fix it before public disclosure.
**Preferred:** Submit through Firecrawl's Bugcrowd vulnerability disclosure
program at <https://bugcrowd.com/engagements/firecrawl-vdp-ess>. Please include:
**Preferred:** Email **help@firecrawl.dev** with:
- A description of the issue and its impact
- Steps to reproduce (a minimal PDF or input that triggers the bug is ideal)
- The version or commit hash of pdf-inspector you tested against
**Alternative:** If you'd rather not use Bugcrowd, email
**help@firecrawl.dev** with the same details.
**Alternative:** Use GitHub's private vulnerability reporting under the
[Security tab](https://github.com/firecrawl/pdf-inspector/security/advisories/new).
We'll acknowledge your report in a timely manner and keep you updated on
remediation progress. Please do not open a public GitHub issue for security
+2 -2
View File
@@ -851,7 +851,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
[[package]]
name = "pdf-inspector"
version = "0.1.8"
version = "0.1.7"
dependencies = [
"env_logger",
"include_dir",
@@ -867,7 +867,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-napi"
version = "0.2.3"
version = "0.2.2"
dependencies = [
"napi",
"napi-build",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-napi"
version = "0.2.3"
version = "0.2.2"
edition = "2021"
[lib]
-16
View File
@@ -83,22 +83,6 @@ for (const region of result[0].regions) {
}
```
### Async variants
`processPdf`, `classifyPdf`, and `extractPagesMarkdown` are synchronous and parse on the calling thread — in Node, that's the event loop. For a one-off call in a script that's fine, but in a server a large document can hold the loop for tens to hundreds of milliseconds.
`processPdfAsync`, `classifyPdfAsync`, and `extractPagesMarkdownAsync` take the same arguments and produce the same results, but run the parse on the libuv thread pool and return a promise, keeping the event loop free. The input buffer is copied before the call returns, so it's safe to reuse or mutate immediately:
```typescript
import { classifyPdfAsync, extractPagesMarkdownAsync } from '@firecrawl/pdf-inspector'
const classification = await classifyPdfAsync(pdf)
if (classification.pdfType === 'TextBased') {
const { pages } = await extractPagesMarkdownAsync(pdf)
// ...
}
```
## Types
```typescript
+6 -6
View File
@@ -8,12 +8,12 @@
"@napi-rs/cli": "^3.4.1",
},
"optionalDependencies": {
"@firecrawl/pdf-inspector-darwin-arm64": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.13.0",
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.13.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.13.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
},
},
},
+7 -7
View File
@@ -1,6 +1,6 @@
{
"name": "@firecrawl/pdf-inspector",
"version": "1.13.0",
"version": "1.12.0",
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
"main": "index.js",
"types": "index.d.ts",
@@ -52,11 +52,11 @@
"@napi-rs/cli": "^3.4.1"
},
"optionalDependencies": {
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.13.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.13.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.13.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.13.0"
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
}
}
+41 -236
View File
@@ -89,11 +89,6 @@ pub struct TextItem {
pub item_type: ItemType,
/// URL for link items, `None` for other types.
pub link_url: Option<String>,
/// Marked Content ID from the content stream's BDC/BMC operator, `None`
/// when the text is not part of marked content. Join with the
/// `page`/`mcid` pairs from [`extractStructureElements`] to attach
/// structure-tree roles (headings, paragraphs, …) in tagged PDFs.
pub mcid: Option<i64>,
}
/// A page's regions for text extraction: (page_index_0based, bboxes).
@@ -158,7 +153,9 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
}
}
fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<PageOcrReasons> {
fn to_napi_page_ocr_reasons(
reasons: Vec<pdf_inspector::PageOcrReasons>,
) -> Vec<PageOcrReasons> {
reasons
.into_iter()
.map(|reason| PageOcrReasons {
@@ -205,31 +202,6 @@ where
}
}
// ---------------------------------------------------------------------------
// Shared implementations (single body behind sync and async entry points)
// ---------------------------------------------------------------------------
fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult> {
let mut opts = pdf_inspector::PdfOptions::new();
if let Some(p) = pages {
opts = opts.pages(p);
}
let result = pdf_inspector::process_pdf_mem_with_options(bytes, opts)
.map_err(|e| to_napi_err(e, "process_pdf"))?;
Ok(to_napi_result(result))
}
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
let result =
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
Ok(PdfClassification {
pdf_type: convert_pdf_type(result.pdf_type),
page_count: result.page_count,
pages_needing_ocr: result.pages_needing_ocr,
confidence: result.confidence as f64,
})
}
// ---------------------------------------------------------------------------
// Public NAPI API
// ---------------------------------------------------------------------------
@@ -238,7 +210,15 @@ fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
#[napi]
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
let bytes: Vec<u8> = buffer.to_vec();
catch_panic("process_pdf", move || process_pdf_impl(&bytes, pages))
catch_panic("process_pdf", move || {
let mut opts = pdf_inspector::PdfOptions::new();
if let Some(p) = pages {
opts = opts.pages(p);
}
let result = pdf_inspector::process_pdf_mem_with_options(&bytes, opts)
.map_err(|e| to_napi_err(e, "process_pdf"))?;
Ok(to_napi_result(result))
})
}
/// Fast detection only — no text extraction or markdown.
@@ -258,7 +238,16 @@ pub fn detect_pdf(buffer: Buffer) -> Result<PdfResult> {
#[napi]
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
let bytes: Vec<u8> = buffer.to_vec();
catch_panic("classify_pdf", move || classify_pdf_impl(&bytes))
catch_panic("classify_pdf", move || {
let result =
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
Ok(PdfClassification {
pdf_type: convert_pdf_type(result.pdf_type),
page_count: result.page_count,
pages_needing_ocr: result.pages_needing_ocr,
confidence: result.confidence as f64,
})
})
}
/// Extract plain text from a PDF Buffer.
@@ -311,61 +300,12 @@ pub fn extract_text_with_positions(
is_strikeout: item.is_strikeout,
item_type,
link_url,
mcid: item.mcid,
}
})
.collect())
})
}
/// One structure-tree element reference from a tagged PDF.
#[napi(object)]
pub struct StructureElementJs {
/// 1-indexed page number (matches `TextItem.page`).
pub page: u32,
/// Marked Content ID from the page's content stream (matches
/// `TextItem.mcid`).
pub mcid: i64,
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
/// Custom tags are resolved through the document's role map; tags with
/// no standard mapping are returned verbatim.
pub role: String,
}
/// Extract structure-tree element references from a tagged PDF.
///
/// Parses the document's structure tree (when present) and returns one
/// entry per marked-content reference, resolved to its 1-indexed page,
/// MCID, and structure type name. Returns an empty array when the PDF is
/// not tagged.
///
/// Join `(page, mcid)` against the `page`/`mcid` fields from
/// [`extractTextWithPositions`] to attach heading levels (H1..H6) and other
/// semantic roles to extracted text.
///
/// Pass 1-indexed page numbers (matching `TextItem.page`) to restrict
/// output; omit `pages` for the whole document. Entries are sorted by
/// `(page, mcid)`.
#[napi]
pub fn extract_structure_elements(
buffer: Buffer,
pages: Option<Vec<u32>>,
) -> Result<Vec<StructureElementJs>> {
let bytes: Vec<u8> = buffer.to_vec();
catch_panic("extract_structure_elements", move || {
let elements = pdf_inspector::extract_structure_elements_mem(&bytes, pages.as_deref())
.map_err(|e| to_napi_err(e, "extract_structure_elements"))?;
Ok(elements
.into_iter()
.map(|e| StructureElementJs {
page: e.page,
mcid: e.mcid,
role: e.role,
})
.collect())
})
}
/// Extract text within bounding-box regions from a PDF.
///
/// For hybrid OCR: layout model detects regions in rendered images,
@@ -693,32 +633,25 @@ pub fn extract_pages_markdown(
) -> Result<PagesExtractionResult> {
let bytes: Vec<u8> = buffer.to_vec();
catch_panic("extract_pages_markdown", move || {
extract_pages_markdown_impl(&bytes, pages.as_deref())
})
}
fn extract_pages_markdown_impl(
bytes: &[u8],
pages: Option<&[u32]>,
) -> Result<PagesExtractionResult> {
let result = pdf_inspector::extract_pages_markdown_mem(bytes, pages)
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
Ok(PagesExtractionResult {
pages: result
.pages
.into_iter()
.map(|r| PageMarkdownResult {
page: r.page,
markdown: r.markdown,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
pages_with_tables: result.pages_with_tables,
pages_with_columns: result.pages_with_columns,
pages_needing_ocr: result.pages_needing_ocr,
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
is_complex: result.is_complex,
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
Ok(PagesExtractionResult {
pages: result
.pages
.into_iter()
.map(|r| PageMarkdownResult {
page: r.page,
markdown: r.markdown,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
pages_with_tables: result.pages_with_tables,
pages_with_columns: result.pages_with_columns,
pages_needing_ocr: result.pages_needing_ocr,
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
is_complex: result.is_complex,
})
})
}
@@ -759,131 +692,3 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
})
.collect()
}
// ---------------------------------------------------------------------------
// Async variants (libuv thread pool via AsyncTask)
//
// The synchronous exports above parse on the calling thread, which in Node is
// the event loop. These `*Async` variants run the same shared implementations
// on the libuv thread pool and hand JavaScript a promise, so servers under
// concurrent load keep answering requests while a document parses. The sync
// exports keep their names, signatures, and behaviour.
//
// Each factory copies the input Buffer to an owned `Vec<u8>` on the calling
// (JS) thread — deliberately. JS execution is single-threaded, so no JS code
// can mutate the buffer while the synchronous part of the call copies it.
// Holding the napi `Buffer` and reading it from the worker instead would be
// zero-copy, but a caller mutating the buffer before the promise settles
// would then race the worker's reads — undefined behavior, not a recoverable
// error (a known napi-rs soundness hazard with cross-thread Buffer access).
// The copy is a one-time memcpy, negligible next to the parse it unblocks.
// ---------------------------------------------------------------------------
pub struct ProcessPdfTask {
bytes: Vec<u8>,
pages: Option<Vec<u32>>,
}
impl Task for ProcessPdfTask {
type Output = PdfResult;
type JsValue = PdfResult;
fn compute(&mut self) -> Result<Self::Output> {
let bytes = std::mem::take(&mut self.bytes);
let pages = self.pages.take();
// AssertUnwindSafe: `bytes`/`pages` are moved into the closure and
// dropped on unwind — no shared state can be observed broken.
catch_panic(
"process_pdf",
panic::AssertUnwindSafe(move || process_pdf_impl(&bytes, pages)),
)
}
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
Ok(output)
}
}
/// Async variant of [`processPdf`]: same result, but the parse runs on the
/// libuv thread pool instead of the event loop and the call returns a
/// promise. The buffer is copied before the call returns, so it may be
/// reused or mutated immediately.
// ts_return_type is required: napi-rs emits `Promise<unknown>` for
// `AsyncTask<T>` returns without it.
#[napi(ts_return_type = "Promise<PdfResult>")]
pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<ProcessPdfTask> {
AsyncTask::new(ProcessPdfTask {
bytes: buffer.to_vec(),
pages,
})
}
pub struct ClassifyPdfTask {
bytes: Vec<u8>,
}
impl Task for ClassifyPdfTask {
type Output = PdfClassification;
type JsValue = PdfClassification;
fn compute(&mut self) -> Result<Self::Output> {
let bytes = std::mem::take(&mut self.bytes);
catch_panic(
"classify_pdf",
panic::AssertUnwindSafe(move || classify_pdf_impl(&bytes)),
)
}
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
Ok(output)
}
}
/// Async variant of [`classifyPdf`]: same result, but the classification runs
/// on the libuv thread pool instead of the event loop and the call returns a
/// promise. The buffer is copied before the call returns, so it may be
/// reused or mutated immediately.
#[napi(ts_return_type = "Promise<PdfClassification>")]
pub fn classify_pdf_async(buffer: Buffer) -> AsyncTask<ClassifyPdfTask> {
AsyncTask::new(ClassifyPdfTask {
bytes: buffer.to_vec(),
})
}
pub struct ExtractPagesMarkdownTask {
bytes: Vec<u8>,
pages: Option<Vec<u32>>,
}
impl Task for ExtractPagesMarkdownTask {
type Output = PagesExtractionResult;
type JsValue = PagesExtractionResult;
fn compute(&mut self) -> Result<Self::Output> {
let bytes = std::mem::take(&mut self.bytes);
let pages = self.pages.take();
catch_panic(
"extract_pages_markdown",
panic::AssertUnwindSafe(move || extract_pages_markdown_impl(&bytes, pages.as_deref())),
)
}
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
Ok(output)
}
}
/// Async variant of [`extractPagesMarkdown`]: same result, but the extraction
/// runs on the libuv thread pool instead of the event loop and the call
/// returns a promise. The buffer is copied before the call returns, so it
/// may be reused or mutated immediately.
#[napi(ts_return_type = "Promise<PagesExtractionResult>")]
pub fn extract_pages_markdown_async(
buffer: Buffer,
pages: Option<Vec<u32>>,
) -> AsyncTask<ExtractPagesMarkdownTask> {
AsyncTask::new(ExtractPagesMarkdownTask {
bytes: buffer.to_vec(),
pages,
})
}
-110
View File
@@ -2,21 +2,16 @@ import { readFileSync } from 'fs';
import { strict as assert } from 'assert';
import {
processPdf,
processPdfAsync,
detectPdf,
classifyPdf,
classifyPdfAsync,
extractText,
extractTextWithPositions,
extractStructureElements,
extractTextInRegions,
detectVectorGridInRegion,
extractPagesMarkdown,
extractPagesMarkdownAsync,
} from './index.js';
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
const taggedFixture = readFileSync('../tests/fixtures/firecrawl_docs_tagged.pdf');
// --- processPdf ---
console.log('Testing processPdf...');
@@ -84,46 +79,6 @@ assert.ok(page1Items.length > 0);
assert.ok(page1Items.every(i => i.page === 1));
console.log(' extractTextWithPositions with pages: OK');
// mcid: undefined on untagged PDFs, numeric on tagged marked content
assert.ok(items.every(i => i.mcid === undefined || typeof i.mcid === 'number'));
const taggedItems = extractTextWithPositions(taggedFixture);
assert.ok(
taggedItems.some(i => typeof i.mcid === 'number'),
'tagged PDF text items should carry Marked Content IDs',
);
console.log(' extractTextWithPositions mcid: OK');
// --- extractStructureElements ---
console.log('Testing extractStructureElements...');
const structureElements = extractStructureElements(taggedFixture);
assert.ok(structureElements.length > 0);
assert.ok(structureElements.every(e => typeof e.page === 'number'));
assert.ok(structureElements.every(e => typeof e.mcid === 'number'));
assert.ok(structureElements.every(e => typeof e.role === 'string' && e.role.length > 0));
assert.ok(
structureElements.some(e => e.role === 'H1'),
'tagged fixture should surface H1 heading roles',
);
// (page, mcid) joins against extractTextWithPositions to recover heading text
const h1Refs = new Set(
structureElements.filter(e => e.role === 'H1').map(e => `${e.page}:${e.mcid}`),
);
const h1Text = taggedItems
.filter(i => typeof i.mcid === 'number' && h1Refs.has(`${i.page}:${i.mcid}`))
.map(i => i.text)
.join('');
assert.ok(h1Text.trim().length > 0, 'H1 join should recover heading text');
// pages filter is 1-indexed, matching TextItem.page
const page1Elements = extractStructureElements(taggedFixture, [1]);
assert.ok(page1Elements.length > 0);
assert.ok(page1Elements.every(e => e.page === 1));
// untagged PDFs yield an empty array
assert.deepEqual(extractStructureElements(fixture), []);
console.log(' extractStructureElements: OK');
// --- extractTextInRegions ---
console.log('Testing extractTextInRegions...');
const regionResults = extractTextInRegions(fixture, [
@@ -169,75 +124,10 @@ assert.equal(picked.pages[0].page, 2);
assert.equal(picked.pages[1].page, 0);
console.log(' extractPagesMarkdown with pages: OK');
// --- Async variants ---
console.log('Testing async variants...');
// processPdfAsync returns a promise and matches the sync result
const asyncResultPromise = processPdfAsync(fixture);
assert.ok(asyncResultPromise instanceof Promise);
const asyncResult = await asyncResultPromise;
assert.equal(asyncResult.pdfType, result.pdfType);
assert.equal(asyncResult.pageCount, result.pageCount);
assert.equal(asyncResult.markdown, result.markdown);
console.log(' processPdfAsync: OK');
// processPdfAsync with pages
const asyncResult2 = await processPdfAsync(fixture, [1]);
assert.equal(asyncResult2.markdown, result2.markdown);
console.log(' processPdfAsync with pages: OK');
// classifyPdfAsync matches the sync result
const asyncClassified = await classifyPdfAsync(fixture);
assert.equal(asyncClassified.pdfType, classified.pdfType);
assert.equal(asyncClassified.pageCount, classified.pageCount);
assert.equal(asyncClassified.confidence, classified.confidence);
assert.deepEqual(asyncClassified.pagesNeedingOcr, classified.pagesNeedingOcr);
console.log(' classifyPdfAsync: OK');
// extractPagesMarkdownAsync matches the sync result
const asyncAllPages = await extractPagesMarkdownAsync(fixture);
assert.equal(asyncAllPages.pages.length, allPages.pages.length);
assert.deepEqual(
asyncAllPages.pages.map(p => p.markdown),
allPages.pages.map(p => p.markdown),
);
assert.equal(asyncAllPages.isComplex, allPages.isComplex);
console.log(' extractPagesMarkdownAsync: OK');
// selected pages preserve caller order
const asyncPicked = await extractPagesMarkdownAsync(fixture, [2, 0]);
assert.equal(asyncPicked.pages.length, 2);
assert.equal(asyncPicked.pages[0].page, 2);
assert.equal(asyncPicked.pages[1].page, 0);
console.log(' extractPagesMarkdownAsync with pages: OK');
// input buffer is copied at call time: mutating it immediately after the
// call must not affect the in-flight parse
const scratch = Buffer.from(fixture);
const inFlight = processPdfAsync(scratch);
scratch.fill(0);
const fromMutated = await inFlight;
assert.equal(fromMutated.markdown, result.markdown);
console.log(' processPdfAsync input copied at call time: OK');
// concurrent async calls all settle
const [c1, c2, c3] = await Promise.all([
processPdfAsync(fixture),
classifyPdfAsync(fixture),
extractPagesMarkdownAsync(fixture),
]);
assert.equal(c1.pdfType, 'TextBased');
assert.equal(c2.pdfType, 'TextBased');
assert.equal(c3.pages.length, 3);
console.log(' concurrent async calls: OK');
// --- Error handling ---
console.log('Testing error handling...');
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
await assert.rejects(processPdfAsync(Buffer.from('not a pdf')), /process_pdf/);
await assert.rejects(classifyPdfAsync(Buffer.from('')), /classify_pdf/);
await assert.rejects(extractPagesMarkdownAsync(Buffer.from('')), /extract_pages_markdown/);
console.log(' error handling: OK');
console.log('\nAll NAPI tests passed!');
-35
View File
@@ -51,20 +51,6 @@ class TextItem:
is_underline: bool
is_strikeout: bool
item_type: str
mcid: Optional[int]
"""Marked Content ID from the content stream's BDC/BMC operator, None when
the text is not part of marked content. Join with the (page, mcid) pairs
from extract_structure_elements to attach structure-tree roles in tagged
PDFs."""
class StructureElement:
"""One structure-tree element reference from a tagged PDF."""
page: int
"""1-indexed page number (matches TextItem.page)."""
mcid: int
"""Marked Content ID from the page's content stream (matches TextItem.mcid)."""
role: str
"""Standard structure type name ("H1".."H6", "P", "Table", "TD", ...)."""
class RegionText:
"""Extracted text for a single region."""
@@ -146,27 +132,6 @@ def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] =
"""Extract text with position information from bytes."""
...
def extract_structure_elements(path: str, pages: Optional[list[int]] = None) -> list[StructureElement]:
"""Extract structure-tree element references from a tagged PDF file.
Returns one entry per marked-content reference, resolved to its 1-indexed
page, MCID, and structure type name ("H1".."H6", "P", "Table", ...), sorted
by (page, mcid). Returns an empty list when the PDF is not tagged.
Args:
path: Path to the PDF file.
pages: Optional list of 1-indexed pages (matching ``TextItem.page``).
When ``None`` (default), the whole document is returned.
"""
...
def extract_structure_elements_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[StructureElement]:
"""Extract structure-tree element references from tagged PDF bytes.
See :func:`extract_structure_elements` for details.
"""
...
def extract_text_in_regions(
path: str,
page_regions: list[tuple[int, list[list[float]]]],
+1 -1
View File
@@ -6,7 +6,7 @@ build-backend = "maturin"
name = "pdf-inspector"
# Bump this to publish to PyPI — CI publishes automatically when the version
# changes on main (same flow as napi/package.json for npm).
version = "0.2.7"
version = "0.2.6"
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
readme = "docs/python.md"
license = { text = "MIT" }
+17 -327
View File
@@ -41,110 +41,15 @@ pub(crate) fn detect_columns(
}
debug!("page {}: detect_columns: {} items", page, page_items.len());
// The width of one ordinary page, used three ways below: as the largest
// credible width for a single text run, as the size of empty gap that marks
// content as detached, and as the span past which those checks run at all.
// This is a heuristic, not a format rule: PDF 2.0 sets no page-size limit,
// and since PDF 1.6 `UserUnit` scales a page's physical size independently
// of its coordinates. 14_400 units (200in at the default 1/72in unit) is
// the traditional Acrobat architectural limit, which makes it a reasonable
// "wider than any ordinary page" mark in coordinate space.
const MAX_PAGE_EXTENT: f32 = 14_400.0;
// A detached cluster is only dropped if it also holds a small minority of
// the items, so a genuine two-part layout keeps its full bounds even when
// the halves are far apart.
const MAX_TRIM_FRACTION: f32 = 0.10;
// Position and width of each item, skipping only non-finite geometry.
let finite_span = |i: &&TextItem| -> Option<(f32, f32)> {
let (left, width) = (i.x, effective_width(i));
(left.is_finite() && (left + width).is_finite()).then_some((left, width))
};
let (min_left, max_right, total) = page_items.iter().filter_map(finite_span).fold(
(f32::INFINITY, f32::NEG_INFINITY, 0usize),
|(lo, hi, n), (left, width)| (lo.min(left), hi.max(left + width), n + 1),
);
// No item had usable geometry, so there is no layout to report.
if total == 0 {
return vec![];
}
// Every threshold below (gutter margins, spanning-item width, the XY-cut
// margin) is a fraction of the page width, so a far item can set the scale
// for the whole page and shrink the effective detection window to a
// rounding error — real gutters then fall inside the margin band and a
// genuine multi-column page collapses to one region.
//
// Anything inside one page extent is ordinary, so the common case keeps the
// plain bounds and skips the work below entirely.
let (x_min, x_max) = if max_right - min_left <= MAX_PAGE_EXTENT {
(min_left, max_right)
} else {
// Discarding content needs positive evidence that it is not part of the
// layout, because a count-based rule alone cannot tell a stray from a
// sparse far sidebar. The evidence is geometric: positions are grouped
// into clusters separated by more than a whole page of continuous
// emptiness. Real content, however sparse, does not leave a void that
// large; a malformed coordinate sits alone beyond one.
let mut spans: Vec<(f32, f32)> = page_items.iter().filter_map(finite_span).collect();
spans.sort_by(|a, b| a.0.total_cmp(&b.0));
let mut core: Option<std::ops::Range<usize>> = None;
let mut start = 0usize;
for i in 1..=spans.len() {
if i < spans.len() && spans[i].0 - spans[i - 1].0 <= MAX_PAGE_EXTENT {
continue;
}
if core.as_ref().is_none_or(|best| i - start > best.len()) {
core = Some(start..i);
}
start = i;
}
let mut core = core.unwrap_or(0..spans.len());
// Only drop the detached clusters when they are a small minority, so a
// genuine two-part layout keeps its full bounds.
let dropped = spans.len() - core.len();
if dropped as f32 > spans.len() as f32 * MAX_TRIM_FRACTION {
core = 0..spans.len();
}
let core = &spans[core];
// Positions cannot be inflated by a bogus width, so the spread of the
// content is a sound scale for judging one. A run much wider than the
// page's own content is a malformed width — the test is relative, so a
// genuinely large page keeps its genuinely long runs.
let (lo, widest_left) = (core[0].0, core[core.len() - 1].0);
let max_run_width = (widest_left - lo) + MAX_PAGE_EXTENT;
let hi = core
.iter()
.filter(|&&(_, width)| width <= max_run_width)
.map(|&(left, width)| left + width)
.fold(widest_left, f32::max);
if lo != min_left || hi != max_right {
debug!(
"page {page}: bounds {min_left}..{max_right} exceed one page; \
dropped {dropped}/{} detached item(s), using {lo}..{hi}",
spans.len()
);
}
(lo, hi)
};
// Hard ceiling on the histogram size, independent of the trimming above:
// the bounds are attacker-influenced, so an unclamped
// `page_width / BIN_WIDTH` lets a crafted PDF force an arbitrarily large
// `vec![0u32; num_bins]` allocation. 65_536 bins covers ~128k points at
// BIN_WIDTH 2.0 — roughly 9x the largest legal page — so this never binds
// on a real layout. Kept as a bound that does not depend on the outlier
// heuristic staying correct.
const MAX_BINS: usize = 65_536;
// Find page bounds
let x_min = page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
let x_max = page_items
.iter()
.map(|i| i.x + effective_width(i))
.fold(f32::NEG_INFINITY, f32::max);
let page_width = x_max - x_min;
if !page_width.is_finite() || page_width < 200.0 {
if page_width < 200.0 {
return vec![ColumnRegion { x_min, x_max }];
}
@@ -152,20 +57,13 @@ pub(crate) fn detect_columns(
return vec![ColumnRegion { x_min, x_max }];
}
// Widen the bins rather than dropping the tail of the page. Clamping the
// count alone would leave anything past MAX_BINS * BIN_WIDTH outside the
// histogram, folded into the last bin, which places gutters at the wrong
// coordinates. Scaling keeps full coverage under the same allocation
// ceiling; only the resolution degrades, and only beyond ~131k points.
let bin_width = BIN_WIDTH.max(page_width / MAX_BINS as f32);
// Build occupancy histogram.
// Exclude items wider than 60% of page width — these are spanning items
// (titles, full-width paragraphs) that would fill the gutter and prevent
// detection of partial-page column layouts (e.g. two-column abstracts on
// a page that also has single-column introduction text).
let wide_threshold = page_width * 0.6;
let num_bins = ((page_width / bin_width).ceil() as usize).clamp(1, MAX_BINS);
let num_bins = ((page_width / BIN_WIDTH).ceil() as usize).max(1);
let mut histogram = vec![0u32; num_bins];
for item in &page_items {
@@ -173,8 +71,8 @@ pub(crate) fn detect_columns(
if w > wide_threshold {
continue;
}
let left = ((item.x - x_min) / bin_width).floor() as usize;
let right = (((item.x + w) - x_min) / bin_width).ceil() as usize;
let left = ((item.x - x_min) / BIN_WIDTH).floor() as usize;
let right = (((item.x + w) - x_min) / BIN_WIDTH).ceil() as usize;
let left = left.min(num_bins);
let right = right.min(num_bins);
for count in histogram.iter_mut().take(right).skip(left) {
@@ -211,12 +109,12 @@ pub(crate) fn detect_columns(
let valleys: Vec<(usize, usize)> = valleys
.into_iter()
.filter(|&(start, end)| {
let width_pts = (end - start) as f32 * bin_width;
let width_pts = (end - start) as f32 * BIN_WIDTH;
if width_pts < MIN_GUTTER_WIDTH {
return false;
}
// Valley center must not be within 5% of page edges
let center_pts = ((start + end) as f32 / 2.0) * bin_width;
let center_pts = ((start + end) as f32 / 2.0) * BIN_WIDTH;
center_pts > margin_threshold && center_pts < (page_width - margin_threshold)
})
.collect();
@@ -234,7 +132,7 @@ pub(crate) fn detect_columns(
&histogram,
num_bins,
x_min,
bin_width,
BIN_WIDTH,
page_width,
margin_threshold,
);
@@ -243,7 +141,7 @@ pub(crate) fn detect_columns(
&rel_valleys,
&page_items,
x_min,
bin_width,
BIN_WIDTH,
x_max,
MIN_ITEMS_PER_COLUMN,
MIN_VERTICAL_SPAN_RATIO,
@@ -284,7 +182,7 @@ pub(crate) fn detect_columns(
&valleys,
&page_items,
x_min,
bin_width,
BIN_WIDTH,
x_max,
MIN_ITEMS_PER_COLUMN,
MIN_VERTICAL_SPAN_RATIO,
@@ -298,7 +196,7 @@ pub(crate) fn detect_columns(
&valleys,
&page_items,
x_min,
bin_width,
BIN_WIDTH,
x_max,
MIN_ITEMS_PER_COLUMN,
MIN_VERTICAL_SPAN_RATIO,
@@ -1929,7 +1827,7 @@ fn split_column_stragglers(lines: Vec<TextLine>) -> (Vec<TextLine>, Vec<TextLine
.unwrap();
let (cs, ce) = segments[core_seg];
let mut core = Vec::with_capacity(ce.saturating_sub(cs));
let mut core = Vec::with_capacity(ce - cs);
let mut stragglers = Vec::new();
for (i, line) in lines.into_iter().enumerate() {
if i >= cs && i < ce {
@@ -2635,214 +2533,6 @@ mod tests {
);
}
#[test]
fn extreme_far_coordinate_does_not_allocate_unboundedly() {
// A crafted PDF can place a text run at an arbitrary coordinate via the
// text matrix. The derived page width must not drive an unbounded
// histogram allocation (previously `page_width / BIN_WIDTH` bins with no
// upper bound would try to reserve terabytes and abort the process).
let mut items = Vec::new();
for i in 0..24 {
items.push(make_item(1, i as f32 * 10.0, 700.0 - i as f32 * 5.0, "A"));
}
// Item placed 1e12 points away — 5e11 bins if left unclamped.
items.push(make_item(1, 1e12, 700.0, "Z"));
// Must return without aborting; content is preserved as a single region.
let cols = detect_columns(&items, 1, false);
assert!(!cols.is_empty());
}
#[test]
fn non_finite_coordinates_never_leak_into_region_bounds() {
// An inf/NaN coordinate must not escape as a column boundary: callers
// treat these as page/column edges.
for bad_x in [f32::INFINITY, f32::NEG_INFINITY, f32::NAN] {
let mut items = Vec::new();
for i in 0..24 {
items.push(make_item(1, i as f32 * 10.0, 700.0 - i as f32 * 5.0, "A"));
}
items.push(make_item(1, bad_x, 700.0, "Z"));
for col in detect_columns(&items, 1, false) {
assert!(
col.x_min.is_finite() && col.x_max.is_finite(),
"bad_x {bad_x} leaked bounds {}..{}",
col.x_min,
col.x_max
);
}
}
}
#[test]
fn all_non_finite_coordinates_yield_no_columns() {
let items: Vec<TextItem> = (0..24)
.map(|i| make_item(1, f32::NAN, 700.0 - i as f32 * 5.0, "A"))
.collect();
assert!(detect_columns(&items, 1, false).is_empty());
}
#[test]
fn one_bad_item_does_not_disable_column_detection() {
// A single stray item should not collapse a clean two-column page to
// one region. Every gutter threshold is a fraction of the page width,
// so an untrimmed outlier pushes real gutters inside the rejected
// margin band. A malformed *width* at an ordinary position poisons the
// bounds just as a malformed position does.
for (label, bad_x, bad_width) in [
("nan position", f32::NAN, 0.0),
("inf position", f32::INFINITY, 0.0),
("far position", 50_000.0, 0.0),
("very far position", 1e12, 0.0),
("huge width", 100.0, 1e12),
("inf width", 100.0, f32::INFINITY),
] {
let mut items = Vec::new();
items.extend(fill_zone(1, 30.0, 280.0, 750.0, 50.0));
items.extend(fill_zone(1, 320.0, 570.0, 750.0, 50.0));
let mut bad = make_item(1, bad_x, 400.0, "Z");
bad.width = bad_width;
items.push(bad);
let cols = detect_columns(&items, 1, false);
assert_eq!(
cols.len(),
2,
"{label}: expected 2 columns, got {}",
cols.len()
);
for col in &cols {
assert!(
col.x_max - col.x_min <= MAX_PAGE_EXTENT_FOR_TEST,
"{label}: region {}..{} exceeds one page",
col.x_min,
col.x_max
);
}
}
}
/// Mirrors `MAX_PAGE_EXTENT` in `detect_columns`.
const MAX_PAGE_EXTENT_FOR_TEST: f32 = 14_400.0;
#[test]
fn very_wide_page_keeps_full_histogram_coverage() {
// Beyond MAX_BINS * BIN_WIDTH (~131k points) the bins must widen rather
// than stop covering the page. Three zones: the first gutter is inside
// the old coverage limit, the second is past it. Because the first
// gutter is found, the XY-cut fallback never runs, so a truncated
// histogram silently reports two columns instead of three.
let mut items = Vec::new();
items.extend(fill_zone(1, 0.0, 60_000.0, 750.0, 700.0));
items.extend(fill_zone(1, 70_000.0, 140_000.0, 750.0, 700.0));
items.extend(fill_zone(1, 160_000.0, 200_000.0, 750.0, 700.0));
let cols = detect_columns(&items, 1, false);
assert_eq!(
cols.len(),
3,
"Expected 3 columns across a 200k-wide page, got {}",
cols.len()
);
assert!(
(140_000.0..=160_000.0).contains(&cols[1].x_max),
"second gutter at {}, expected inside the real 140k..160k gap",
cols[1].x_max
);
}
#[test]
fn large_page_with_legitimately_long_runs_is_kept() {
// On a very large page, individual runs can exceed one ordinary page's
// width. They are real content, so they must not be judged malformed:
// the page keeps its columns and its full right edge.
let mut items = Vec::new();
for row in 0..30 {
let y = 750.0 - row as f32 * 14.0;
let mut left = make_item(1, 0.0, y, "Left run");
left.width = 20_000.0;
let mut right = make_item(1, 25_000.0, y, "Right run");
right.width = 20_000.0;
items.extend([left, right]);
}
let cols = detect_columns(&items, 1, false);
assert!(
!cols.is_empty(),
"a page of long-but-valid runs must still report a layout"
);
let right_edge = cols
.iter()
.map(|c| c.x_max)
.fold(f32::NEG_INFINITY, f32::max);
assert!(
right_edge > 44_000.0,
"long runs were treated as malformed: right edge {right_edge}, expected ~45_000"
);
}
#[test]
fn sparse_far_sidebar_on_a_large_page_is_kept() {
// A large-format page with a thin, sparsely-populated sidebar far from
// the main block. The sidebar is a small minority of the items, so an
// item-count rule alone would discard it — but nothing about its
// geometry says it is invalid, so its bounds must survive.
let mut items = Vec::new();
items.extend(fill_zone(1, 0.0, 12_000.0, 750.0, 500.0));
for i in 0..12 {
items.push(make_item(1, 24_000.0, 750.0 - i as f32 * 14.0, "Sidebar"));
}
let cols = detect_columns(&items, 1, false);
let right_edge = cols
.iter()
.map(|c| c.x_max)
.fold(f32::NEG_INFINITY, f32::max);
assert!(
right_edge > 24_000.0,
"sidebar was trimmed away: right edge {right_edge}, expected >24_000"
);
}
#[test]
fn genuinely_wide_layout_keeps_its_true_bounds() {
// A large-format page whose content really is spread beyond one
// ordinary page must not be trimmed to the median cluster: its far
// items are the majority, not strays.
let mut items = Vec::new();
items.extend(fill_zone(1, 100.0, 20_000.0, 750.0, 600.0));
items.extend(fill_zone(1, 22_000.0, 40_000.0, 750.0, 600.0));
let cols = detect_columns(&items, 1, false);
let widest = cols
.iter()
.map(|c| c.x_max)
.fold(f32::NEG_INFINITY, f32::max);
assert!(
widest > 35_000.0,
"wide layout was trimmed: right edge {widest}, expected ~40_000"
);
}
#[test]
fn oversized_but_legal_page_is_not_trimmed() {
// A wide-format page well inside the 14_400pt spec limit must keep its
// real bounds — outlier trimming is only for spans beyond a legal page.
let mut items = Vec::new();
items.extend(fill_zone(1, 100.0, 4_000.0, 750.0, 400.0));
items.extend(fill_zone(1, 4_400.0, 8_000.0, 750.0, 400.0));
let cols = detect_columns(&items, 1, false);
assert_eq!(cols.len(), 2, "Expected 2 columns, got {}", cols.len());
assert!(
cols[1].x_max > 7_000.0,
"right column should keep its true extent, got {}",
cols[1].x_max
);
}
#[test]
fn two_column_regression_guard() {
// Standard 2-column layout with clear gutter at center
+5 -311
View File
@@ -2,51 +2,11 @@
use crate::types::{ItemType, TextItem};
use lopdf::{Document, Object, ObjectId};
use std::collections::{HashMap, HashSet};
use std::collections::HashMap;
use super::fonts::{resolve_array, resolve_dict};
use super::get_number;
/// Upper bound on the number of form-field nodes visited during a single
/// `extract_form_fields` pass. A crafted PDF can chain thousands of distinct
/// `/Kids` fields to blow the stack even without an outright reference cycle,
/// so we cap total traversal work in addition to detecting cycles.
const MAX_FORM_FIELD_NODES: usize = 100_000;
/// Upper bound on `/Kids` recursion depth. Real AcroForm hierarchies are only
/// a few levels deep (fields → child fields → widgets); a crafted PDF can chain
/// tens of thousands of distinct fields into a linear `/Kids` list that would
/// overflow the stack via depth-first recursion long before the node budget is
/// reached. This depth cap bounds the stack independently of total node count.
const MAX_FORM_FIELD_DEPTH: usize = 100;
/// Traversal budget for the AcroForm field walk. Bounds both the number of
/// distinct nodes visited *and* the total number of `/Fields`/`/Kids` entries
/// examined.
///
/// Counting `visited` alone is not enough: invalid entries (non-references) and
/// duplicate references never grow `visited`, so an oversized array full of them
/// would iterate to completion no matter how large. Charging every examined
/// entry against the same budget makes it a real cap on traversal work.
pub(crate) struct FieldWalkBudget {
visited: HashSet<ObjectId>,
examined: usize,
}
impl FieldWalkBudget {
fn new() -> Self {
Self {
visited: HashSet::new(),
examined: 0,
}
}
/// True once the budget is spent; callers must stop iterating and recursing.
fn exhausted(&self) -> bool {
self.visited.len() >= MAX_FORM_FIELD_NODES || self.examined >= MAX_FORM_FIELD_NODES
}
}
pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> Vec<TextItem> {
let mut links = Vec::new();
@@ -186,12 +146,9 @@ pub(crate) fn extract_form_fields(
Err(_) => return items,
};
// Borrow the array rather than cloning it: a crafted `/Fields` can be huge,
// and cloning would pay an O(n) allocation/copy before the budget check
// below can stop the work.
let fields = match acroform.get(b"Fields") {
Ok(obj) => match resolve_array(doc, obj) {
Some(arr) => arr,
Some(arr) => arr.clone(),
None => return items,
},
Err(_) => return items,
@@ -201,19 +158,7 @@ pub(crate) fn extract_form_fields(
}
let annotation_pages = annotation_page_map(doc, page_map);
// Bound the walk so a crafted PDF cannot send us into unbounded recursion
// via a `/Kids` cycle, a deep chain, or an oversized array of invalid or
// duplicate entries.
let mut budget = FieldWalkBudget::new();
for field_obj in fields {
// Stop once the budget is spent so a `/Fields` array wider than the
// budget can't burn CPU iterating entries whose walk would no-op. Charge
// every entry (including invalid ones) against the budget.
if budget.exhausted() {
break;
}
budget.examined += 1;
for field_obj in &fields {
if let Ok(field_ref) = field_obj.as_reference() {
walk_form_fields(
doc,
@@ -223,8 +168,6 @@ pub(crate) fn extract_form_fields(
page_map,
&annotation_pages,
&mut items,
&mut budget,
0,
);
}
}
@@ -259,7 +202,6 @@ fn annotation_page_map(
}
/// Recursively walk the form field tree, extracting leaf field values.
#[allow(clippy::too_many_arguments)]
pub(crate) fn walk_form_fields(
doc: &Document,
field_id: ObjectId,
@@ -268,22 +210,7 @@ pub(crate) fn walk_form_fields(
page_map: &HashMap<ObjectId, u32>,
annotation_pages: &HashMap<ObjectId, u32>,
items: &mut Vec<TextItem>,
budget: &mut FieldWalkBudget,
depth: usize,
) {
// Guard against `/Kids` cycles and pathologically large field trees.
// Exceeding the depth cap means the chain is too deep to be a legitimate
// form (and would overflow the stack); an exhausted budget means the tree is
// too large. Both checks run *before* inserting so the visited set can never
// grow past the budget.
if depth > MAX_FORM_FIELD_DEPTH || budget.exhausted() {
return;
}
// Revisiting an object ID means we hit a `/Kids` cycle.
if !budget.visited.insert(field_id) {
return;
}
let field_dict = match doc.get_dictionary(field_id) {
Ok(d) => d,
Err(_) => return,
@@ -314,19 +241,9 @@ pub(crate) fn walk_form_fields(
// Check for /Kids — if present, recurse into children
if let Ok(kids_obj) = field_dict.get(b"Kids") {
// Iterate the borrowed array directly — cloning a crafted, oversized
// `/Kids` would allocate and copy every entry before the budget check
// below could stop the work.
if let Some(kids) = resolve_array(doc, kids_obj) {
for kid in kids {
// Stop once the budget is spent so a `/Kids` array wider than the
// budget can't burn CPU iterating entries whose walk would no-op.
// Charge every entry (including invalid/duplicate ones) against
// the budget so this is a true traversal-work cap.
if budget.exhausted() {
break;
}
budget.examined += 1;
let kids = kids.clone();
for kid in &kids {
if let Ok(kid_ref) = kid.as_reference() {
walk_form_fields(
doc,
@@ -336,8 +253,6 @@ pub(crate) fn walk_form_fields(
page_map,
annotation_pages,
items,
budget,
depth + 1,
);
}
}
@@ -496,225 +411,4 @@ mod tests {
assert_eq!(items[0].page, 2);
assert_eq!(items[0].text, "customer: Alice");
}
#[test]
fn kids_self_cycle_does_not_overflow_stack() {
// A crafted AcroForm field that lists itself in `/Kids` must not send
// the traversal into unbounded recursion.
let mut doc = Document::new();
let field_id = doc.new_object_id();
doc.set_object(
field_id,
dictionary! {
"FT" => "Tx",
"T" => Object::string_literal("loop"),
"Kids" => vec![Object::Reference(field_id)],
},
);
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"AcroForm" => dictionary! {
"Fields" => vec![Object::Reference(field_id)],
},
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let page_map = HashMap::new();
// Completes (rather than overflowing the stack) and yields no items.
let items = extract_form_fields(&doc, &page_map);
assert!(items.is_empty());
}
#[test]
fn kids_mutual_cycle_terminates() {
// Two fields that reference each other via `/Kids` form a cycle that
// must also terminate.
let mut doc = Document::new();
let field_a = doc.new_object_id();
let field_b = doc.new_object_id();
doc.set_object(
field_a,
dictionary! {
"T" => Object::string_literal("a"),
"Kids" => vec![Object::Reference(field_b)],
},
);
doc.set_object(
field_b,
dictionary! {
"T" => Object::string_literal("b"),
"Kids" => vec![Object::Reference(field_a)],
},
);
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"AcroForm" => dictionary! {
"Fields" => vec![Object::Reference(field_a)],
},
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let page_map = HashMap::new();
let items = extract_form_fields(&doc, &page_map);
assert!(items.is_empty());
}
#[test]
fn deep_acyclic_kids_chain_does_not_overflow_stack() {
// A long chain of *distinct* fields (no cycle) must also terminate:
// the visited set alone would still recurse to the chain length, so
// the depth cap is what prevents a stack overflow here.
let mut doc = Document::new();
let n = MAX_FORM_FIELD_DEPTH * 500;
let ids: Vec<ObjectId> = (0..=n).map(|_| doc.new_object_id()).collect();
for i in 0..n {
doc.set_object(
ids[i],
dictionary! {
"FT" => "Tx",
"Kids" => vec![Object::Reference(ids[i + 1])],
},
);
}
// Leaf carries a value; it sits far below the depth cap so it is never
// reached, proving traversal stops early rather than crashing.
doc.set_object(
ids[n],
dictionary! {
"FT" => "Tx",
"T" => Object::string_literal("leaf"),
"V" => Object::string_literal("x"),
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
},
);
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"AcroForm" => dictionary! {
"Fields" => vec![Object::Reference(ids[0])],
},
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let page_map = HashMap::new();
let items = extract_form_fields(&doc, &page_map);
assert!(items.is_empty());
}
#[test]
fn wide_tree_traversal_stops_at_node_budget() {
// A single field with a `/Kids` array wider than the node budget must
// stop traversal at the cap rather than growing `visited` (and the work)
// without bound. Each processed leaf emits one item, so the item count
// is bounded by the budget and reaches right up to it (a couple of
// slots go to the root and the boundary node charged against the cap).
let mut doc = Document::new();
let fanout = MAX_FORM_FIELD_NODES + 50;
let leaf_ids: Vec<ObjectId> = (0..fanout).map(|_| doc.new_object_id()).collect();
for &leaf in &leaf_ids {
doc.set_object(
leaf,
dictionary! {
"FT" => "Tx",
"V" => Object::string_literal("v"),
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
},
);
}
let kids: Vec<Object> = leaf_ids.iter().map(|&id| Object::Reference(id)).collect();
let root_id = doc.add_object(dictionary! {
"T" => Object::string_literal("root"),
"Kids" => kids,
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"AcroForm" => dictionary! {
"Fields" => vec![Object::Reference(root_id)],
},
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let page_map = HashMap::new();
let items = extract_form_fields(&doc, &page_map);
// Extraction stops at the budget: bounded above by the cap, and it gets
// right up to it (allowing a small delta for the root/boundary nodes
// charged against the budget).
assert!(items.len() <= MAX_FORM_FIELD_NODES);
assert!(items.len() >= MAX_FORM_FIELD_NODES - 3);
}
#[test]
fn wide_top_level_fields_stop_at_node_budget() {
// A top-level `/Fields` array wider than the budget must also stop at
// the cap: the item count is bounded by the budget and reaches right up
// to it.
let mut doc = Document::new();
let fanout = MAX_FORM_FIELD_NODES + 50;
let leaf_ids: Vec<ObjectId> = (0..fanout).map(|_| doc.new_object_id()).collect();
for &leaf in &leaf_ids {
doc.set_object(
leaf,
dictionary! {
"FT" => "Tx",
"V" => Object::string_literal("v"),
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
},
);
}
let fields: Vec<Object> = leaf_ids.iter().map(|&id| Object::Reference(id)).collect();
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"AcroForm" => dictionary! {
"Fields" => fields,
},
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let page_map = HashMap::new();
let items = extract_form_fields(&doc, &page_map);
assert!(items.len() <= MAX_FORM_FIELD_NODES);
assert!(items.len() >= MAX_FORM_FIELD_NODES - 3);
}
#[test]
fn duplicate_and_invalid_kids_entries_stop_at_budget() {
// Duplicate references and non-reference junk never grow `visited`, so
// without charging examined entries against the budget an oversized
// array of them would iterate to completion. The walk must still
// terminate and extract the single real leaf exactly once.
let mut doc = Document::new();
let leaf_id = doc.new_object_id();
doc.set_object(
leaf_id,
dictionary! {
"FT" => "Tx",
"V" => Object::string_literal("v"),
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
},
);
// A `/Kids` array far wider than the budget: half duplicate references
// to the same leaf, half invalid (null) entries.
let mut kids: Vec<Object> = Vec::new();
for i in 0..(MAX_FORM_FIELD_NODES * 2) {
if i % 2 == 0 {
kids.push(Object::Reference(leaf_id));
} else {
kids.push(Object::Null);
}
}
let root_id = doc.add_object(dictionary! {
"T" => Object::string_literal("root"),
"Kids" => kids,
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"AcroForm" => dictionary! {
"Fields" => vec![Object::Reference(root_id)],
},
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let page_map = HashMap::new();
let items = extract_form_fields(&doc, &page_map);
assert_eq!(items.len(), 1);
}
}
+3 -40
View File
@@ -4566,13 +4566,9 @@ pub fn glyph_to_char(name: &str) -> Option<char> {
}
}
// Try to parse uniXXXX format.
// Use `get` rather than a byte-length check + slice: `name` can contain
// non-ASCII bytes (e.g. U+FFFD from lossy UTF-8 decoding of an attacker
// controlled /Differences name), so byte index 7 may not be a char
// boundary and `&name[3..7]` would panic.
if let Some(hex) = name.strip_prefix("uni").and_then(|rest| rest.get(..4)) {
if let Ok(code) = u32::from_str_radix(hex, 16) {
// Try to parse uniXXXX format
if name.starts_with("uni") && name.len() >= 7 {
if let Ok(code) = u32::from_str_radix(&name[3..7], 16) {
// Strip PUA F000 offset: uniF0XX → U+00XX (Windows Symbol encoding convention)
let code = if (0xF000..=0xF0FF).contains(&code) {
code - 0xF000
@@ -4592,36 +4588,3 @@ pub fn glyph_to_char(name: &str) -> Option<char> {
None
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn uni_hex_parsing() {
assert_eq!(glyph_to_char("uni0041"), Some('A'));
assert_eq!(glyph_to_char("uni00e9"), Some('\u{00e9}'));
// PUA F0xx symbol-encoding offset is stripped.
assert_eq!(glyph_to_char("uniF041"), Some('A'));
}
#[test]
fn u_hex_parsing() {
assert_eq!(glyph_to_char("u0041"), Some('A'));
assert_eq!(glyph_to_char("u1F600"), Some('\u{1F600}'));
}
#[test]
fn non_ascii_uni_name_does_not_panic() {
// A crafted /Differences name like `/uni#80#80#80#80` decodes via
// from_utf8_lossy into "uni" followed by four U+FFFD replacements.
// Byte index 7 lands mid-character, so a naive `&name[3..7]` slice
// would panic. It must be handled gracefully instead.
let crafted = format!("uni{0}{0}{0}{0}", '\u{FFFD}');
assert_eq!(glyph_to_char(&crafted), None);
// Assorted non-ASCII bytes right after the "uni" prefix.
assert_eq!(glyph_to_char("uni\u{FFFD}bc"), None);
assert_eq!(glyph_to_char("uni\u{00e9}00"), None);
}
}
-74
View File
@@ -657,80 +657,6 @@ pub fn extract_pages_markdown<P: AsRef<Path>>(
extract_pages_markdown_mem(&buffer, pages)
}
// =========================================================================
// Structure-tree element extraction (tagged PDFs)
// =========================================================================
/// One structure-tree element reference from a tagged PDF, resolved to a
/// page and Marked Content ID.
///
/// Join `(page, mcid)` against [`TextItem::page`] / [`TextItem::mcid`] from
/// [`extract_text_with_positions`] to attach semantic roles (heading levels,
/// paragraphs, table cells, …) to extracted text.
#[derive(Debug, Clone)]
pub struct StructureElement {
/// 1-indexed page number (matches [`TextItem::page`]).
pub page: u32,
/// Marked Content ID from the page's content stream (matches
/// [`TextItem::mcid`]).
pub mcid: i64,
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
/// Custom tags are resolved through the document's `/RoleMap`; tags
/// with no standard mapping are returned verbatim.
pub role: String,
}
/// Extract structure-tree element references from a tagged PDF in memory.
///
/// Parses `/StructTreeRoot` (when present) and returns one entry per
/// marked-content reference, resolved to its 1-indexed page, MCID, and
/// structure type name. Returns an empty list when the PDF is not tagged.
///
/// Pass `Some(&[...])` with 1-indexed page numbers (matching
/// [`TextItem::page`]) to restrict output to those pages; pass `None` for
/// the whole document. Entries are sorted by `(page, mcid)`.
pub fn extract_structure_elements_mem(
buffer: &[u8],
pages: Option<&[u32]>,
) -> Result<Vec<StructureElement>, PdfError> {
validate_pdf_bytes(buffer)?;
let (doc, _page_count) = load_document_from_mem(buffer)?;
let Some(tree) = structure_tree::StructTree::from_doc(&doc) else {
return Ok(Vec::new());
};
let page_ids = doc.get_pages();
let roles = tree.mcid_to_roles(&page_ids);
let page_filter: Option<HashSet<u32>> = pages.map(|p| p.iter().copied().collect());
let mut elements: Vec<StructureElement> = roles
.into_iter()
.filter(|(page, _)| page_filter.as_ref().is_none_or(|f| f.contains(page)))
.flat_map(|(page, mcids)| {
mcids.into_iter().map(move |(mcid, role)| StructureElement {
page,
mcid,
role: role.name().to_string(),
})
})
.collect();
elements.sort_unstable_by_key(|e| (e.page, e.mcid));
Ok(elements)
}
/// Path-based wrapper for [`extract_structure_elements_mem`].
///
/// Reads the PDF from disk and extracts structure-tree element references.
/// Pass `None` for `pages` to return the whole document, or `Some(&[...])`
/// to restrict to specific 1-indexed pages.
pub fn extract_structure_elements<P: AsRef<Path>>(
path: P,
pages: Option<&[u32]>,
) -> Result<Vec<StructureElement>, PdfError> {
validate_pdf_file(&path)?;
let buffer = std::fs::read(path.as_ref())?;
extract_structure_elements_mem(&buffer, pages)
}
// =========================================================================
// Region-based text extraction (for hybrid OCR pipelines)
// =========================================================================
-192
View File
@@ -171,74 +171,6 @@ pub(crate) fn is_toc_marker_heading(text: &str) -> bool {
/// equation and absent from name-plus-number headings. A bare trailing colon
/// is NOT a fragment signal either: real headings frequently end with colons
/// ("Procedure:", "Steps for Using the Microscope:").
/// True when the line opens with a section number ("3.", "2.1.4", "IV)").
///
/// Mirrors the acceptance of `heading::parse_numbering` rather than the
/// stricter `convert::starts_with_section_number`, which deliberately
/// requires two components because it bypasses isolation checks. Here a
/// single "1." counts: numbering is independent evidence of a heading, and
/// `heading.rs` applies its numbered-prefix allowance *after* consulting
/// `is_heading_fragment`, so without this exemption a numbered
/// sentence-case heading would be vetoed before that allowance can run.
fn starts_with_numbering_prefix(t: &str) -> bool {
let Some(first) = t.split_whitespace().next() else {
return false;
};
let has_delimiter = first.ends_with(['.', ')', ':']);
let token = first.trim_end_matches(['.', ')', ':']);
if token.is_empty() {
return false;
}
let parts: Vec<&str> = token.split('.').collect();
let decimal = parts
.iter()
.all(|p| !p.is_empty() && p.len() <= 3 && p.chars().all(|c| c.is_ascii_digit()));
if decimal {
// "1." / "2.1." carry a delimiter; "2.3 Title" is written without
// one, so a multi-component number is accepted bare. A bare single
// number ("3 apples") is not — that is ordinary prose.
return has_delimiter || parts.len() >= 2;
}
// Roman numerals go through the heading parser's own grammar so the two
// agree: uppercase I/V/X/L/C only, at most 8 characters. A looser rule
// here would exempt markers the parser rejects — "iv)" or "d)" from an
// alphabetical list — letting an ordinary list item bypass the veto and
// reach heading promotion.
//
// A delimiter is also required: a bare leading "I" is the pronoun far
// more often than a section number.
has_delimiter && crate::markdown::heading::roman_value(token).is_some()
}
/// True when the line reads as a title rather than a sentence: every
/// content word (ignoring minor words) starts uppercase. Used to spare real
/// headings from the dangling-verb veto — "Bond Yields" is a section title,
/// "the method yields" is a stranded clause, and only the casing tells them
/// apart.
fn looks_title_case(t: &str) -> bool {
const MINOR: &[&str] = &[
"a", "an", "the", "of", "and", "or", "for", "to", "in", "on", "at", "by", "with", "from",
"as", "is", "are", "that", "than", "into",
];
let mut content = 0usize;
let mut capitalized = 0usize;
for w in t.split_whitespace() {
let cleaned: String = w.chars().filter(|c| c.is_alphabetic()).collect();
if cleaned.is_empty() {
continue;
}
if MINOR.contains(&cleaned.to_lowercase().as_str()) {
continue;
}
content += 1;
if cleaned.chars().next().is_some_and(char::is_uppercase) {
capitalized += 1;
}
}
// A single content word ("Yields") is a title by default.
content == 0 || capitalized == content
}
pub(crate) fn is_heading_fragment(text: &str) -> bool {
let t = text.trim_end();
@@ -312,133 +244,9 @@ pub(crate) fn is_heading_fragment(text: &str) -> bool {
if t.ends_with(':') && t.split_whitespace().any(is_equation_number) {
return true;
}
// Dangling clause: a stranded sentence lead-in ends on a relational
// verb with no terminal punctuation — "Note that the exact error equals"
// left ahead of its formula when a phantom table dissolved.
//
// Gated on the line reading as prose rather than a title. Case is the
// discriminator the trailing word alone cannot provide: a heading is
// title case ("Bond Yields", "The Method Yields") while a stranded
// lead-in is sentence case ("the method yields"). Without this gate the
// veto eats real headings — "Bond Yields", "Crop Yields" and any wrapped
// title-case heading the preprocessor failed to merge.
if !t.ends_with(['.', '!', '?', ':', ';', ')', ']'])
&& !looks_title_case(t)
&& !starts_with_numbering_prefix(t)
{
if let Some(last) = t.split_whitespace().next_back() {
let word: String = last
.trim_matches(|c: char| !c.is_alphanumeric())
.to_lowercase();
// Relational verbs only, and only those with no common noun
// sense. "yields" was dropped for exactly that reason: "Bond
// Yields" is a real section title. Function words, copulas and
// auxiliaries were measured and rejected outright — a heading
// that wraps across lines ends on those, and suppressing them
// destroyed real IRS Publication 17 headings.
const DANGLING_TAIL: &[&str] =
&["equals", "denotes", "implies", "satisfies", "signifies"];
if DANGLING_TAIL.contains(&word.as_str()) {
return true;
}
}
}
false
}
#[cfg(test)]
mod fragment_heading_tests {
use super::is_heading_fragment;
#[test]
fn dangling_tail_marks_stranded_clause() {
// opendataloader 01030000000144: left behind when a phantom table
// dissolved, ahead of its formula on the next line.
assert!(is_heading_fragment("Note that the exact error equals"));
assert!(is_heading_fragment("The remainder term satisfies"));
assert!(is_heading_fragment("we conclude that the sum equals"));
}
#[test]
fn real_headings_survive() {
assert!(!is_heading_fragment("Introduction"));
assert!(!is_heading_fragment("Error Analysis"));
assert!(!is_heading_fragment("Materials and Methods"));
assert!(!is_heading_fragment("Results"));
assert!(!is_heading_fragment("3.2 Richardson Extrapolation"));
assert!(!is_heading_fragment("Discussion and Conclusions"));
// Terminal punctuation means the clause is complete.
assert!(!is_heading_fragment("What is a Derivative?"));
assert!(!is_heading_fragment("Procedure:"));
assert!(!is_heading_fragment("Note that this is important."));
}
#[test]
fn title_case_headings_ending_in_a_verb_survive() {
// "yields" is also a plural noun; these are real section titles.
assert!(!is_heading_fragment("Bond Yields"));
assert!(!is_heading_fragment("Crop Yields"));
assert!(!is_heading_fragment("Dividend Yields"));
assert!(!is_heading_fragment("Yields"));
// A wrapped title-case heading whose first line ends on a listed
// verb must survive even if the preprocessor failed to merge it.
assert!(!is_heading_fragment("The Theorem Implies"));
assert!(!is_heading_fragment("What This Denotes"));
}
#[test]
fn numbered_sentence_case_headings_survive() {
// heading.rs consults is_heading_fragment BEFORE applying its
// numbered-prefix allowance, so the veto must not pre-empt it.
assert!(!is_heading_fragment("1. What the model implies"));
assert!(!is_heading_fragment("2.3 How the estimator satisfies"));
assert!(!is_heading_fragment("IV) What this denotes"));
// Without numbering the same wording is still a stranded clause.
assert!(is_heading_fragment("What the model implies"));
// A bare leading number or pronoun is prose, not numbering.
assert!(is_heading_fragment("3 apples and what that implies"));
assert!(is_heading_fragment("I think the model implies"));
// Markers heading::parse_numbering rejects must not be exempted
// either, or an ordinary list item bypasses the veto: lowercase
// roman, alphabetical markers, and over-long tokens.
assert!(is_heading_fragment("iv) the estimator satisfies"));
assert!(is_heading_fragment("d) the value implies"));
// Unsupported character (M is outside the parser's I/V/X/L/C set).
assert!(is_heading_fragment("MMMM. the value implies"));
// Over-long token: nine valid characters, so this exercises the
// 8-character bound rather than the character set.
assert!(is_heading_fragment("IIIIIIIII. the value implies"));
// Eight is still within the bound and stays exempt.
assert!(!is_heading_fragment("IIIIIIII. What this implies"));
// Uppercase roman within the parser's grammar is still exempt.
assert!(!is_heading_fragment("IV. What this denotes"));
assert!(!is_heading_fragment("XII) What this implies"));
}
#[test]
fn wrapped_headings_are_not_fragments() {
// A heading that wraps across lines ends on a function word. These
// are real headings from IRS Publication 17 and must survive.
assert!(!is_heading_fragment("Casualty and"));
assert!(!is_heading_fragment("Rule 10. You Must Be at"));
assert!(!is_heading_fragment("Higher Standard Deduction for"));
assert!(!is_heading_fragment("Qualifying Child of"));
assert!(!is_heading_fragment("When Can I Withdraw or"));
// Copulas and auxiliaries also end real wrapped headings.
assert!(!is_heading_fragment("Rule 15. Your AGI Must Be"));
assert!(!is_heading_fragment("What Medical Expenses Are"));
assert!(!is_heading_fragment("Rule 13. You Must Have"));
assert!(!is_heading_fragment("When Can a Roth IRA Be"));
}
#[test]
fn dangling_check_is_case_insensitive() {
// All-caps is not sentence case, so the veto must not fire there.
assert!(!is_heading_fragment("THE REMAINDER EQUALS"));
}
}
/// Compute the Y-gap threshold for paragraph break detection.
///
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
+1 -3
View File
@@ -127,9 +127,7 @@ fn visual_style(line: &TextLine) -> Option<VisualStyle> {
})
}
/// Shared with `analysis::starts_with_numbering_prefix` so the veto
/// exemption and the heading parser agree on what a roman numeral is.
pub(super) fn roman_value(token: &str) -> Option<u32> {
fn roman_value(token: &str) -> Option<u32> {
if token.is_empty() || token.len() > 8 {
return None;
}
-89
View File
@@ -271,12 +271,6 @@ pub struct PyTextItem {
pub is_strikeout: bool,
#[pyo3(get)]
pub item_type: String,
/// Marked Content ID from the content stream's BDC/BMC operator, None
/// when the text is not part of marked content. Join with the
/// (page, mcid) pairs from extract_structure_elements to attach
/// structure-tree roles (headings, paragraphs, ...) in tagged PDFs.
#[pyo3(get)]
pub mcid: Option<i64>,
}
#[pymethods]
@@ -292,32 +286,6 @@ impl PyTextItem {
}
}
/// One structure-tree element reference from a tagged PDF.
#[pyclass(name = "StructureElement")]
#[derive(Clone)]
pub struct PyStructureElement {
/// 1-indexed page number (matches TextItem.page).
#[pyo3(get)]
pub page: u32,
/// Marked Content ID from the page's content stream (matches
/// TextItem.mcid).
#[pyo3(get)]
pub mcid: i64,
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", ...).
#[pyo3(get)]
pub role: String,
}
#[pymethods]
impl PyStructureElement {
fn __repr__(&self) -> String {
format!(
"StructureElement(page={}, mcid={}, role='{}')",
self.page, self.mcid, self.role
)
}
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
@@ -388,18 +356,6 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
is_underline: item.is_underline,
is_strikeout: item.is_strikeout,
item_type: item_type_str(&item.item_type),
mcid: item.mcid,
})
.collect()
}
fn convert_structure_elements(elements: Vec<crate::StructureElement>) -> Vec<PyStructureElement> {
elements
.into_iter()
.map(|e| PyStructureElement {
page: e.page,
mcid: e.mcid,
role: e.role,
})
.collect()
}
@@ -657,48 +613,6 @@ fn extract_pages_markdown_bytes(
Ok(to_py_pages_result(result))
}
/// Extract structure-tree element references from a tagged PDF file.
///
/// Parses the document's structure tree (when present) and returns one
/// entry per marked-content reference, resolved to its 1-indexed page,
/// MCID, and structure type name ("H1".."H6", "P", "Table", ...). Returns
/// an empty list when the PDF is not tagged.
///
/// Join (page, mcid) against the page/mcid attributes from
/// [`extract_text_with_positions`] to attach heading levels and other
/// semantic roles to extracted text.
///
/// Args:
/// path: Path to the PDF file.
/// pages: Optional list of 1-indexed pages (matching TextItem.page).
/// When None (default), the whole document is returned.
///
/// Returns:
/// List of StructureElement sorted by (page, mcid).
#[pyfunction]
#[pyo3(signature = (path, pages=None))]
fn extract_structure_elements(
path: &str,
pages: Option<Vec<u32>>,
) -> PyResult<Vec<PyStructureElement>> {
let elements = crate::extract_structure_elements(path, pages.as_deref()).map_err(to_py_err)?;
Ok(convert_structure_elements(elements))
}
/// Extract structure-tree element references from tagged PDF bytes.
///
/// See [`extract_structure_elements`] for details.
#[pyfunction]
#[pyo3(signature = (data, pages=None))]
fn extract_structure_elements_bytes(
data: &[u8],
pages: Option<Vec<u32>>,
) -> PyResult<Vec<PyStructureElement>> {
let elements =
crate::extract_structure_elements_mem(data, pages.as_deref()).map_err(to_py_err)?;
Ok(convert_structure_elements(elements))
}
/// Python module definition.
#[pymodule]
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
@@ -706,7 +620,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<PyPageOcrReasons>()?;
m.add_class::<PyPdfClassification>()?;
m.add_class::<PyTextItem>()?;
m.add_class::<PyStructureElement>()?;
m.add_class::<PyRegionText>()?;
m.add_class::<PyPageRegionTexts>()?;
m.add_class::<PyPageMarkdown>()?;
@@ -721,8 +634,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_function(wrap_pyfunction!(extract_text_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_with_positions, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_structure_elements, m)?)?;
m.add_function(wrap_pyfunction!(extract_structure_elements_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
+38 -819
View File
File diff suppressed because it is too large Load Diff
+10 -308
View File
@@ -435,72 +435,6 @@ fn revised_table_cell_indices(
.collect()
}
/// Index of candidate "body" items (larger-font attachment targets) sorted by
/// Y, so script-attachment checks scan a narrow Y window instead of the whole
/// page per candidate.
struct ScriptBodyIndex<'a> {
/// (y, item), sorted ascending by y
by_y: Vec<(f32, &'a TextItem)>,
/// widest vertical attachment window any body item can produce
max_window: f32,
}
impl<'a> ScriptBodyIndex<'a> {
fn new(items: &'a [TextItem]) -> Self {
// Smallest table-candidate font is 6pt, so any possible attachment
// target is at least 6 x 1.2 pt.
let mut by_y: Vec<(f32, &TextItem)> = items
.iter()
.filter(|i| i.font_size >= 6.0 * 1.2)
.map(|i| (i.y, i))
.collect();
by_y.sort_by(|a, b| a.0.total_cmp(&b.0));
let max_window = by_y
.iter()
.map(|(_, i)| i.font_size * 0.8)
.fold(0.0f32, f32::max);
Self { by_y, max_window }
}
/// True when a small-font item is horizontally attached to a larger-font
/// item at a script baseline offset — a sub/superscript in running text
/// or math (equation subscripts, footnote markers). Script attachments
/// are not table cells; without this filter, display equations with
/// sub/superscripts form phantom small-font table regions (e.g. TeX
/// papers where log subscripts cluster with footnote lines into a fake
/// 3-column table). A genuine baseline offset is required so same-line
/// table neighbours (a small cell beside a larger label cell) are never
/// classified as scripts.
///
/// `min_anchor_size` additionally constrains what counts as an
/// attachment target: the small-font pass accepts any sufficiently
/// larger item (0.0), while the body-font pass requires a heading-sized
/// anchor so a body-size table cell beside a slightly larger label with
/// baseline jitter is never treated as a script.
fn is_script_attachment(&self, small: &TextItem, min_anchor_size: f32) -> bool {
let attach_gap = small.font_size.max(4.0) * 0.6;
let lo = self
.by_y
.partition_point(|(y, _)| *y < small.y - self.max_window);
self.by_y[lo..]
.iter()
.take_while(|(y, _)| *y <= small.y + self.max_window)
.any(|(_, body)| {
let dy = (small.y - body.y).abs();
body.font_size >= small.font_size * 1.2
&& body.font_size >= min_anchor_size
&& dy > body.font_size * 0.05
&& dy <= body.font_size * 0.8
&& {
let gap_after_body = small.x - (body.x + body.width);
let gap_before_body = body.x - (small.x + small.width);
(-attach_gap..=attach_gap).contains(&gap_after_body)
|| (-attach_gap..=attach_gap).contains(&gap_before_body)
}
})
}
}
/// Detect tables in a set of text items from a single page
pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bool) -> Vec<Table> {
detect_tables_with_page_width(items, base_font_size, skip_body_font, content_width(items))
@@ -549,27 +483,6 @@ pub(crate) fn detect_tables_with_page_width(
// === Pass 1: Small-font tables (existing behavior) ===
let table_font_threshold = base_font_size * 0.90;
// Mark sub/superscript attachments once per pass. They stay candidates —
// the masks only remove them from region qualification and column/row
// geometry.
//
// The two passes need different anchor thresholds. In the small-font pass
// any sufficiently larger neighbour is a plausible base for a script. In
// the body-font pass the candidates are themselves body-sized
// (0.85..1.05x), so a merely "slightly larger" neighbour is usually a bold
// label or an adjacent column header, not the base of a superscript —
// treating it as one would strip real cells out of the geometry and lose
// the table. Requiring a heading-sized anchor (>= 1.15x base) keeps the
// body pass to genuine scripts hanging off headings.
let script_index = ScriptBodyIndex::new(items);
let script_flags: Vec<bool> = items
.iter()
.map(|item| script_index.is_script_attachment(item, 0.0))
.collect();
let body_script_flags: Vec<bool> = items
.iter()
.map(|item| script_index.is_script_attachment(item, base_font_size * 1.15))
.collect();
let table_candidates: Vec<(usize, &TextItem)> = items
.iter()
.enumerate()
@@ -581,14 +494,7 @@ pub(crate) fn detect_tables_with_page_width(
.collect();
if table_candidates.len() >= 6 {
// Qualify regions from non-script items: a cluster of sub/superscripts
// must not, on its own, mark out a table region.
let region_evidence: Vec<(usize, &TextItem)> = table_candidates
.iter()
.filter(|(idx, _)| !script_flags[*idx])
.cloned()
.collect();
let regions = find_table_regions(&region_evidence);
let regions = find_table_regions(&table_candidates);
for (y_min, y_max) in regions {
let region_items: Vec<(usize, &TextItem)> = table_candidates
@@ -602,9 +508,7 @@ pub(crate) fn detect_tables_with_page_width(
}
if let Some(mut table) =
detect_table_in_region(&region_items, TableDetectionMode::SmallFont, &|i| {
script_flags[i]
})
detect_table_in_region(&region_items, TableDetectionMode::SmallFont)
{
// Try to recover body-font header row above the small-font table
recover_header_row(&mut table, items, table_font_threshold);
@@ -649,20 +553,8 @@ pub(crate) fn detect_tables_with_page_width(
body_font_low,
body_font_high,
);
// Scripts are NOT filtered out of the candidate set here, mirroring
// the small-font pass: they must stay eligible for cell assignment so
// a sub/superscript that belongs inside a table cell keeps its text.
// The heading-anchored `body_script_flags` mask removes them from
// geometry only.
if body_candidates.len() >= 6 {
// Same reasoning as the small-font pass: scripts do not qualify
// regions, but remain available for cell assignment within one.
let region_evidence: Vec<(usize, &TextItem)> = body_candidates
.iter()
.filter(|(idx, _)| !body_script_flags[*idx])
.cloned()
.collect();
let regions = find_table_regions_strict(&region_evidence);
let regions = find_table_regions_strict(&body_candidates);
log::debug!("body-font: {} strict regions found", regions.len());
for (y_min, y_max, _x_min, _x_max) in &regions {
@@ -688,9 +580,7 @@ pub(crate) fn detect_tables_with_page_width(
}
if let Some(table) =
detect_table_in_region(&region_items, TableDetectionMode::BodyFont, &|i| {
body_script_flags[i]
})
detect_table_in_region(&region_items, TableDetectionMode::BodyFont)
{
tables.push(table);
}
@@ -918,30 +808,10 @@ fn find_table_regions_strict(items: &[(usize, &TextItem)]) -> Vec<(f32, f32, f32
regions
}
/// Detect a table within a specific region.
///
/// `is_script` marks items that are sub/superscript attachments. Those are
/// excluded from the *geometry* — they must not be able to create a column,
/// which is how equation subscript clusters used to fabricate phantom grids —
/// but they remain eligible for cell assignment, so legitimate cell content
/// (exponents in an engineering-notation table, footnote markers) stays in
/// the cell it belongs to instead of leaking out into the reading order.
fn detect_table_in_region(
items: &[(usize, &TextItem)],
mode: TableDetectionMode,
is_script: &dyn Fn(usize) -> bool,
) -> Option<Table> {
// Column geometry from non-script items only.
let geometry_items: Vec<(usize, &TextItem)> = items
.iter()
.filter(|(idx, _)| !is_script(*idx))
.cloned()
.collect();
// A region that is *entirely* scripts has no table structure at all.
if geometry_items.is_empty() {
return None;
}
let columns = find_column_boundaries(&geometry_items, mode);
/// Detect a table within a specific region
fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode) -> Option<Table> {
// Find column boundaries
let columns = find_column_boundaries(items, mode);
let min_cols = 2;
if columns.len() < min_cols || columns.len() > 25 {
log::debug!(
@@ -952,8 +822,8 @@ fn detect_table_in_region(
return None;
}
// Find row boundaries (geometry items only, same reasoning)
let rows = find_row_boundaries(&geometry_items);
// Find row boundaries
let rows = find_row_boundaries(items);
let min_rows = 2;
if rows.len() < min_rows {
log::debug!(
@@ -972,11 +842,6 @@ fn detect_table_in_region(
);
// Verify this looks like a table: multiple items should align to columns
// Validate against ALL items, including scripts. Columns are derived from
// non-script geometry so scripts cannot *create* a column, but excluding
// them from validation too would let a region manufacture alignment: drop
// the awkward items and whatever remains looks like a tidy grid. Block
// diagrams did exactly that. Everything in the region must fit.
let col_alignment = check_column_alignment(items, &columns, mode);
let min_alignment = match mode {
TableDetectionMode::SmallFont => 0.5,
@@ -1047,29 +912,6 @@ fn detect_table_in_region(
cells.push(row_cells);
}
// Validation 0 (small-font pass only): reject tiny all-numeric
// fragments. A <=2-row grid whose every cell is a bare 1-2 digit number
// carries no tabular information — in practice these are
// exponent/subscript clusters from display math that happen to align.
// Body-font tables are not subject to this veto: their cells cannot be
// script glyphs.
if matches!(mode, TableDetectionMode::SmallFont) {
let nonempty_cells: Vec<&String> =
cells.iter().flatten().filter(|c| !c.is_empty()).collect();
if rows.len() <= 2
&& !nonempty_cells.is_empty()
&& nonempty_cells
.iter()
.all(|c| c.len() <= 2 && c.chars().all(|ch| ch.is_ascii_digit()))
{
log::debug!(
" validation 0 fail: tiny all-numeric fragment ({} cells)",
nonempty_cells.len()
);
return None;
}
}
// Validation 1: some rows should have content in first column.
// Use a lower threshold (25%) for tables with wrapped cells where
// continuation lines leave the first column empty.
@@ -2135,146 +1977,6 @@ fn try_add_label_column(
#[cfg(test)]
mod tests {
fn make_item(text: &str, x: f32, y: f32, font_size: f32, width: f32) -> TextItem {
TextItem {
text: text.to_string(),
x,
y,
width,
height: font_size,
font: "TestFont".to_string(),
font_size,
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
}
#[test]
fn script_attachment_detects_subscript_after_body_text() {
let body = make_item("log", 100.0, 500.0, 10.0, 15.0);
let sub = make_item("10", 115.5, 497.0, 7.0, 7.0);
let items = vec![body, sub.clone()];
assert!(ScriptBodyIndex::new(&items).is_script_attachment(&sub, 0.0));
}
#[test]
fn script_attachment_detects_superscript_footnote_marker() {
let body = make_item("Hartley", 200.0, 500.0, 10.0, 35.0);
let sup = make_item("2", 235.8, 504.0, 6.6, 3.5);
let items = vec![body, sup.clone()];
assert!(ScriptBodyIndex::new(&items).is_script_attachment(&sup, 0.0));
}
#[test]
fn script_attachment_ignores_small_cell_far_from_body_text() {
let body = make_item("Revenue", 100.0, 500.0, 10.0, 40.0);
let cell = make_item("1,234", 180.0, 500.0, 7.0, 20.0);
let items = vec![body, cell.clone()];
assert!(!ScriptBodyIndex::new(&items).is_script_attachment(&cell, 0.0));
}
#[test]
fn body_pass_anchor_spares_cells_beside_slightly_larger_labels() {
// A body-font table cell (10pt) sitting beside a slightly larger,
// NON-heading label (12.5pt) with a little baseline jitter. The
// small-font pass treats any larger neighbour as a possible script
// base, but the body pass must not: at body sizes a slightly larger
// neighbour is a bold label or column header, and flagging the cell
// would strip it out of the table geometry and lose the table.
// Cell at the low end of the body band (0.85x base) beside a 10.5pt
// label. 10.5 clears the inherent 1.2x-of-cell rule (10.2) but falls
// below the body pass's heading anchor (11.5), which is exactly the
// band where the two masks must disagree.
let label = make_item("Revenue", 100.0, 500.0, 10.5, 40.0);
let cell = make_item("1,234", 141.0, 496.5, 8.5, 22.0);
let items = vec![label, cell.clone()];
let index = ScriptBodyIndex::new(&items);
let base = 10.0;
assert!(
index.is_script_attachment(&cell, 0.0),
"small-font pass anchor should still see this as an attachment"
);
assert!(
!index.is_script_attachment(&cell, base * 1.15),
"body pass must not treat a cell beside a slightly larger label \
as a script that removes real cells from the geometry"
);
// A genuine heading-sized anchor still qualifies in the body pass.
let heading = make_item("Section", 100.0, 500.0, 20.0, 60.0);
let sup = make_item("3", 161.0, 508.0, 10.0, 5.0);
let h_items = vec![heading, sup.clone()];
assert!(
ScriptBodyIndex::new(&h_items).is_script_attachment(&sup, base * 1.15),
"script hanging off a heading must still be excluded in the body pass"
);
}
#[test]
fn script_attachment_ignores_same_baseline_neighbor_cell() {
// A small cell beside a larger label on the SAME baseline is a table
// layout, not a subscript — a genuine baseline offset is required.
let label = make_item("Total", 100.0, 500.0, 10.0, 25.0);
let cell = make_item("42", 127.0, 500.0, 7.5, 9.0);
let items = vec![label, cell.clone()];
assert!(!ScriptBodyIndex::new(&items).is_script_attachment(&cell, 0.0));
}
#[test]
fn script_attachment_ignores_neighbor_on_different_line() {
let body = make_item("Header", 100.0, 500.0, 10.0, 30.0);
let cell = make_item("42", 131.0, 486.0, 7.0, 10.0);
let items = vec![body, cell.clone()];
assert!(!ScriptBodyIndex::new(&items).is_script_attachment(&cell, 0.0));
}
/// Equation-subscript + footnote layout from Shannon entropy.pdf page 1,
/// with real coordinates. Without the larger-font anchors the small items
/// alone DO form a phantom table — proving the layout reaches detection —
/// and adding the anchors must suppress it.
fn shannon_page1_small_items() -> Vec<TextItem> {
vec![
make_item("2", 267.4, 133.9, 7.4, 3.7),
make_item("10", 306.2, 133.9, 7.4, 7.4),
make_item("10", 342.7, 133.9, 7.4, 7.4),
make_item("10", 325.0, 118.9, 7.4, 7.4),
make_item("Bell System Technical Journal,", 295.7, 101.9, 8.0, 95.0),
make_item(
"April 1924, p. 324; Certain Topics in",
396.7,
101.9,
8.0,
130.0,
),
make_item("v. 47, April 1928, p. 617.", 250.9, 92.5, 8.0, 90.0),
make_item("Bell System Technical Journal,", 264.2, 82.6, 8.0, 95.0),
make_item("July 1928, p. 535.", 364.3, 82.6, 8.0, 65.0),
]
}
#[test]
fn equation_scripts_do_not_form_phantom_table() {
let bare = shannon_page1_small_items();
assert!(
!detect_tables(&bare, 10.0, false).is_empty(),
"test layout must form a phantom table when the filter cannot fire"
);
let mut items = shannon_page1_small_items();
items.push(make_item("log", 253.0, 137.0, 10.0, 13.5));
items.push(make_item("log", 291.5, 137.0, 10.0, 13.5));
items.push(make_item("log", 328.0, 137.0, 10.0, 13.5));
items.push(make_item("log", 310.3, 122.0, 10.0, 13.5));
let tables = detect_tables(&items, 10.0, false);
assert!(
tables.is_empty(),
"equation scripts + footnotes must not become a table: {tables:?}"
);
}
use super::*;
use crate::types::ItemType;
+1 -19
View File
@@ -594,7 +594,7 @@ fn hex_to_unicode_string(hex: &str) -> Option<String> {
let bytes: Option<Vec<u8>> = (0..hex.len())
.step_by(2)
.map(|i| u8::from_str_radix(hex.get(i..i + 2)?, 16).ok())
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
.collect();
let bytes = bytes?;
@@ -2606,24 +2606,6 @@ endcmap
assert_eq!(cmap.lookup(0x0025), Some("B".to_string()));
}
#[test]
fn test_hex_to_unicode_non_ascii_no_panic() {
// A destination containing a multi-byte char makes the byte length even
// while a byte offset can land inside a char. Slicing must not panic;
// it should be rejected gracefully.
assert_eq!(hex_to_unicode_string("XéY"), None);
assert_eq!(hex_to_unicode_string("\u{fffd}0"), None);
}
#[test]
fn test_parse_bfchar_non_ascii_destination_no_panic() {
// Crafted /ToUnicode CMap: a non-hex, non-ASCII destination previously
// triggered a char-boundary panic in hex_to_unicode_string.
let cmap_content = "beginbfchar <0041> <XéY> endbfchar";
// Must not panic; the malformed entry is simply skipped.
let _ = ToUnicodeCMap::parse(cmap_content.as_bytes());
}
#[test]
fn test_parse_bfchar_1byte() {
// This is the pattern that caused the CJK bug: codespace is <0000><FFFF>
-71
View File
@@ -1429,77 +1429,6 @@ fn test_firecrawl_tagged_pdf_struct_tree() {
assert_eq!(fence_count % 2, 0, "Code fences should be balanced");
}
#[test]
fn test_tagged_pdf_text_items_carry_mcid() {
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
assert!(
items.iter().any(|i| i.mcid.is_some()),
"Tagged PDF text items should carry Marked Content IDs"
);
}
#[test]
fn test_extract_structure_elements_tagged_pdf() {
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
assert!(!elements.is_empty(), "Tagged PDF should yield elements");
assert!(
elements.iter().any(|e| e.role == "H1"),
"Should surface H1 heading roles"
);
assert!(
elements.iter().all(|e| !e.role.is_empty()),
"Every element should carry a role name"
);
// Sorted by (page, mcid) for deterministic output
assert!(
elements
.windows(2)
.all(|w| (w[0].page, w[0].mcid) <= (w[1].page, w[1].mcid)),
"Elements should be sorted by (page, mcid)"
);
// The advertised join: (page, mcid) pairs must line up with the
// mcid-carrying TextItems from positioned extraction, and joining the
// H1 entries must recover non-empty heading text.
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
let h1_refs: std::collections::HashSet<(u32, i64)> = elements
.iter()
.filter(|e| e.role == "H1")
.map(|e| (e.page, e.mcid))
.collect();
let h1_text: String = items
.iter()
.filter(|i| i.mcid.is_some_and(|mcid| h1_refs.contains(&(i.page, mcid))))
.map(|i| i.text.as_str())
.collect();
assert!(
!h1_text.trim().is_empty(),
"Joining H1 structure elements to text items should recover heading text"
);
// Page filter is 1-indexed (matching TextItem.page) and equals the
// corresponding subset of the full document result.
let page1 = pdf_inspector::extract_structure_elements_mem(&buf, Some(&[1])).unwrap();
assert!(!page1.is_empty(), "Page 1 should have elements");
assert!(page1.iter().all(|e| e.page == 1));
let full_page1_count = elements.iter().filter(|e| e.page == 1).count();
assert_eq!(page1.len(), full_page1_count);
}
#[test]
fn test_extract_structure_elements_untagged_pdf_empty() {
let buf = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
assert!(
elements.is_empty(),
"Untagged PDF should yield no structure elements, got {:?}",
elements
);
}
#[test]
fn test_identity_h_no_tounicode_suppresses_garbage() {
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
-73
View File
@@ -203,79 +203,6 @@ class TestExtractTextWithPositions:
assert len(items) > 0
assert all(item.page == 1 for item in items)
def test_mcid(self):
# Untagged fixture: mcid is None or int, never anything else
items = pdf_inspector.extract_text_with_positions(
fixture_path("thermo-freon12.pdf")
)
assert all(item.mcid is None or isinstance(item.mcid, int) for item in items)
# Tagged fixture: marked content carries MCIDs
tagged = pdf_inspector.extract_text_with_positions(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert any(item.mcid is not None for item in tagged)
# ---------------------------------------------------------------------------
# extract_structure_elements / extract_structure_elements_bytes
# ---------------------------------------------------------------------------
class TestExtractStructureElements:
def test_tagged_file(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert len(elements) > 0
assert all(isinstance(e.page, int) for e in elements)
assert all(isinstance(e.mcid, int) for e in elements)
assert all(isinstance(e.role, str) and len(e.role) > 0 for e in elements)
assert any(e.role == "H1" for e in elements)
def test_join_with_text_items(self):
# (page, mcid) joins against extract_text_with_positions to recover
# heading text
path = fixture_path("firecrawl_docs_tagged.pdf")
elements = pdf_inspector.extract_structure_elements(path)
items = pdf_inspector.extract_text_with_positions(path)
h1_refs = {(e.page, e.mcid) for e in elements if e.role == "H1"}
h1_text = "".join(
item.text
for item in items
if item.mcid is not None and (item.page, item.mcid) in h1_refs
)
assert len(h1_text.strip()) > 0
def test_with_pages(self):
# pages filter is 1-indexed, matching TextItem.page
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf"), pages=[1]
)
assert len(elements) > 0
assert all(e.page == 1 for e in elements)
def test_bytes(self):
data = fixture_bytes("firecrawl_docs_tagged.pdf")
elements = pdf_inspector.extract_structure_elements_bytes(data)
assert len(elements) > 0
assert any(e.role == "H1" for e in elements)
def test_untagged_returns_empty(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("thermo-freon12.pdf")
)
assert elements == []
def test_repr(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert "StructureElement" in repr(elements[0])
def test_not_a_pdf(self):
with pytest.raises(ValueError):
pdf_inspector.extract_structure_elements_bytes(b"not a pdf")
# ---------------------------------------------------------------------------
# extract_text_in_regions / extract_text_in_regions_bytes
+2 -2
View File
@@ -724,7 +724,7 @@ checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
[[package]]
name = "pdf-inspector"
version = "0.1.8"
version = "0.1.7"
dependencies = [
"env_logger",
"include_dir",
@@ -740,7 +740,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-wasm"
version = "0.1.4"
version = "0.1.3"
dependencies = [
"console_error_panic_hook",
"js-sys",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-wasm"
version = "0.1.4"
version = "0.1.3"
edition = "2021"
authors = ["Firecrawl Team"]
description = "Browser WebAssembly bindings for pdf-inspector"