Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
afca0c5ac8 | ||
|
|
764ec6e081 | ||
|
|
631f3ff192 | ||
|
|
ba1807dce5 | ||
|
|
597e125731 | ||
|
|
b0b42c4949 | ||
|
|
dd68cc221f | ||
|
|
20c5a0a82a |
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.8"
|
||||
version = "0.1.7"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
|
||||
+3
-4
@@ -5,15 +5,14 @@
|
||||
If you believe you've found a security vulnerability in pdf-inspector, please
|
||||
report it privately so we can fix it before public disclosure.
|
||||
|
||||
**Preferred:** Submit through Firecrawl's Bugcrowd vulnerability disclosure
|
||||
program at <https://bugcrowd.com/engagements/firecrawl-vdp-ess>. Please include:
|
||||
**Preferred:** Email **help@firecrawl.dev** with:
|
||||
|
||||
- A description of the issue and its impact
|
||||
- Steps to reproduce (a minimal PDF or input that triggers the bug is ideal)
|
||||
- The version or commit hash of pdf-inspector you tested against
|
||||
|
||||
**Alternative:** If you'd rather not use Bugcrowd, email
|
||||
**help@firecrawl.dev** with the same details.
|
||||
**Alternative:** Use GitHub's private vulnerability reporting under the
|
||||
[Security tab](https://github.com/firecrawl/pdf-inspector/security/advisories/new).
|
||||
|
||||
We'll acknowledge your report in a timely manner and keep you updated on
|
||||
remediation progress. Please do not open a public GitHub issue for security
|
||||
|
||||
Generated
+2
-2
@@ -851,7 +851,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.8"
|
||||
version = "0.1.7"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
@@ -867,7 +867,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.3"
|
||||
version = "0.2.2"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.3"
|
||||
version = "0.2.2"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
@@ -83,22 +83,6 @@ for (const region of result[0].regions) {
|
||||
}
|
||||
```
|
||||
|
||||
### Async variants
|
||||
|
||||
`processPdf`, `classifyPdf`, and `extractPagesMarkdown` are synchronous and parse on the calling thread — in Node, that's the event loop. For a one-off call in a script that's fine, but in a server a large document can hold the loop for tens to hundreds of milliseconds.
|
||||
|
||||
`processPdfAsync`, `classifyPdfAsync`, and `extractPagesMarkdownAsync` take the same arguments and produce the same results, but run the parse on the libuv thread pool and return a promise, keeping the event loop free. The input buffer is copied before the call returns, so it's safe to reuse or mutate immediately:
|
||||
|
||||
```typescript
|
||||
import { classifyPdfAsync, extractPagesMarkdownAsync } from '@firecrawl/pdf-inspector'
|
||||
|
||||
const classification = await classifyPdfAsync(pdf)
|
||||
if (classification.pdfType === 'TextBased') {
|
||||
const { pages } = await extractPagesMarkdownAsync(pdf)
|
||||
// ...
|
||||
}
|
||||
```
|
||||
|
||||
## Types
|
||||
|
||||
```typescript
|
||||
|
||||
+6
-6
@@ -8,12 +8,12 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+7
-7
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.13.0",
|
||||
"version": "1.12.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -52,11 +52,11 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.13.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.13.0"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
|
||||
}
|
||||
}
|
||||
|
||||
+41
-236
@@ -89,11 +89,6 @@ pub struct TextItem {
|
||||
pub item_type: ItemType,
|
||||
/// URL for link items, `None` for other types.
|
||||
pub link_url: Option<String>,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator, `None`
|
||||
/// when the text is not part of marked content. Join with the
|
||||
/// `page`/`mcid` pairs from [`extractStructureElements`] to attach
|
||||
/// structure-tree roles (headings, paragraphs, …) in tagged PDFs.
|
||||
pub mcid: Option<i64>,
|
||||
}
|
||||
|
||||
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
||||
@@ -158,7 +153,9 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<PageOcrReasons> {
|
||||
fn to_napi_page_ocr_reasons(
|
||||
reasons: Vec<pdf_inspector::PageOcrReasons>,
|
||||
) -> Vec<PageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PageOcrReasons {
|
||||
@@ -205,31 +202,6 @@ where
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Shared implementations (single body behind sync and async entry points)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||
let mut opts = pdf_inspector::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = pdf_inspector::process_pdf_mem_with_options(bytes, opts)
|
||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
}
|
||||
|
||||
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: convert_pdf_type(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public NAPI API
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -238,7 +210,15 @@ fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||
#[napi]
|
||||
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("process_pdf", move || process_pdf_impl(&bytes, pages))
|
||||
catch_panic("process_pdf", move || {
|
||||
let mut opts = pdf_inspector::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = pdf_inspector::process_pdf_mem_with_options(&bytes, opts)
|
||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
})
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
@@ -258,7 +238,16 @@ pub fn detect_pdf(buffer: Buffer) -> Result<PdfResult> {
|
||||
#[napi]
|
||||
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("classify_pdf", move || classify_pdf_impl(&bytes))
|
||||
catch_panic("classify_pdf", move || {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: convert_pdf_type(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from a PDF Buffer.
|
||||
@@ -311,61 +300,12 @@ pub fn extract_text_with_positions(
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type,
|
||||
link_url,
|
||||
mcid: item.mcid,
|
||||
}
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// One structure-tree element reference from a tagged PDF.
|
||||
#[napi(object)]
|
||||
pub struct StructureElementJs {
|
||||
/// 1-indexed page number (matches `TextItem.page`).
|
||||
pub page: u32,
|
||||
/// Marked Content ID from the page's content stream (matches
|
||||
/// `TextItem.mcid`).
|
||||
pub mcid: i64,
|
||||
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
|
||||
/// Custom tags are resolved through the document's role map; tags with
|
||||
/// no standard mapping are returned verbatim.
|
||||
pub role: String,
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from a tagged PDF.
|
||||
///
|
||||
/// Parses the document's structure tree (when present) and returns one
|
||||
/// entry per marked-content reference, resolved to its 1-indexed page,
|
||||
/// MCID, and structure type name. Returns an empty array when the PDF is
|
||||
/// not tagged.
|
||||
///
|
||||
/// Join `(page, mcid)` against the `page`/`mcid` fields from
|
||||
/// [`extractTextWithPositions`] to attach heading levels (H1..H6) and other
|
||||
/// semantic roles to extracted text.
|
||||
///
|
||||
/// Pass 1-indexed page numbers (matching `TextItem.page`) to restrict
|
||||
/// output; omit `pages` for the whole document. Entries are sorted by
|
||||
/// `(page, mcid)`.
|
||||
#[napi]
|
||||
pub fn extract_structure_elements(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> Result<Vec<StructureElementJs>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_structure_elements", move || {
|
||||
let elements = pdf_inspector::extract_structure_elements_mem(&bytes, pages.as_deref())
|
||||
.map_err(|e| to_napi_err(e, "extract_structure_elements"))?;
|
||||
Ok(elements
|
||||
.into_iter()
|
||||
.map(|e| StructureElementJs {
|
||||
page: e.page,
|
||||
mcid: e.mcid,
|
||||
role: e.role,
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF.
|
||||
///
|
||||
/// For hybrid OCR: layout model detects regions in rendered images,
|
||||
@@ -693,32 +633,25 @@ pub fn extract_pages_markdown(
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_pages_markdown", move || {
|
||||
extract_pages_markdown_impl(&bytes, pages.as_deref())
|
||||
})
|
||||
}
|
||||
|
||||
fn extract_pages_markdown_impl(
|
||||
bytes: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(bytes, pages)
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|r| PageMarkdownResult {
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|r| PageMarkdownResult {
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
@@ -759,131 +692,3 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Async variants (libuv thread pool via AsyncTask)
|
||||
//
|
||||
// The synchronous exports above parse on the calling thread, which in Node is
|
||||
// the event loop. These `*Async` variants run the same shared implementations
|
||||
// on the libuv thread pool and hand JavaScript a promise, so servers under
|
||||
// concurrent load keep answering requests while a document parses. The sync
|
||||
// exports keep their names, signatures, and behaviour.
|
||||
//
|
||||
// Each factory copies the input Buffer to an owned `Vec<u8>` on the calling
|
||||
// (JS) thread — deliberately. JS execution is single-threaded, so no JS code
|
||||
// can mutate the buffer while the synchronous part of the call copies it.
|
||||
// Holding the napi `Buffer` and reading it from the worker instead would be
|
||||
// zero-copy, but a caller mutating the buffer before the promise settles
|
||||
// would then race the worker's reads — undefined behavior, not a recoverable
|
||||
// error (a known napi-rs soundness hazard with cross-thread Buffer access).
|
||||
// The copy is a one-time memcpy, negligible next to the parse it unblocks.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct ProcessPdfTask {
|
||||
bytes: Vec<u8>,
|
||||
pages: Option<Vec<u32>>,
|
||||
}
|
||||
|
||||
impl Task for ProcessPdfTask {
|
||||
type Output = PdfResult;
|
||||
type JsValue = PdfResult;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
let pages = self.pages.take();
|
||||
// AssertUnwindSafe: `bytes`/`pages` are moved into the closure and
|
||||
// dropped on unwind — no shared state can be observed broken.
|
||||
catch_panic(
|
||||
"process_pdf",
|
||||
panic::AssertUnwindSafe(move || process_pdf_impl(&bytes, pages)),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Async variant of [`processPdf`]: same result, but the parse runs on the
|
||||
/// libuv thread pool instead of the event loop and the call returns a
|
||||
/// promise. The buffer is copied before the call returns, so it may be
|
||||
/// reused or mutated immediately.
|
||||
// ts_return_type is required: napi-rs emits `Promise<unknown>` for
|
||||
// `AsyncTask<T>` returns without it.
|
||||
#[napi(ts_return_type = "Promise<PdfResult>")]
|
||||
pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<ProcessPdfTask> {
|
||||
AsyncTask::new(ProcessPdfTask {
|
||||
bytes: buffer.to_vec(),
|
||||
pages,
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ClassifyPdfTask {
|
||||
bytes: Vec<u8>,
|
||||
}
|
||||
|
||||
impl Task for ClassifyPdfTask {
|
||||
type Output = PdfClassification;
|
||||
type JsValue = PdfClassification;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
catch_panic(
|
||||
"classify_pdf",
|
||||
panic::AssertUnwindSafe(move || classify_pdf_impl(&bytes)),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Async variant of [`classifyPdf`]: same result, but the classification runs
|
||||
/// on the libuv thread pool instead of the event loop and the call returns a
|
||||
/// promise. The buffer is copied before the call returns, so it may be
|
||||
/// reused or mutated immediately.
|
||||
#[napi(ts_return_type = "Promise<PdfClassification>")]
|
||||
pub fn classify_pdf_async(buffer: Buffer) -> AsyncTask<ClassifyPdfTask> {
|
||||
AsyncTask::new(ClassifyPdfTask {
|
||||
bytes: buffer.to_vec(),
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ExtractPagesMarkdownTask {
|
||||
bytes: Vec<u8>,
|
||||
pages: Option<Vec<u32>>,
|
||||
}
|
||||
|
||||
impl Task for ExtractPagesMarkdownTask {
|
||||
type Output = PagesExtractionResult;
|
||||
type JsValue = PagesExtractionResult;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
let pages = self.pages.take();
|
||||
catch_panic(
|
||||
"extract_pages_markdown",
|
||||
panic::AssertUnwindSafe(move || extract_pages_markdown_impl(&bytes, pages.as_deref())),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Async variant of [`extractPagesMarkdown`]: same result, but the extraction
|
||||
/// runs on the libuv thread pool instead of the event loop and the call
|
||||
/// returns a promise. The buffer is copied before the call returns, so it
|
||||
/// may be reused or mutated immediately.
|
||||
#[napi(ts_return_type = "Promise<PagesExtractionResult>")]
|
||||
pub fn extract_pages_markdown_async(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> AsyncTask<ExtractPagesMarkdownTask> {
|
||||
AsyncTask::new(ExtractPagesMarkdownTask {
|
||||
bytes: buffer.to_vec(),
|
||||
pages,
|
||||
})
|
||||
}
|
||||
|
||||
-110
@@ -2,21 +2,16 @@ import { readFileSync } from 'fs';
|
||||
import { strict as assert } from 'assert';
|
||||
import {
|
||||
processPdf,
|
||||
processPdfAsync,
|
||||
detectPdf,
|
||||
classifyPdf,
|
||||
classifyPdfAsync,
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractStructureElements,
|
||||
extractTextInRegions,
|
||||
detectVectorGridInRegion,
|
||||
extractPagesMarkdown,
|
||||
extractPagesMarkdownAsync,
|
||||
} from './index.js';
|
||||
|
||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||
const taggedFixture = readFileSync('../tests/fixtures/firecrawl_docs_tagged.pdf');
|
||||
|
||||
// --- processPdf ---
|
||||
console.log('Testing processPdf...');
|
||||
@@ -84,46 +79,6 @@ assert.ok(page1Items.length > 0);
|
||||
assert.ok(page1Items.every(i => i.page === 1));
|
||||
console.log(' extractTextWithPositions with pages: OK');
|
||||
|
||||
// mcid: undefined on untagged PDFs, numeric on tagged marked content
|
||||
assert.ok(items.every(i => i.mcid === undefined || typeof i.mcid === 'number'));
|
||||
const taggedItems = extractTextWithPositions(taggedFixture);
|
||||
assert.ok(
|
||||
taggedItems.some(i => typeof i.mcid === 'number'),
|
||||
'tagged PDF text items should carry Marked Content IDs',
|
||||
);
|
||||
console.log(' extractTextWithPositions mcid: OK');
|
||||
|
||||
// --- extractStructureElements ---
|
||||
console.log('Testing extractStructureElements...');
|
||||
const structureElements = extractStructureElements(taggedFixture);
|
||||
assert.ok(structureElements.length > 0);
|
||||
assert.ok(structureElements.every(e => typeof e.page === 'number'));
|
||||
assert.ok(structureElements.every(e => typeof e.mcid === 'number'));
|
||||
assert.ok(structureElements.every(e => typeof e.role === 'string' && e.role.length > 0));
|
||||
assert.ok(
|
||||
structureElements.some(e => e.role === 'H1'),
|
||||
'tagged fixture should surface H1 heading roles',
|
||||
);
|
||||
|
||||
// (page, mcid) joins against extractTextWithPositions to recover heading text
|
||||
const h1Refs = new Set(
|
||||
structureElements.filter(e => e.role === 'H1').map(e => `${e.page}:${e.mcid}`),
|
||||
);
|
||||
const h1Text = taggedItems
|
||||
.filter(i => typeof i.mcid === 'number' && h1Refs.has(`${i.page}:${i.mcid}`))
|
||||
.map(i => i.text)
|
||||
.join('');
|
||||
assert.ok(h1Text.trim().length > 0, 'H1 join should recover heading text');
|
||||
|
||||
// pages filter is 1-indexed, matching TextItem.page
|
||||
const page1Elements = extractStructureElements(taggedFixture, [1]);
|
||||
assert.ok(page1Elements.length > 0);
|
||||
assert.ok(page1Elements.every(e => e.page === 1));
|
||||
|
||||
// untagged PDFs yield an empty array
|
||||
assert.deepEqual(extractStructureElements(fixture), []);
|
||||
console.log(' extractStructureElements: OK');
|
||||
|
||||
// --- extractTextInRegions ---
|
||||
console.log('Testing extractTextInRegions...');
|
||||
const regionResults = extractTextInRegions(fixture, [
|
||||
@@ -169,75 +124,10 @@ assert.equal(picked.pages[0].page, 2);
|
||||
assert.equal(picked.pages[1].page, 0);
|
||||
console.log(' extractPagesMarkdown with pages: OK');
|
||||
|
||||
// --- Async variants ---
|
||||
console.log('Testing async variants...');
|
||||
|
||||
// processPdfAsync returns a promise and matches the sync result
|
||||
const asyncResultPromise = processPdfAsync(fixture);
|
||||
assert.ok(asyncResultPromise instanceof Promise);
|
||||
const asyncResult = await asyncResultPromise;
|
||||
assert.equal(asyncResult.pdfType, result.pdfType);
|
||||
assert.equal(asyncResult.pageCount, result.pageCount);
|
||||
assert.equal(asyncResult.markdown, result.markdown);
|
||||
console.log(' processPdfAsync: OK');
|
||||
|
||||
// processPdfAsync with pages
|
||||
const asyncResult2 = await processPdfAsync(fixture, [1]);
|
||||
assert.equal(asyncResult2.markdown, result2.markdown);
|
||||
console.log(' processPdfAsync with pages: OK');
|
||||
|
||||
// classifyPdfAsync matches the sync result
|
||||
const asyncClassified = await classifyPdfAsync(fixture);
|
||||
assert.equal(asyncClassified.pdfType, classified.pdfType);
|
||||
assert.equal(asyncClassified.pageCount, classified.pageCount);
|
||||
assert.equal(asyncClassified.confidence, classified.confidence);
|
||||
assert.deepEqual(asyncClassified.pagesNeedingOcr, classified.pagesNeedingOcr);
|
||||
console.log(' classifyPdfAsync: OK');
|
||||
|
||||
// extractPagesMarkdownAsync matches the sync result
|
||||
const asyncAllPages = await extractPagesMarkdownAsync(fixture);
|
||||
assert.equal(asyncAllPages.pages.length, allPages.pages.length);
|
||||
assert.deepEqual(
|
||||
asyncAllPages.pages.map(p => p.markdown),
|
||||
allPages.pages.map(p => p.markdown),
|
||||
);
|
||||
assert.equal(asyncAllPages.isComplex, allPages.isComplex);
|
||||
console.log(' extractPagesMarkdownAsync: OK');
|
||||
|
||||
// selected pages preserve caller order
|
||||
const asyncPicked = await extractPagesMarkdownAsync(fixture, [2, 0]);
|
||||
assert.equal(asyncPicked.pages.length, 2);
|
||||
assert.equal(asyncPicked.pages[0].page, 2);
|
||||
assert.equal(asyncPicked.pages[1].page, 0);
|
||||
console.log(' extractPagesMarkdownAsync with pages: OK');
|
||||
|
||||
// input buffer is copied at call time: mutating it immediately after the
|
||||
// call must not affect the in-flight parse
|
||||
const scratch = Buffer.from(fixture);
|
||||
const inFlight = processPdfAsync(scratch);
|
||||
scratch.fill(0);
|
||||
const fromMutated = await inFlight;
|
||||
assert.equal(fromMutated.markdown, result.markdown);
|
||||
console.log(' processPdfAsync input copied at call time: OK');
|
||||
|
||||
// concurrent async calls all settle
|
||||
const [c1, c2, c3] = await Promise.all([
|
||||
processPdfAsync(fixture),
|
||||
classifyPdfAsync(fixture),
|
||||
extractPagesMarkdownAsync(fixture),
|
||||
]);
|
||||
assert.equal(c1.pdfType, 'TextBased');
|
||||
assert.equal(c2.pdfType, 'TextBased');
|
||||
assert.equal(c3.pages.length, 3);
|
||||
console.log(' concurrent async calls: OK');
|
||||
|
||||
// --- Error handling ---
|
||||
console.log('Testing error handling...');
|
||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
|
||||
await assert.rejects(processPdfAsync(Buffer.from('not a pdf')), /process_pdf/);
|
||||
await assert.rejects(classifyPdfAsync(Buffer.from('')), /classify_pdf/);
|
||||
await assert.rejects(extractPagesMarkdownAsync(Buffer.from('')), /extract_pages_markdown/);
|
||||
console.log(' error handling: OK');
|
||||
|
||||
console.log('\nAll NAPI tests passed!');
|
||||
|
||||
@@ -51,20 +51,6 @@ class TextItem:
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
mcid: Optional[int]
|
||||
"""Marked Content ID from the content stream's BDC/BMC operator, None when
|
||||
the text is not part of marked content. Join with the (page, mcid) pairs
|
||||
from extract_structure_elements to attach structure-tree roles in tagged
|
||||
PDFs."""
|
||||
|
||||
class StructureElement:
|
||||
"""One structure-tree element reference from a tagged PDF."""
|
||||
page: int
|
||||
"""1-indexed page number (matches TextItem.page)."""
|
||||
mcid: int
|
||||
"""Marked Content ID from the page's content stream (matches TextItem.mcid)."""
|
||||
role: str
|
||||
"""Standard structure type name ("H1".."H6", "P", "Table", "TD", ...)."""
|
||||
|
||||
class RegionText:
|
||||
"""Extracted text for a single region."""
|
||||
@@ -146,27 +132,6 @@ def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] =
|
||||
"""Extract text with position information from bytes."""
|
||||
...
|
||||
|
||||
def extract_structure_elements(path: str, pages: Optional[list[int]] = None) -> list[StructureElement]:
|
||||
"""Extract structure-tree element references from a tagged PDF file.
|
||||
|
||||
Returns one entry per marked-content reference, resolved to its 1-indexed
|
||||
page, MCID, and structure type name ("H1".."H6", "P", "Table", ...), sorted
|
||||
by (page, mcid). Returns an empty list when the PDF is not tagged.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
pages: Optional list of 1-indexed pages (matching ``TextItem.page``).
|
||||
When ``None`` (default), the whole document is returned.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_structure_elements_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[StructureElement]:
|
||||
"""Extract structure-tree element references from tagged PDF bytes.
|
||||
|
||||
See :func:`extract_structure_elements` for details.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_text_in_regions(
|
||||
path: str,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ build-backend = "maturin"
|
||||
name = "pdf-inspector"
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.7"
|
||||
version = "0.2.6"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
+17
-327
@@ -41,110 +41,15 @@ pub(crate) fn detect_columns(
|
||||
}
|
||||
debug!("page {}: detect_columns: {} items", page, page_items.len());
|
||||
|
||||
// The width of one ordinary page, used three ways below: as the largest
|
||||
// credible width for a single text run, as the size of empty gap that marks
|
||||
// content as detached, and as the span past which those checks run at all.
|
||||
// This is a heuristic, not a format rule: PDF 2.0 sets no page-size limit,
|
||||
// and since PDF 1.6 `UserUnit` scales a page's physical size independently
|
||||
// of its coordinates. 14_400 units (200in at the default 1/72in unit) is
|
||||
// the traditional Acrobat architectural limit, which makes it a reasonable
|
||||
// "wider than any ordinary page" mark in coordinate space.
|
||||
const MAX_PAGE_EXTENT: f32 = 14_400.0;
|
||||
// A detached cluster is only dropped if it also holds a small minority of
|
||||
// the items, so a genuine two-part layout keeps its full bounds even when
|
||||
// the halves are far apart.
|
||||
const MAX_TRIM_FRACTION: f32 = 0.10;
|
||||
|
||||
// Position and width of each item, skipping only non-finite geometry.
|
||||
let finite_span = |i: &&TextItem| -> Option<(f32, f32)> {
|
||||
let (left, width) = (i.x, effective_width(i));
|
||||
(left.is_finite() && (left + width).is_finite()).then_some((left, width))
|
||||
};
|
||||
|
||||
let (min_left, max_right, total) = page_items.iter().filter_map(finite_span).fold(
|
||||
(f32::INFINITY, f32::NEG_INFINITY, 0usize),
|
||||
|(lo, hi, n), (left, width)| (lo.min(left), hi.max(left + width), n + 1),
|
||||
);
|
||||
|
||||
// No item had usable geometry, so there is no layout to report.
|
||||
if total == 0 {
|
||||
return vec![];
|
||||
}
|
||||
|
||||
// Every threshold below (gutter margins, spanning-item width, the XY-cut
|
||||
// margin) is a fraction of the page width, so a far item can set the scale
|
||||
// for the whole page and shrink the effective detection window to a
|
||||
// rounding error — real gutters then fall inside the margin band and a
|
||||
// genuine multi-column page collapses to one region.
|
||||
//
|
||||
// Anything inside one page extent is ordinary, so the common case keeps the
|
||||
// plain bounds and skips the work below entirely.
|
||||
let (x_min, x_max) = if max_right - min_left <= MAX_PAGE_EXTENT {
|
||||
(min_left, max_right)
|
||||
} else {
|
||||
// Discarding content needs positive evidence that it is not part of the
|
||||
// layout, because a count-based rule alone cannot tell a stray from a
|
||||
// sparse far sidebar. The evidence is geometric: positions are grouped
|
||||
// into clusters separated by more than a whole page of continuous
|
||||
// emptiness. Real content, however sparse, does not leave a void that
|
||||
// large; a malformed coordinate sits alone beyond one.
|
||||
let mut spans: Vec<(f32, f32)> = page_items.iter().filter_map(finite_span).collect();
|
||||
spans.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
let mut core: Option<std::ops::Range<usize>> = None;
|
||||
let mut start = 0usize;
|
||||
for i in 1..=spans.len() {
|
||||
if i < spans.len() && spans[i].0 - spans[i - 1].0 <= MAX_PAGE_EXTENT {
|
||||
continue;
|
||||
}
|
||||
if core.as_ref().is_none_or(|best| i - start > best.len()) {
|
||||
core = Some(start..i);
|
||||
}
|
||||
start = i;
|
||||
}
|
||||
let mut core = core.unwrap_or(0..spans.len());
|
||||
|
||||
// Only drop the detached clusters when they are a small minority, so a
|
||||
// genuine two-part layout keeps its full bounds.
|
||||
let dropped = spans.len() - core.len();
|
||||
if dropped as f32 > spans.len() as f32 * MAX_TRIM_FRACTION {
|
||||
core = 0..spans.len();
|
||||
}
|
||||
let core = &spans[core];
|
||||
|
||||
// Positions cannot be inflated by a bogus width, so the spread of the
|
||||
// content is a sound scale for judging one. A run much wider than the
|
||||
// page's own content is a malformed width — the test is relative, so a
|
||||
// genuinely large page keeps its genuinely long runs.
|
||||
let (lo, widest_left) = (core[0].0, core[core.len() - 1].0);
|
||||
let max_run_width = (widest_left - lo) + MAX_PAGE_EXTENT;
|
||||
let hi = core
|
||||
.iter()
|
||||
.filter(|&&(_, width)| width <= max_run_width)
|
||||
.map(|&(left, width)| left + width)
|
||||
.fold(widest_left, f32::max);
|
||||
|
||||
if lo != min_left || hi != max_right {
|
||||
debug!(
|
||||
"page {page}: bounds {min_left}..{max_right} exceed one page; \
|
||||
dropped {dropped}/{} detached item(s), using {lo}..{hi}",
|
||||
spans.len()
|
||||
);
|
||||
}
|
||||
(lo, hi)
|
||||
};
|
||||
|
||||
// Hard ceiling on the histogram size, independent of the trimming above:
|
||||
// the bounds are attacker-influenced, so an unclamped
|
||||
// `page_width / BIN_WIDTH` lets a crafted PDF force an arbitrarily large
|
||||
// `vec![0u32; num_bins]` allocation. 65_536 bins covers ~128k points at
|
||||
// BIN_WIDTH 2.0 — roughly 9x the largest legal page — so this never binds
|
||||
// on a real layout. Kept as a bound that does not depend on the outlier
|
||||
// heuristic staying correct.
|
||||
const MAX_BINS: usize = 65_536;
|
||||
// Find page bounds
|
||||
let x_min = page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let x_max = page_items
|
||||
.iter()
|
||||
.map(|i| i.x + effective_width(i))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
|
||||
let page_width = x_max - x_min;
|
||||
if !page_width.is_finite() || page_width < 200.0 {
|
||||
if page_width < 200.0 {
|
||||
return vec![ColumnRegion { x_min, x_max }];
|
||||
}
|
||||
|
||||
@@ -152,20 +57,13 @@ pub(crate) fn detect_columns(
|
||||
return vec![ColumnRegion { x_min, x_max }];
|
||||
}
|
||||
|
||||
// Widen the bins rather than dropping the tail of the page. Clamping the
|
||||
// count alone would leave anything past MAX_BINS * BIN_WIDTH outside the
|
||||
// histogram, folded into the last bin, which places gutters at the wrong
|
||||
// coordinates. Scaling keeps full coverage under the same allocation
|
||||
// ceiling; only the resolution degrades, and only beyond ~131k points.
|
||||
let bin_width = BIN_WIDTH.max(page_width / MAX_BINS as f32);
|
||||
|
||||
// Build occupancy histogram.
|
||||
// Exclude items wider than 60% of page width — these are spanning items
|
||||
// (titles, full-width paragraphs) that would fill the gutter and prevent
|
||||
// detection of partial-page column layouts (e.g. two-column abstracts on
|
||||
// a page that also has single-column introduction text).
|
||||
let wide_threshold = page_width * 0.6;
|
||||
let num_bins = ((page_width / bin_width).ceil() as usize).clamp(1, MAX_BINS);
|
||||
let num_bins = ((page_width / BIN_WIDTH).ceil() as usize).max(1);
|
||||
let mut histogram = vec![0u32; num_bins];
|
||||
|
||||
for item in &page_items {
|
||||
@@ -173,8 +71,8 @@ pub(crate) fn detect_columns(
|
||||
if w > wide_threshold {
|
||||
continue;
|
||||
}
|
||||
let left = ((item.x - x_min) / bin_width).floor() as usize;
|
||||
let right = (((item.x + w) - x_min) / bin_width).ceil() as usize;
|
||||
let left = ((item.x - x_min) / BIN_WIDTH).floor() as usize;
|
||||
let right = (((item.x + w) - x_min) / BIN_WIDTH).ceil() as usize;
|
||||
let left = left.min(num_bins);
|
||||
let right = right.min(num_bins);
|
||||
for count in histogram.iter_mut().take(right).skip(left) {
|
||||
@@ -211,12 +109,12 @@ pub(crate) fn detect_columns(
|
||||
let valleys: Vec<(usize, usize)> = valleys
|
||||
.into_iter()
|
||||
.filter(|&(start, end)| {
|
||||
let width_pts = (end - start) as f32 * bin_width;
|
||||
let width_pts = (end - start) as f32 * BIN_WIDTH;
|
||||
if width_pts < MIN_GUTTER_WIDTH {
|
||||
return false;
|
||||
}
|
||||
// Valley center must not be within 5% of page edges
|
||||
let center_pts = ((start + end) as f32 / 2.0) * bin_width;
|
||||
let center_pts = ((start + end) as f32 / 2.0) * BIN_WIDTH;
|
||||
center_pts > margin_threshold && center_pts < (page_width - margin_threshold)
|
||||
})
|
||||
.collect();
|
||||
@@ -234,7 +132,7 @@ pub(crate) fn detect_columns(
|
||||
&histogram,
|
||||
num_bins,
|
||||
x_min,
|
||||
bin_width,
|
||||
BIN_WIDTH,
|
||||
page_width,
|
||||
margin_threshold,
|
||||
);
|
||||
@@ -243,7 +141,7 @@ pub(crate) fn detect_columns(
|
||||
&rel_valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
bin_width,
|
||||
BIN_WIDTH,
|
||||
x_max,
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
@@ -284,7 +182,7 @@ pub(crate) fn detect_columns(
|
||||
&valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
bin_width,
|
||||
BIN_WIDTH,
|
||||
x_max,
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
@@ -298,7 +196,7 @@ pub(crate) fn detect_columns(
|
||||
&valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
bin_width,
|
||||
BIN_WIDTH,
|
||||
x_max,
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
@@ -1929,7 +1827,7 @@ fn split_column_stragglers(lines: Vec<TextLine>) -> (Vec<TextLine>, Vec<TextLine
|
||||
.unwrap();
|
||||
|
||||
let (cs, ce) = segments[core_seg];
|
||||
let mut core = Vec::with_capacity(ce.saturating_sub(cs));
|
||||
let mut core = Vec::with_capacity(ce - cs);
|
||||
let mut stragglers = Vec::new();
|
||||
for (i, line) in lines.into_iter().enumerate() {
|
||||
if i >= cs && i < ce {
|
||||
@@ -2635,214 +2533,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extreme_far_coordinate_does_not_allocate_unboundedly() {
|
||||
// A crafted PDF can place a text run at an arbitrary coordinate via the
|
||||
// text matrix. The derived page width must not drive an unbounded
|
||||
// histogram allocation (previously `page_width / BIN_WIDTH` bins with no
|
||||
// upper bound would try to reserve terabytes and abort the process).
|
||||
let mut items = Vec::new();
|
||||
for i in 0..24 {
|
||||
items.push(make_item(1, i as f32 * 10.0, 700.0 - i as f32 * 5.0, "A"));
|
||||
}
|
||||
// Item placed 1e12 points away — 5e11 bins if left unclamped.
|
||||
items.push(make_item(1, 1e12, 700.0, "Z"));
|
||||
|
||||
// Must return without aborting; content is preserved as a single region.
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert!(!cols.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_finite_coordinates_never_leak_into_region_bounds() {
|
||||
// An inf/NaN coordinate must not escape as a column boundary: callers
|
||||
// treat these as page/column edges.
|
||||
for bad_x in [f32::INFINITY, f32::NEG_INFINITY, f32::NAN] {
|
||||
let mut items = Vec::new();
|
||||
for i in 0..24 {
|
||||
items.push(make_item(1, i as f32 * 10.0, 700.0 - i as f32 * 5.0, "A"));
|
||||
}
|
||||
items.push(make_item(1, bad_x, 700.0, "Z"));
|
||||
|
||||
for col in detect_columns(&items, 1, false) {
|
||||
assert!(
|
||||
col.x_min.is_finite() && col.x_max.is_finite(),
|
||||
"bad_x {bad_x} leaked bounds {}..{}",
|
||||
col.x_min,
|
||||
col.x_max
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn all_non_finite_coordinates_yield_no_columns() {
|
||||
let items: Vec<TextItem> = (0..24)
|
||||
.map(|i| make_item(1, f32::NAN, 700.0 - i as f32 * 5.0, "A"))
|
||||
.collect();
|
||||
|
||||
assert!(detect_columns(&items, 1, false).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_bad_item_does_not_disable_column_detection() {
|
||||
// A single stray item should not collapse a clean two-column page to
|
||||
// one region. Every gutter threshold is a fraction of the page width,
|
||||
// so an untrimmed outlier pushes real gutters inside the rejected
|
||||
// margin band. A malformed *width* at an ordinary position poisons the
|
||||
// bounds just as a malformed position does.
|
||||
for (label, bad_x, bad_width) in [
|
||||
("nan position", f32::NAN, 0.0),
|
||||
("inf position", f32::INFINITY, 0.0),
|
||||
("far position", 50_000.0, 0.0),
|
||||
("very far position", 1e12, 0.0),
|
||||
("huge width", 100.0, 1e12),
|
||||
("inf width", 100.0, f32::INFINITY),
|
||||
] {
|
||||
let mut items = Vec::new();
|
||||
items.extend(fill_zone(1, 30.0, 280.0, 750.0, 50.0));
|
||||
items.extend(fill_zone(1, 320.0, 570.0, 750.0, 50.0));
|
||||
let mut bad = make_item(1, bad_x, 400.0, "Z");
|
||||
bad.width = bad_width;
|
||||
items.push(bad);
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert_eq!(
|
||||
cols.len(),
|
||||
2,
|
||||
"{label}: expected 2 columns, got {}",
|
||||
cols.len()
|
||||
);
|
||||
for col in &cols {
|
||||
assert!(
|
||||
col.x_max - col.x_min <= MAX_PAGE_EXTENT_FOR_TEST,
|
||||
"{label}: region {}..{} exceeds one page",
|
||||
col.x_min,
|
||||
col.x_max
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mirrors `MAX_PAGE_EXTENT` in `detect_columns`.
|
||||
const MAX_PAGE_EXTENT_FOR_TEST: f32 = 14_400.0;
|
||||
|
||||
#[test]
|
||||
fn very_wide_page_keeps_full_histogram_coverage() {
|
||||
// Beyond MAX_BINS * BIN_WIDTH (~131k points) the bins must widen rather
|
||||
// than stop covering the page. Three zones: the first gutter is inside
|
||||
// the old coverage limit, the second is past it. Because the first
|
||||
// gutter is found, the XY-cut fallback never runs, so a truncated
|
||||
// histogram silently reports two columns instead of three.
|
||||
let mut items = Vec::new();
|
||||
items.extend(fill_zone(1, 0.0, 60_000.0, 750.0, 700.0));
|
||||
items.extend(fill_zone(1, 70_000.0, 140_000.0, 750.0, 700.0));
|
||||
items.extend(fill_zone(1, 160_000.0, 200_000.0, 750.0, 700.0));
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert_eq!(
|
||||
cols.len(),
|
||||
3,
|
||||
"Expected 3 columns across a 200k-wide page, got {}",
|
||||
cols.len()
|
||||
);
|
||||
assert!(
|
||||
(140_000.0..=160_000.0).contains(&cols[1].x_max),
|
||||
"second gutter at {}, expected inside the real 140k..160k gap",
|
||||
cols[1].x_max
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn large_page_with_legitimately_long_runs_is_kept() {
|
||||
// On a very large page, individual runs can exceed one ordinary page's
|
||||
// width. They are real content, so they must not be judged malformed:
|
||||
// the page keeps its columns and its full right edge.
|
||||
let mut items = Vec::new();
|
||||
for row in 0..30 {
|
||||
let y = 750.0 - row as f32 * 14.0;
|
||||
let mut left = make_item(1, 0.0, y, "Left run");
|
||||
left.width = 20_000.0;
|
||||
let mut right = make_item(1, 25_000.0, y, "Right run");
|
||||
right.width = 20_000.0;
|
||||
items.extend([left, right]);
|
||||
}
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert!(
|
||||
!cols.is_empty(),
|
||||
"a page of long-but-valid runs must still report a layout"
|
||||
);
|
||||
let right_edge = cols
|
||||
.iter()
|
||||
.map(|c| c.x_max)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
assert!(
|
||||
right_edge > 44_000.0,
|
||||
"long runs were treated as malformed: right edge {right_edge}, expected ~45_000"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_far_sidebar_on_a_large_page_is_kept() {
|
||||
// A large-format page with a thin, sparsely-populated sidebar far from
|
||||
// the main block. The sidebar is a small minority of the items, so an
|
||||
// item-count rule alone would discard it — but nothing about its
|
||||
// geometry says it is invalid, so its bounds must survive.
|
||||
let mut items = Vec::new();
|
||||
items.extend(fill_zone(1, 0.0, 12_000.0, 750.0, 500.0));
|
||||
for i in 0..12 {
|
||||
items.push(make_item(1, 24_000.0, 750.0 - i as f32 * 14.0, "Sidebar"));
|
||||
}
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
let right_edge = cols
|
||||
.iter()
|
||||
.map(|c| c.x_max)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
assert!(
|
||||
right_edge > 24_000.0,
|
||||
"sidebar was trimmed away: right edge {right_edge}, expected >24_000"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn genuinely_wide_layout_keeps_its_true_bounds() {
|
||||
// A large-format page whose content really is spread beyond one
|
||||
// ordinary page must not be trimmed to the median cluster: its far
|
||||
// items are the majority, not strays.
|
||||
let mut items = Vec::new();
|
||||
items.extend(fill_zone(1, 100.0, 20_000.0, 750.0, 600.0));
|
||||
items.extend(fill_zone(1, 22_000.0, 40_000.0, 750.0, 600.0));
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
let widest = cols
|
||||
.iter()
|
||||
.map(|c| c.x_max)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
assert!(
|
||||
widest > 35_000.0,
|
||||
"wide layout was trimmed: right edge {widest}, expected ~40_000"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn oversized_but_legal_page_is_not_trimmed() {
|
||||
// A wide-format page well inside the 14_400pt spec limit must keep its
|
||||
// real bounds — outlier trimming is only for spans beyond a legal page.
|
||||
let mut items = Vec::new();
|
||||
items.extend(fill_zone(1, 100.0, 4_000.0, 750.0, 400.0));
|
||||
items.extend(fill_zone(1, 4_400.0, 8_000.0, 750.0, 400.0));
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert_eq!(cols.len(), 2, "Expected 2 columns, got {}", cols.len());
|
||||
assert!(
|
||||
cols[1].x_max > 7_000.0,
|
||||
"right column should keep its true extent, got {}",
|
||||
cols[1].x_max
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn two_column_regression_guard() {
|
||||
// Standard 2-column layout with clear gutter at center
|
||||
|
||||
+5
-311
@@ -2,51 +2,11 @@
|
||||
|
||||
use crate::types::{ItemType, TextItem};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{resolve_array, resolve_dict};
|
||||
use super::get_number;
|
||||
|
||||
/// Upper bound on the number of form-field nodes visited during a single
|
||||
/// `extract_form_fields` pass. A crafted PDF can chain thousands of distinct
|
||||
/// `/Kids` fields to blow the stack even without an outright reference cycle,
|
||||
/// so we cap total traversal work in addition to detecting cycles.
|
||||
const MAX_FORM_FIELD_NODES: usize = 100_000;
|
||||
|
||||
/// Upper bound on `/Kids` recursion depth. Real AcroForm hierarchies are only
|
||||
/// a few levels deep (fields → child fields → widgets); a crafted PDF can chain
|
||||
/// tens of thousands of distinct fields into a linear `/Kids` list that would
|
||||
/// overflow the stack via depth-first recursion long before the node budget is
|
||||
/// reached. This depth cap bounds the stack independently of total node count.
|
||||
const MAX_FORM_FIELD_DEPTH: usize = 100;
|
||||
|
||||
/// Traversal budget for the AcroForm field walk. Bounds both the number of
|
||||
/// distinct nodes visited *and* the total number of `/Fields`/`/Kids` entries
|
||||
/// examined.
|
||||
///
|
||||
/// Counting `visited` alone is not enough: invalid entries (non-references) and
|
||||
/// duplicate references never grow `visited`, so an oversized array full of them
|
||||
/// would iterate to completion no matter how large. Charging every examined
|
||||
/// entry against the same budget makes it a real cap on traversal work.
|
||||
pub(crate) struct FieldWalkBudget {
|
||||
visited: HashSet<ObjectId>,
|
||||
examined: usize,
|
||||
}
|
||||
|
||||
impl FieldWalkBudget {
|
||||
fn new() -> Self {
|
||||
Self {
|
||||
visited: HashSet::new(),
|
||||
examined: 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// True once the budget is spent; callers must stop iterating and recursing.
|
||||
fn exhausted(&self) -> bool {
|
||||
self.visited.len() >= MAX_FORM_FIELD_NODES || self.examined >= MAX_FORM_FIELD_NODES
|
||||
}
|
||||
}
|
||||
|
||||
pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> Vec<TextItem> {
|
||||
let mut links = Vec::new();
|
||||
|
||||
@@ -186,12 +146,9 @@ pub(crate) fn extract_form_fields(
|
||||
Err(_) => return items,
|
||||
};
|
||||
|
||||
// Borrow the array rather than cloning it: a crafted `/Fields` can be huge,
|
||||
// and cloning would pay an O(n) allocation/copy before the budget check
|
||||
// below can stop the work.
|
||||
let fields = match acroform.get(b"Fields") {
|
||||
Ok(obj) => match resolve_array(doc, obj) {
|
||||
Some(arr) => arr,
|
||||
Some(arr) => arr.clone(),
|
||||
None => return items,
|
||||
},
|
||||
Err(_) => return items,
|
||||
@@ -201,19 +158,7 @@ pub(crate) fn extract_form_fields(
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
// Bound the walk so a crafted PDF cannot send us into unbounded recursion
|
||||
// via a `/Kids` cycle, a deep chain, or an oversized array of invalid or
|
||||
// duplicate entries.
|
||||
let mut budget = FieldWalkBudget::new();
|
||||
|
||||
for field_obj in fields {
|
||||
// Stop once the budget is spent so a `/Fields` array wider than the
|
||||
// budget can't burn CPU iterating entries whose walk would no-op. Charge
|
||||
// every entry (including invalid ones) against the budget.
|
||||
if budget.exhausted() {
|
||||
break;
|
||||
}
|
||||
budget.examined += 1;
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(
|
||||
doc,
|
||||
@@ -223,8 +168,6 @@ pub(crate) fn extract_form_fields(
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
&mut budget,
|
||||
0,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -259,7 +202,6 @@ fn annotation_page_map(
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
field_id: ObjectId,
|
||||
@@ -268,22 +210,7 @@ pub(crate) fn walk_form_fields(
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
budget: &mut FieldWalkBudget,
|
||||
depth: usize,
|
||||
) {
|
||||
// Guard against `/Kids` cycles and pathologically large field trees.
|
||||
// Exceeding the depth cap means the chain is too deep to be a legitimate
|
||||
// form (and would overflow the stack); an exhausted budget means the tree is
|
||||
// too large. Both checks run *before* inserting so the visited set can never
|
||||
// grow past the budget.
|
||||
if depth > MAX_FORM_FIELD_DEPTH || budget.exhausted() {
|
||||
return;
|
||||
}
|
||||
// Revisiting an object ID means we hit a `/Kids` cycle.
|
||||
if !budget.visited.insert(field_id) {
|
||||
return;
|
||||
}
|
||||
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
Ok(d) => d,
|
||||
Err(_) => return,
|
||||
@@ -314,19 +241,9 @@ pub(crate) fn walk_form_fields(
|
||||
|
||||
// Check for /Kids — if present, recurse into children
|
||||
if let Ok(kids_obj) = field_dict.get(b"Kids") {
|
||||
// Iterate the borrowed array directly — cloning a crafted, oversized
|
||||
// `/Kids` would allocate and copy every entry before the budget check
|
||||
// below could stop the work.
|
||||
if let Some(kids) = resolve_array(doc, kids_obj) {
|
||||
for kid in kids {
|
||||
// Stop once the budget is spent so a `/Kids` array wider than the
|
||||
// budget can't burn CPU iterating entries whose walk would no-op.
|
||||
// Charge every entry (including invalid/duplicate ones) against
|
||||
// the budget so this is a true traversal-work cap.
|
||||
if budget.exhausted() {
|
||||
break;
|
||||
}
|
||||
budget.examined += 1;
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(
|
||||
doc,
|
||||
@@ -336,8 +253,6 @@ pub(crate) fn walk_form_fields(
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
budget,
|
||||
depth + 1,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -496,225 +411,4 @@ mod tests {
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn kids_self_cycle_does_not_overflow_stack() {
|
||||
// A crafted AcroForm field that lists itself in `/Kids` must not send
|
||||
// the traversal into unbounded recursion.
|
||||
let mut doc = Document::new();
|
||||
let field_id = doc.new_object_id();
|
||||
doc.set_object(
|
||||
field_id,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("loop"),
|
||||
"Kids" => vec![Object::Reference(field_id)],
|
||||
},
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(field_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
// Completes (rather than overflowing the stack) and yields no items.
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn kids_mutual_cycle_terminates() {
|
||||
// Two fields that reference each other via `/Kids` form a cycle that
|
||||
// must also terminate.
|
||||
let mut doc = Document::new();
|
||||
let field_a = doc.new_object_id();
|
||||
let field_b = doc.new_object_id();
|
||||
doc.set_object(
|
||||
field_a,
|
||||
dictionary! {
|
||||
"T" => Object::string_literal("a"),
|
||||
"Kids" => vec![Object::Reference(field_b)],
|
||||
},
|
||||
);
|
||||
doc.set_object(
|
||||
field_b,
|
||||
dictionary! {
|
||||
"T" => Object::string_literal("b"),
|
||||
"Kids" => vec![Object::Reference(field_a)],
|
||||
},
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(field_a)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn deep_acyclic_kids_chain_does_not_overflow_stack() {
|
||||
// A long chain of *distinct* fields (no cycle) must also terminate:
|
||||
// the visited set alone would still recurse to the chain length, so
|
||||
// the depth cap is what prevents a stack overflow here.
|
||||
let mut doc = Document::new();
|
||||
let n = MAX_FORM_FIELD_DEPTH * 500;
|
||||
let ids: Vec<ObjectId> = (0..=n).map(|_| doc.new_object_id()).collect();
|
||||
for i in 0..n {
|
||||
doc.set_object(
|
||||
ids[i],
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"Kids" => vec![Object::Reference(ids[i + 1])],
|
||||
},
|
||||
);
|
||||
}
|
||||
// Leaf carries a value; it sits far below the depth cap so it is never
|
||||
// reached, proving traversal stops early rather than crashing.
|
||||
doc.set_object(
|
||||
ids[n],
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("leaf"),
|
||||
"V" => Object::string_literal("x"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(ids[0])],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_tree_traversal_stops_at_node_budget() {
|
||||
// A single field with a `/Kids` array wider than the node budget must
|
||||
// stop traversal at the cap rather than growing `visited` (and the work)
|
||||
// without bound. Each processed leaf emits one item, so the item count
|
||||
// is bounded by the budget and reaches right up to it (a couple of
|
||||
// slots go to the root and the boundary node charged against the cap).
|
||||
let mut doc = Document::new();
|
||||
let fanout = MAX_FORM_FIELD_NODES + 50;
|
||||
let leaf_ids: Vec<ObjectId> = (0..fanout).map(|_| doc.new_object_id()).collect();
|
||||
for &leaf in &leaf_ids {
|
||||
doc.set_object(
|
||||
leaf,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"V" => Object::string_literal("v"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
}
|
||||
let kids: Vec<Object> = leaf_ids.iter().map(|&id| Object::Reference(id)).collect();
|
||||
let root_id = doc.add_object(dictionary! {
|
||||
"T" => Object::string_literal("root"),
|
||||
"Kids" => kids,
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(root_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
// Extraction stops at the budget: bounded above by the cap, and it gets
|
||||
// right up to it (allowing a small delta for the root/boundary nodes
|
||||
// charged against the budget).
|
||||
assert!(items.len() <= MAX_FORM_FIELD_NODES);
|
||||
assert!(items.len() >= MAX_FORM_FIELD_NODES - 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_top_level_fields_stop_at_node_budget() {
|
||||
// A top-level `/Fields` array wider than the budget must also stop at
|
||||
// the cap: the item count is bounded by the budget and reaches right up
|
||||
// to it.
|
||||
let mut doc = Document::new();
|
||||
let fanout = MAX_FORM_FIELD_NODES + 50;
|
||||
let leaf_ids: Vec<ObjectId> = (0..fanout).map(|_| doc.new_object_id()).collect();
|
||||
for &leaf in &leaf_ids {
|
||||
doc.set_object(
|
||||
leaf,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"V" => Object::string_literal("v"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
}
|
||||
let fields: Vec<Object> = leaf_ids.iter().map(|&id| Object::Reference(id)).collect();
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => fields,
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.len() <= MAX_FORM_FIELD_NODES);
|
||||
assert!(items.len() >= MAX_FORM_FIELD_NODES - 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn duplicate_and_invalid_kids_entries_stop_at_budget() {
|
||||
// Duplicate references and non-reference junk never grow `visited`, so
|
||||
// without charging examined entries against the budget an oversized
|
||||
// array of them would iterate to completion. The walk must still
|
||||
// terminate and extract the single real leaf exactly once.
|
||||
let mut doc = Document::new();
|
||||
let leaf_id = doc.new_object_id();
|
||||
doc.set_object(
|
||||
leaf_id,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"V" => Object::string_literal("v"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
// A `/Kids` array far wider than the budget: half duplicate references
|
||||
// to the same leaf, half invalid (null) entries.
|
||||
let mut kids: Vec<Object> = Vec::new();
|
||||
for i in 0..(MAX_FORM_FIELD_NODES * 2) {
|
||||
if i % 2 == 0 {
|
||||
kids.push(Object::Reference(leaf_id));
|
||||
} else {
|
||||
kids.push(Object::Null);
|
||||
}
|
||||
}
|
||||
let root_id = doc.add_object(dictionary! {
|
||||
"T" => Object::string_literal("root"),
|
||||
"Kids" => kids,
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(root_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert_eq!(items.len(), 1);
|
||||
}
|
||||
}
|
||||
|
||||
+3
-40
@@ -4566,13 +4566,9 @@ pub fn glyph_to_char(name: &str) -> Option<char> {
|
||||
}
|
||||
}
|
||||
|
||||
// Try to parse uniXXXX format.
|
||||
// Use `get` rather than a byte-length check + slice: `name` can contain
|
||||
// non-ASCII bytes (e.g. U+FFFD from lossy UTF-8 decoding of an attacker
|
||||
// controlled /Differences name), so byte index 7 may not be a char
|
||||
// boundary and `&name[3..7]` would panic.
|
||||
if let Some(hex) = name.strip_prefix("uni").and_then(|rest| rest.get(..4)) {
|
||||
if let Ok(code) = u32::from_str_radix(hex, 16) {
|
||||
// Try to parse uniXXXX format
|
||||
if name.starts_with("uni") && name.len() >= 7 {
|
||||
if let Ok(code) = u32::from_str_radix(&name[3..7], 16) {
|
||||
// Strip PUA F000 offset: uniF0XX → U+00XX (Windows Symbol encoding convention)
|
||||
let code = if (0xF000..=0xF0FF).contains(&code) {
|
||||
code - 0xF000
|
||||
@@ -4592,36 +4588,3 @@ pub fn glyph_to_char(name: &str) -> Option<char> {
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn uni_hex_parsing() {
|
||||
assert_eq!(glyph_to_char("uni0041"), Some('A'));
|
||||
assert_eq!(glyph_to_char("uni00e9"), Some('\u{00e9}'));
|
||||
// PUA F0xx symbol-encoding offset is stripped.
|
||||
assert_eq!(glyph_to_char("uniF041"), Some('A'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u_hex_parsing() {
|
||||
assert_eq!(glyph_to_char("u0041"), Some('A'));
|
||||
assert_eq!(glyph_to_char("u1F600"), Some('\u{1F600}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_ascii_uni_name_does_not_panic() {
|
||||
// A crafted /Differences name like `/uni#80#80#80#80` decodes via
|
||||
// from_utf8_lossy into "uni" followed by four U+FFFD replacements.
|
||||
// Byte index 7 lands mid-character, so a naive `&name[3..7]` slice
|
||||
// would panic. It must be handled gracefully instead.
|
||||
let crafted = format!("uni{0}{0}{0}{0}", '\u{FFFD}');
|
||||
assert_eq!(glyph_to_char(&crafted), None);
|
||||
|
||||
// Assorted non-ASCII bytes right after the "uni" prefix.
|
||||
assert_eq!(glyph_to_char("uni\u{FFFD}bc"), None);
|
||||
assert_eq!(glyph_to_char("uni\u{00e9}00"), None);
|
||||
}
|
||||
}
|
||||
|
||||
-74
@@ -657,80 +657,6 @@ pub fn extract_pages_markdown<P: AsRef<Path>>(
|
||||
extract_pages_markdown_mem(&buffer, pages)
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Structure-tree element extraction (tagged PDFs)
|
||||
// =========================================================================
|
||||
|
||||
/// One structure-tree element reference from a tagged PDF, resolved to a
|
||||
/// page and Marked Content ID.
|
||||
///
|
||||
/// Join `(page, mcid)` against [`TextItem::page`] / [`TextItem::mcid`] from
|
||||
/// [`extract_text_with_positions`] to attach semantic roles (heading levels,
|
||||
/// paragraphs, table cells, …) to extracted text.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct StructureElement {
|
||||
/// 1-indexed page number (matches [`TextItem::page`]).
|
||||
pub page: u32,
|
||||
/// Marked Content ID from the page's content stream (matches
|
||||
/// [`TextItem::mcid`]).
|
||||
pub mcid: i64,
|
||||
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
|
||||
/// Custom tags are resolved through the document's `/RoleMap`; tags
|
||||
/// with no standard mapping are returned verbatim.
|
||||
pub role: String,
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from a tagged PDF in memory.
|
||||
///
|
||||
/// Parses `/StructTreeRoot` (when present) and returns one entry per
|
||||
/// marked-content reference, resolved to its 1-indexed page, MCID, and
|
||||
/// structure type name. Returns an empty list when the PDF is not tagged.
|
||||
///
|
||||
/// Pass `Some(&[...])` with 1-indexed page numbers (matching
|
||||
/// [`TextItem::page`]) to restrict output to those pages; pass `None` for
|
||||
/// the whole document. Entries are sorted by `(page, mcid)`.
|
||||
pub fn extract_structure_elements_mem(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<Vec<StructureElement>, PdfError> {
|
||||
validate_pdf_bytes(buffer)?;
|
||||
let (doc, _page_count) = load_document_from_mem(buffer)?;
|
||||
let Some(tree) = structure_tree::StructTree::from_doc(&doc) else {
|
||||
return Ok(Vec::new());
|
||||
};
|
||||
let page_ids = doc.get_pages();
|
||||
let roles = tree.mcid_to_roles(&page_ids);
|
||||
|
||||
let page_filter: Option<HashSet<u32>> = pages.map(|p| p.iter().copied().collect());
|
||||
let mut elements: Vec<StructureElement> = roles
|
||||
.into_iter()
|
||||
.filter(|(page, _)| page_filter.as_ref().is_none_or(|f| f.contains(page)))
|
||||
.flat_map(|(page, mcids)| {
|
||||
mcids.into_iter().map(move |(mcid, role)| StructureElement {
|
||||
page,
|
||||
mcid,
|
||||
role: role.name().to_string(),
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
elements.sort_unstable_by_key(|e| (e.page, e.mcid));
|
||||
Ok(elements)
|
||||
}
|
||||
|
||||
/// Path-based wrapper for [`extract_structure_elements_mem`].
|
||||
///
|
||||
/// Reads the PDF from disk and extracts structure-tree element references.
|
||||
/// Pass `None` for `pages` to return the whole document, or `Some(&[...])`
|
||||
/// to restrict to specific 1-indexed pages.
|
||||
pub fn extract_structure_elements<P: AsRef<Path>>(
|
||||
path: P,
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<Vec<StructureElement>, PdfError> {
|
||||
validate_pdf_file(&path)?;
|
||||
let buffer = std::fs::read(path.as_ref())?;
|
||||
extract_structure_elements_mem(&buffer, pages)
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Region-based text extraction (for hybrid OCR pipelines)
|
||||
// =========================================================================
|
||||
|
||||
@@ -150,6 +150,61 @@ pub(crate) fn merge_heading_lines(
|
||||
/// Merge drop caps with the appropriate line.
|
||||
/// A drop cap is a single large letter at the start of a paragraph.
|
||||
/// Due to PDF coordinate sorting, the drop cap may appear AFTER the line it belongs to.
|
||||
/// True when the text ends a sentence, as opposed to merely ending in a
|
||||
/// period. An abbreviation or list marker ("e.g.", "Fig.", "Mr.", "1.")
|
||||
/// closes with a period mid-sentence, so treating those as paragraph
|
||||
/// boundaries would let a drop cap be prepended to a continuation.
|
||||
fn ends_sentence(text: &str) -> bool {
|
||||
let t = text.trim_end();
|
||||
if t.ends_with(['!', '?']) {
|
||||
return true;
|
||||
}
|
||||
let Some(stripped) = t.strip_suffix('.') else {
|
||||
return false;
|
||||
};
|
||||
let last = stripped.split_whitespace().next_back().unwrap_or("");
|
||||
if last.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Abbreviations carry an internal period between very short segments
|
||||
// ("e.g.", "i.e.", "U.S."). Domains and decimals have the same shape but
|
||||
// longer or numeric segments ("example.com.", "3.14."), and those end
|
||||
// sentences perfectly well, so require every segment to be short and
|
||||
// alphabetic before reading the internal period as an abbreviation.
|
||||
if last.contains('.')
|
||||
&& last
|
||||
.split('.')
|
||||
.filter(|seg| !seg.is_empty())
|
||||
// Characters, not bytes: a two-letter non-ASCII abbreviation
|
||||
// ("т.е.", "ú.d.") measures four or more bytes and would
|
||||
// otherwise be read as a completed sentence.
|
||||
.all(|seg| seg.chars().count() <= 2 && seg.chars().all(char::is_alphabetic))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// Enumerators stand alone on their line ("1.", "ii.", "IV."). A number
|
||||
// or numeral in the tail of a sentence does not — "published in 2020.",
|
||||
// "He scored 5." and "after World War II." all end sentences, and
|
||||
// treating them as markers would block a legitimate drop-cap merge.
|
||||
if stripped.split_whitespace().count() == 1 {
|
||||
let is_numeric = last.chars().all(|c| c.is_ascii_digit());
|
||||
let is_roman = last
|
||||
.chars()
|
||||
.all(|c| matches!(c.to_ascii_uppercase(), 'I' | 'V' | 'X' | 'L' | 'C'));
|
||||
if is_numeric || is_roman {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
const ABBREVIATIONS: &[&str] = &[
|
||||
"Fig", "No", "Mr", "Mrs", "Ms", "Dr", "St", "vs", "etc", "al", "Ed", "Eq", "Ch", "pp",
|
||||
"Vol", "cf", "Prof", "Inc", "Ltd", "Jr", "Sr",
|
||||
];
|
||||
!ABBREVIATIONS.iter().any(|a| a.eq_ignore_ascii_case(last))
|
||||
}
|
||||
|
||||
pub(crate) fn merge_drop_caps(lines: Vec<TextLine>, base_size: f32) -> Vec<TextLine> {
|
||||
let mut result: Vec<TextLine> = Vec::with_capacity(lines.len());
|
||||
|
||||
@@ -169,6 +224,169 @@ pub(crate) fn merge_drop_caps(lines: Vec<TextLine>, base_size: f32) -> Vec<TextL
|
||||
.map(|c| c.is_uppercase())
|
||||
.unwrap_or(false);
|
||||
|
||||
// Embedded drop cap: a two-line cap's baseline aligns with the
|
||||
// paragraph's SECOND line, so Y-grouping puts the glyph at the start
|
||||
// of that line rather than on a line of its own. Left there it
|
||||
// surfaces mid-sentence once the paragraph is joined — Shannon's
|
||||
// "A Mathematical Theory of Communication" reads "...which exchange
|
||||
// T bandwidth for signal-to-noise ratio...". Detect it, prepend the
|
||||
// character to the paragraph's first line, and drop it from this one.
|
||||
//
|
||||
// The size gate is 1.8x rather than 2.5x because bitmap (Type3) caps
|
||||
// report their glyph bbox rather than the em box, so a two-line cap
|
||||
// can measure as little as ~1.9x the body size.
|
||||
if line.items.len() > 1 {
|
||||
let first = &line.items[0];
|
||||
// The remainder must be a substantive body run: a lone label or
|
||||
// math fragment beside a large glyph is not a drop-cap paragraph.
|
||||
let rest_letters: usize = line.items[1..]
|
||||
.iter()
|
||||
.map(|i| i.text.chars().filter(|c| c.is_alphabetic()).count())
|
||||
.sum();
|
||||
let is_embedded_cap = first.font_size >= base_size * 1.8
|
||||
&& first.text.trim().chars().count() == 1
|
||||
&& first
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(char::is_uppercase)
|
||||
&& line.items[1..]
|
||||
.iter()
|
||||
.all(|i| i.font_size < base_size * 1.5)
|
||||
&& line.items[1..].iter().any(|i| i.x > first.x)
|
||||
&& rest_letters >= 8;
|
||||
if is_embedded_cap {
|
||||
let drop_char = first.text.trim().chars().next().unwrap();
|
||||
let cap_x = first.x;
|
||||
let line_y = line.y;
|
||||
// Text on the cap's own line, pushed right to clear the glyph.
|
||||
let rest_x = line.items[1].x;
|
||||
|
||||
// Walk up the run of lines the cap has indented. A drop cap
|
||||
// pushes every line it covers to the right of the glyph, so
|
||||
// the paragraph's first line is the TOPMOST line sharing that
|
||||
// indent — however many lines the cap spans. Using the indent
|
||||
// rather than the cap's font size is what makes this work for
|
||||
// three- and four-line initials as well as two-line ones;
|
||||
// deriving a line count from the em size does not survive
|
||||
// contact with real documents, where 36-47pt initials sit
|
||||
// over 11-14pt leading.
|
||||
//
|
||||
// A cap in a different column has no such run (its neighbours
|
||||
// sit at an unrelated x), so it is left alone — which is
|
||||
// correct when the cap's own line already carries the rest of
|
||||
// the word.
|
||||
const INDENT_TOLERANCE: f32 = 2.0;
|
||||
const MAX_CAP_LINES: usize = 8;
|
||||
let max_step = base_size * 2.5;
|
||||
let mut target_idx = result.len();
|
||||
let mut expected_y = line_y;
|
||||
while target_idx > 0 && result.len() - target_idx < MAX_CAP_LINES {
|
||||
let cand = &result[target_idx - 1];
|
||||
let step = cand.y - expected_y;
|
||||
let shares_indent = cand
|
||||
.items
|
||||
.first()
|
||||
.is_some_and(|i| (i.x - rest_x).abs() <= INDENT_TOLERANCE);
|
||||
if cand.page != line.page || step <= 0.0 || step > max_step || !shares_indent {
|
||||
break;
|
||||
}
|
||||
expected_y = cand.y;
|
||||
target_idx -= 1;
|
||||
}
|
||||
|
||||
// The topmost line of the run is the paragraph's first line.
|
||||
// The line above THAT tells us whether it starts a paragraph.
|
||||
let before_target = target_idx
|
||||
.checked_sub(1)
|
||||
.and_then(|i| result.get(i))
|
||||
.filter(|l| l.page == line.page)
|
||||
.map(|l| (l.text().trim_end().to_string(), l.y));
|
||||
// Leading within the run: the step from the target down to the
|
||||
// next line of the paragraph, which is the cap's own line when
|
||||
// the run is a single line.
|
||||
let run_step = result
|
||||
.get(target_idx)
|
||||
.map(|t| {
|
||||
let below_y = result.get(target_idx + 1).map_or(line_y, |b| b.y);
|
||||
t.y - below_y
|
||||
})
|
||||
.unwrap_or(0.0);
|
||||
let step_for_gap = if run_step > 0.0 {
|
||||
run_step
|
||||
} else {
|
||||
base_size * 1.2
|
||||
};
|
||||
|
||||
let target = (target_idx < result.len())
|
||||
.then(|| &mut result[target_idx])
|
||||
.filter(|prev| {
|
||||
let prev_text = prev.text();
|
||||
let prev_trimmed = prev_text.trim();
|
||||
// A hyphen on the line above means the target resumes
|
||||
// a split word, so it continues a paragraph rather
|
||||
// than starting one (polkuja_ylakoulu: "ylakou-" +
|
||||
// "lulaisten").
|
||||
//
|
||||
// Case cannot serve as a continuation signal here: the
|
||||
// target legitimately starts lowercase, because the
|
||||
// cap removes the word's first letter and leaves
|
||||
// "ver the course..." for "Over".
|
||||
let continues_previous = before_target
|
||||
.as_ref()
|
||||
.is_some_and(|(b, _)| b.ends_with('-'));
|
||||
// The target must START a paragraph: extra leading
|
||||
// above it, a completed sentence on the line above, or
|
||||
// nothing above it at all.
|
||||
let starts_paragraph = match before_target.as_ref() {
|
||||
None => true,
|
||||
Some((text, y)) => {
|
||||
y - prev.y > step_for_gap * 1.15 || ends_sentence(text)
|
||||
}
|
||||
};
|
||||
!continues_previous
|
||||
&& starts_paragraph
|
||||
&& prev.page == line.page
|
||||
&& prev.y > line_y
|
||||
// Indented past the cap glyph, not merely to its
|
||||
// right by an arbitrary amount.
|
||||
&& prev
|
||||
.items
|
||||
.first()
|
||||
.is_some_and(|i| i.x > cap_x && i.x - cap_x <= first.font_size * 2.0)
|
||||
// Body text, so headings, labels and table
|
||||
// fragments are never rewritten.
|
||||
&& prev_trimmed
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(char::is_alphabetic)
|
||||
&& prev_trimmed.chars().filter(|c| c.is_alphabetic()).count() >= 8
|
||||
});
|
||||
if let Some(prev_line) = target {
|
||||
if let Some(first_item) = prev_line.items.first_mut() {
|
||||
// A mid-word cap ("T" + "HE recent") joins directly.
|
||||
// Leading whitespace only marks a word boundary when
|
||||
// the cap is itself a single-letter word, since the
|
||||
// paragraph's indent can also arrive as whitespace.
|
||||
const SINGLE_LETTER_WORDS: &[char] = &['A', 'I', 'O', 'U', 'Y', 'E'];
|
||||
let had_leading_ws = first_item.text.starts_with(char::is_whitespace)
|
||||
&& SINGLE_LETTER_WORDS.contains(&drop_char);
|
||||
let rest = first_item.text.trim_start().to_string();
|
||||
first_item.text = if had_leading_ws {
|
||||
format!("{} {}", drop_char, rest)
|
||||
} else {
|
||||
format!("{}{}", drop_char, rest)
|
||||
};
|
||||
}
|
||||
let mut line = line.clone();
|
||||
line.items.remove(0);
|
||||
result.push(line);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if is_drop_cap {
|
||||
let drop_char = trimmed.chars().next().unwrap();
|
||||
|
||||
@@ -598,6 +816,314 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn make_item_at(text: &str, font_size: f32, x: f32) -> TextItem {
|
||||
let mut item = make_item(text, font_size, None);
|
||||
item.x = x;
|
||||
item.width = text.len() as f32 * font_size * 0.5;
|
||||
item
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_moves_to_paragraph_start() {
|
||||
// A two-line cap baseline-aligns with the paragraph's SECOND line,
|
||||
// so it lands as that line's first item (Shannon entropy.pdf p.1).
|
||||
let first_line = TextLine {
|
||||
items: vec![make_item_at(
|
||||
"HE recent development which exchange",
|
||||
10.0,
|
||||
90.0,
|
||||
)],
|
||||
// 16pt baseline step under a 25pt cap: a genuine two-line cap.
|
||||
y: 716.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let second_line = TextLine {
|
||||
items: vec![
|
||||
make_item_at("T", 25.0, 72.0),
|
||||
make_item_at("bandwidth for signal-to-noise ratio", 10.0, 90.0),
|
||||
],
|
||||
y: 700.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let result = merge_drop_caps(vec![first_line, second_line], 10.0);
|
||||
assert_eq!(result.len(), 2);
|
||||
assert!(
|
||||
result[0].text().starts_with("THE recent"),
|
||||
"cap should prepend to the paragraph start: {}",
|
||||
result[0].text()
|
||||
);
|
||||
assert!(
|
||||
result[1].text().starts_with("bandwidth"),
|
||||
"cap must be removed from the second line: {}",
|
||||
result[1].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_walks_a_multi_line_initial_to_the_paragraph_start() {
|
||||
// A 47pt initial over 13pt leading covers four lines, so the
|
||||
// paragraph's first line is three lines above the cap rather than
|
||||
// immediately above it (polkuja_ylakoulu). The indented run, not the
|
||||
// cap's em size, is what locates it.
|
||||
let mut lines = vec![TextLine {
|
||||
items: vec![make_item_at("Previous paragraph ends here.", 10.0, 72.0)],
|
||||
y: 766.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
}];
|
||||
for (i, text) in [
|
||||
"rilaiset mediasisallot ovat tarkea osa",
|
||||
"useimpien ylakoululaisten elamaa ja",
|
||||
"muuta tekstia jatkuu tassa viela",
|
||||
]
|
||||
.iter()
|
||||
.enumerate()
|
||||
{
|
||||
lines.push(TextLine {
|
||||
items: vec![make_item_at(text, 10.0, 90.0)],
|
||||
y: 753.0 - 13.0 * i as f32,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
});
|
||||
}
|
||||
lines.push(TextLine {
|
||||
items: vec![
|
||||
make_item_at("E", 47.0, 72.0),
|
||||
make_item_at("loppuosa tekstista tassa", 10.0, 90.0),
|
||||
],
|
||||
y: 714.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
});
|
||||
|
||||
let result = merge_drop_caps(lines, 10.0);
|
||||
assert!(
|
||||
result[1].text().starts_with("Erilaiset"),
|
||||
"cap belongs on the topmost line of the indented run: {}",
|
||||
result[1].text()
|
||||
);
|
||||
assert!(
|
||||
result[2].text().starts_with("useimpien"),
|
||||
"intervening run lines must be untouched: {}",
|
||||
result[2].text()
|
||||
);
|
||||
assert!(
|
||||
result[4].text().starts_with("loppuosa"),
|
||||
"cap must be removed from its own line: {}",
|
||||
result[4].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_ignores_non_paragraph_neighbours() {
|
||||
// Same geometry, but the preceding line is a short label rather than
|
||||
// body text, so it must not be rewritten.
|
||||
let label = TextLine {
|
||||
items: vec![make_item_at("Fig. 2", 10.0, 90.0)],
|
||||
y: 716.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let second_line = TextLine {
|
||||
items: vec![
|
||||
make_item_at("T", 25.0, 72.0),
|
||||
make_item_at("bandwidth for signal-to-noise ratio", 10.0, 90.0),
|
||||
],
|
||||
y: 700.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let result = merge_drop_caps(vec![label, second_line], 10.0);
|
||||
assert_eq!(result[0].text().trim(), "Fig. 2");
|
||||
assert!(
|
||||
result[1].text().starts_with('T'),
|
||||
"cap stays put: {}",
|
||||
result[1].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_keeps_a_space_for_standalone_word_caps() {
|
||||
// Leading whitespace on the paragraph's first item marks the cap as
|
||||
// a word of its own rather than the first letter of one.
|
||||
let mut lead = make_item_at("long time ago in a galaxy far away", 10.0, 90.0);
|
||||
lead.text = " long time ago in a galaxy far away".to_string();
|
||||
let first_line = TextLine {
|
||||
items: vec![lead],
|
||||
y: 716.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let second_line = TextLine {
|
||||
items: vec![
|
||||
make_item_at("A", 25.0, 72.0),
|
||||
make_item_at("continued here with more body text", 10.0, 90.0),
|
||||
],
|
||||
y: 700.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let result = merge_drop_caps(vec![first_line, second_line], 10.0);
|
||||
assert!(
|
||||
result[0].text().starts_with("A long time ago"),
|
||||
"standalone-word cap keeps one space: {}",
|
||||
result[0].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_skips_hyphenation_continuation_targets() {
|
||||
// The line above the RUN ends on a hyphen, so the run's topmost line
|
||||
// resumes a split word rather than starting a paragraph. It sits at
|
||||
// the paragraph margin (x=72), outside the cap's indent, so it is not
|
||||
// part of the run itself.
|
||||
let split_word = TextLine {
|
||||
items: vec![make_item_at(
|
||||
"mediasisallot ovat osa useimpien ylakou-",
|
||||
10.0,
|
||||
72.0,
|
||||
)],
|
||||
y: 728.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let run_top = TextLine {
|
||||
items: vec![make_item_at(
|
||||
"lulaisten elamaa ja muuta tekstia",
|
||||
10.0,
|
||||
90.0,
|
||||
)],
|
||||
y: 714.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let cap_line = TextLine {
|
||||
items: vec![
|
||||
make_item_at("E", 25.0, 72.0),
|
||||
make_item_at("jatkuu tassa lisaa leipatekstia", 10.0, 90.0),
|
||||
],
|
||||
y: 700.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let result = merge_drop_caps(vec![split_word, run_top, cap_line], 10.0);
|
||||
assert!(
|
||||
result[1].text().starts_with("lulaisten"),
|
||||
"a run resuming a split word must not receive the cap: {}",
|
||||
result[1].text()
|
||||
);
|
||||
assert!(
|
||||
result[2].text().starts_with('E'),
|
||||
"cap stays put when no valid target exists: {}",
|
||||
result[2].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_indent_is_not_a_word_boundary() {
|
||||
// The paragraph's first line is indented to clear the cap, and that
|
||||
// indent can arrive as leading whitespace. A mid-word cap must still
|
||||
// join directly — "T HE recent" would be the defect this fixes.
|
||||
let mut lead = make_item_at("HE recent development and more body text", 10.0, 90.0);
|
||||
lead.text = " HE recent development and more body text".to_string();
|
||||
let first_line = TextLine {
|
||||
items: vec![lead],
|
||||
y: 716.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let cap_line = TextLine {
|
||||
items: vec![
|
||||
make_item_at("T", 25.0, 72.0),
|
||||
make_item_at("bandwidth for signal-to-noise ratio", 10.0, 90.0),
|
||||
],
|
||||
y: 700.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let result = merge_drop_caps(vec![first_line, cap_line], 10.0);
|
||||
assert!(
|
||||
result[0].text().starts_with("THE recent"),
|
||||
"indent must not be read as a word boundary: {}",
|
||||
result[0].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ends_sentence_rejects_abbreviations_and_markers() {
|
||||
use super::ends_sentence;
|
||||
assert!(ends_sentence("This completes the thought."));
|
||||
assert!(ends_sentence("Is that so?"));
|
||||
assert!(ends_sentence("Stop!"));
|
||||
// Periods that do not end a sentence.
|
||||
assert!(!ends_sentence("as shown in Fig."));
|
||||
assert!(!ends_sentence("see e.g."));
|
||||
// Non-ASCII abbreviations. The two-CHARACTER segment is the case
|
||||
// that distinguishes a character count from a byte count: "пр" is
|
||||
// 2 chars but 4 bytes, so a byte-based bound would reject it and
|
||||
// read the line as a completed sentence.
|
||||
assert!(!ends_sentence("и т.пр."));
|
||||
assert!(!ends_sentence("см. т.е."));
|
||||
assert!(!ends_sentence("napr. ú.d."));
|
||||
assert!(!ends_sentence("reviewed by Dr."));
|
||||
// Standalone enumerators, any case.
|
||||
assert!(!ends_sentence("1."));
|
||||
assert!(!ends_sentence("IV."));
|
||||
assert!(!ends_sentence("ii."));
|
||||
assert!(!ends_sentence("xii."));
|
||||
// Numbers and numerals that genuinely end a sentence must count,
|
||||
// or a legitimate drop-cap merge is blocked.
|
||||
assert!(ends_sentence("The paper was published in 2020."));
|
||||
assert!(ends_sentence("He scored 5."));
|
||||
assert!(ends_sentence("after World War II."));
|
||||
assert!(ends_sentence("the constant equals 3.14."));
|
||||
assert!(ends_sentence("documented at example.com."));
|
||||
assert!(!ends_sentence("a trailing clause with no period"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_drop_cap_allows_first_paragraph_on_a_new_page() {
|
||||
// The line two back is on the previous page, so its y is unrelated
|
||||
// and must not be used as leading evidence.
|
||||
let prev_page_tail = TextLine {
|
||||
items: vec![make_item_at(
|
||||
"tail of the previous page body text",
|
||||
10.0,
|
||||
90.0,
|
||||
)],
|
||||
y: 90.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let first_line = TextLine {
|
||||
items: vec![make_item_at(
|
||||
"HE recent development which exchange",
|
||||
10.0,
|
||||
90.0,
|
||||
)],
|
||||
y: 716.0,
|
||||
page: 2,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let cap_line = TextLine {
|
||||
items: vec![
|
||||
make_item_at("T", 25.0, 72.0),
|
||||
make_item_at("bandwidth for signal-to-noise ratio", 10.0, 90.0),
|
||||
],
|
||||
y: 700.0,
|
||||
page: 2,
|
||||
adaptive_threshold: 0.10,
|
||||
};
|
||||
let result = merge_drop_caps(vec![prev_page_tail, first_line, cap_line], 10.0);
|
||||
assert!(
|
||||
result[1].text().starts_with("THE recent"),
|
||||
"a page break must not suppress the merge: {}",
|
||||
result[1].text()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_struct_tree_headings() {
|
||||
// Two consecutive lines tagged as H2 via struct tree, same font size as body
|
||||
|
||||
@@ -271,12 +271,6 @@ pub struct PyTextItem {
|
||||
pub is_strikeout: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator, None
|
||||
/// when the text is not part of marked content. Join with the
|
||||
/// (page, mcid) pairs from extract_structure_elements to attach
|
||||
/// structure-tree roles (headings, paragraphs, ...) in tagged PDFs.
|
||||
#[pyo3(get)]
|
||||
pub mcid: Option<i64>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -292,32 +286,6 @@ impl PyTextItem {
|
||||
}
|
||||
}
|
||||
|
||||
/// One structure-tree element reference from a tagged PDF.
|
||||
#[pyclass(name = "StructureElement")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyStructureElement {
|
||||
/// 1-indexed page number (matches TextItem.page).
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Marked Content ID from the page's content stream (matches
|
||||
/// TextItem.mcid).
|
||||
#[pyo3(get)]
|
||||
pub mcid: i64,
|
||||
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", ...).
|
||||
#[pyo3(get)]
|
||||
pub role: String,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyStructureElement {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"StructureElement(page={}, mcid={}, role='{}')",
|
||||
self.page, self.mcid, self.role
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -388,18 +356,6 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
mcid: item.mcid,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_structure_elements(elements: Vec<crate::StructureElement>) -> Vec<PyStructureElement> {
|
||||
elements
|
||||
.into_iter()
|
||||
.map(|e| PyStructureElement {
|
||||
page: e.page,
|
||||
mcid: e.mcid,
|
||||
role: e.role,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
@@ -657,48 +613,6 @@ fn extract_pages_markdown_bytes(
|
||||
Ok(to_py_pages_result(result))
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from a tagged PDF file.
|
||||
///
|
||||
/// Parses the document's structure tree (when present) and returns one
|
||||
/// entry per marked-content reference, resolved to its 1-indexed page,
|
||||
/// MCID, and structure type name ("H1".."H6", "P", "Table", ...). Returns
|
||||
/// an empty list when the PDF is not tagged.
|
||||
///
|
||||
/// Join (page, mcid) against the page/mcid attributes from
|
||||
/// [`extract_text_with_positions`] to attach heading levels and other
|
||||
/// semantic roles to extracted text.
|
||||
///
|
||||
/// Args:
|
||||
/// path: Path to the PDF file.
|
||||
/// pages: Optional list of 1-indexed pages (matching TextItem.page).
|
||||
/// When None (default), the whole document is returned.
|
||||
///
|
||||
/// Returns:
|
||||
/// List of StructureElement sorted by (page, mcid).
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn extract_structure_elements(
|
||||
path: &str,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<Vec<PyStructureElement>> {
|
||||
let elements = crate::extract_structure_elements(path, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(convert_structure_elements(elements))
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from tagged PDF bytes.
|
||||
///
|
||||
/// See [`extract_structure_elements`] for details.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn extract_structure_elements_bytes(
|
||||
data: &[u8],
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<Vec<PyStructureElement>> {
|
||||
let elements =
|
||||
crate::extract_structure_elements_mem(data, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(convert_structure_elements(elements))
|
||||
}
|
||||
|
||||
/// Python module definition.
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
@@ -706,7 +620,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPageOcrReasons>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyStructureElement>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
m.add_class::<PyPageRegionTexts>()?;
|
||||
m.add_class::<PyPageMarkdown>()?;
|
||||
@@ -721,8 +634,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_function(wrap_pyfunction!(extract_text_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_structure_elements, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_structure_elements_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
|
||||
|
||||
+38
-819
File diff suppressed because it is too large
Load Diff
+1
-19
@@ -594,7 +594,7 @@ fn hex_to_unicode_string(hex: &str) -> Option<String> {
|
||||
|
||||
let bytes: Option<Vec<u8>> = (0..hex.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(hex.get(i..i + 2)?, 16).ok())
|
||||
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
|
||||
.collect();
|
||||
let bytes = bytes?;
|
||||
|
||||
@@ -2606,24 +2606,6 @@ endcmap
|
||||
assert_eq!(cmap.lookup(0x0025), Some("B".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_hex_to_unicode_non_ascii_no_panic() {
|
||||
// A destination containing a multi-byte char makes the byte length even
|
||||
// while a byte offset can land inside a char. Slicing must not panic;
|
||||
// it should be rejected gracefully.
|
||||
assert_eq!(hex_to_unicode_string("XéY"), None);
|
||||
assert_eq!(hex_to_unicode_string("\u{fffd}0"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_non_ascii_destination_no_panic() {
|
||||
// Crafted /ToUnicode CMap: a non-hex, non-ASCII destination previously
|
||||
// triggered a char-boundary panic in hex_to_unicode_string.
|
||||
let cmap_content = "beginbfchar <0041> <XéY> endbfchar";
|
||||
// Must not panic; the malformed entry is simply skipped.
|
||||
let _ = ToUnicodeCMap::parse(cmap_content.as_bytes());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_1byte() {
|
||||
// This is the pattern that caused the CJK bug: codespace is <0000><FFFF>
|
||||
|
||||
@@ -1429,77 +1429,6 @@ fn test_firecrawl_tagged_pdf_struct_tree() {
|
||||
assert_eq!(fence_count % 2, 0, "Code fences should be balanced");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_tagged_pdf_text_items_carry_mcid() {
|
||||
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
|
||||
assert!(
|
||||
items.iter().any(|i| i.mcid.is_some()),
|
||||
"Tagged PDF text items should carry Marked Content IDs"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_structure_elements_tagged_pdf() {
|
||||
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
|
||||
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
|
||||
assert!(!elements.is_empty(), "Tagged PDF should yield elements");
|
||||
assert!(
|
||||
elements.iter().any(|e| e.role == "H1"),
|
||||
"Should surface H1 heading roles"
|
||||
);
|
||||
assert!(
|
||||
elements.iter().all(|e| !e.role.is_empty()),
|
||||
"Every element should carry a role name"
|
||||
);
|
||||
|
||||
// Sorted by (page, mcid) for deterministic output
|
||||
assert!(
|
||||
elements
|
||||
.windows(2)
|
||||
.all(|w| (w[0].page, w[0].mcid) <= (w[1].page, w[1].mcid)),
|
||||
"Elements should be sorted by (page, mcid)"
|
||||
);
|
||||
|
||||
// The advertised join: (page, mcid) pairs must line up with the
|
||||
// mcid-carrying TextItems from positioned extraction, and joining the
|
||||
// H1 entries must recover non-empty heading text.
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
|
||||
let h1_refs: std::collections::HashSet<(u32, i64)> = elements
|
||||
.iter()
|
||||
.filter(|e| e.role == "H1")
|
||||
.map(|e| (e.page, e.mcid))
|
||||
.collect();
|
||||
let h1_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.mcid.is_some_and(|mcid| h1_refs.contains(&(i.page, mcid))))
|
||||
.map(|i| i.text.as_str())
|
||||
.collect();
|
||||
assert!(
|
||||
!h1_text.trim().is_empty(),
|
||||
"Joining H1 structure elements to text items should recover heading text"
|
||||
);
|
||||
|
||||
// Page filter is 1-indexed (matching TextItem.page) and equals the
|
||||
// corresponding subset of the full document result.
|
||||
let page1 = pdf_inspector::extract_structure_elements_mem(&buf, Some(&[1])).unwrap();
|
||||
assert!(!page1.is_empty(), "Page 1 should have elements");
|
||||
assert!(page1.iter().all(|e| e.page == 1));
|
||||
let full_page1_count = elements.iter().filter(|e| e.page == 1).count();
|
||||
assert_eq!(page1.len(), full_page1_count);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_structure_elements_untagged_pdf_empty() {
|
||||
let buf = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
|
||||
assert!(
|
||||
elements.is_empty(),
|
||||
"Untagged PDF should yield no structure elements, got {:?}",
|
||||
elements
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_identity_h_no_tounicode_suppresses_garbage() {
|
||||
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
|
||||
|
||||
@@ -1,16 +1,12 @@
|
||||
Reprinted with corrections from *The Bell System Technical Journal,* Vol. 27, pp. 379–423, 623–656, July, October, 1948.
|
||||
|
||||
## A Mathematical Theory of Communication
|
||||
# A Mathematical Theory of Communication
|
||||
|
||||
### By C. E. SHANNON
|
||||
## By C. E. SHANNON
|
||||
|
||||
INTRODUCTION
|
||||
|
||||
HE recent development of various methods of modulation such as PCM and PPM which exchange
|
||||
|
||||
# Tbandwidth for signal-to-noise ratio has intensified the interest in a general theory of communication. A
|
||||
|
||||
basis for such a theory is contained in the important papers of Nyquist¹ and Hartley² on this subject. In the present paper we will extend the theory to include a number of new factors, in particular the effect of noise in the channel, and the savings possible due to the statistical structure of the original message and due to the nature of the final destination of the information. The fundamental problem of communication is that of reproducing at one point either exactly or ap- proximately a message selected at another point. Frequently the messages have *meaning*; that is they refer to or are correlated according to some system with certain physical or conceptual entities. These semantic aspects of communication are irrelevant to the engineering problem. The significant aspect is that the actual message is one *selected from a set* of possible messages. The system must be designed to operate for each possible selection, not just the one which will actually be chosen since this is unknown at the time of design. If the number of messages in the set is finite then this number or any monotonic function of this number can be regarded as a measure of the information produced when one message is chosen from the set, all choices being equally likely. As was pointed out by Hartley the most natural choice is the logarithmic function. Although this definition must be generalized considerably when we consider the influence of the statistics of the message and when we have a continuous range of messages, we will in all cases use an essentially logarithmic measure. The logarithmic measure is more convenient for various reasons:
|
||||
THE recent development of various methods of modulation such as PCM and PPM which exchange bandwidth for signal-to-noise ratio has intensified the interest in a general theory of communication. A basis for such a theory is contained in the important papers of Nyquist¹ and Hartley² on this subject. In the present paper we will extend the theory to include a number of new factors, in particular the effect of noise in the channel, and the savings possible due to the statistical structure of the original message and due to the nature of the final destination of the information. The fundamental problem of communication is that of reproducing at one point either exactly or ap- proximately a message selected at another point. Frequently the messages have *meaning*; that is they refer to or are correlated according to some system with certain physical or conceptual entities. These semantic aspects of communication are irrelevant to the engineering problem. The significant aspect is that the actual message is one *selected from a set* of possible messages. The system must be designed to operate for each possible selection, not just the one which will actually be chosen since this is unknown at the time of design. If the number of messages in the set is finite then this number or any monotonic function of this number can be regarded as a measure of the information produced when one message is chosen from the set, all choices being equally likely. As was pointed out by Hartley the most natural choice is the logarithmic function. Although this definition must be generalized considerably when we consider the influence of the statistics of the message and when we have a continuous range of messages, we will in all cases use an essentially logarithmic measure. The logarithmic measure is more convenient for various reasons:
|
||||
|
||||
1. It is practically more useful. Parameters of engineering importance such as time, bandwidth, number of relays, etc., tend to vary linearly with the logarithm of the number of possibilities. For example, adding one relay to a group doubles the number of possible states of the relays. It adds 1 to the base 2 logarithm of this number. Doubling the time roughly squares the number of possible messages, or doubles the logarithm, etc.
|
||||
2. It is nearer to our intuitive feeling as to the proper measure. This is closely related to (1) since we in- tuitively measures entities by linear comparison with common standards. One feels, for example, that two punched cards should have twice the capacity of one for information storage, and two identical channels twice the capacity of one for transmitting information.
|
||||
|
||||
@@ -203,79 +203,6 @@ class TestExtractTextWithPositions:
|
||||
assert len(items) > 0
|
||||
assert all(item.page == 1 for item in items)
|
||||
|
||||
def test_mcid(self):
|
||||
# Untagged fixture: mcid is None or int, never anything else
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert all(item.mcid is None or isinstance(item.mcid, int) for item in items)
|
||||
# Tagged fixture: marked content carries MCIDs
|
||||
tagged = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("firecrawl_docs_tagged.pdf")
|
||||
)
|
||||
assert any(item.mcid is not None for item in tagged)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_structure_elements / extract_structure_elements_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractStructureElements:
|
||||
def test_tagged_file(self):
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("firecrawl_docs_tagged.pdf")
|
||||
)
|
||||
assert len(elements) > 0
|
||||
assert all(isinstance(e.page, int) for e in elements)
|
||||
assert all(isinstance(e.mcid, int) for e in elements)
|
||||
assert all(isinstance(e.role, str) and len(e.role) > 0 for e in elements)
|
||||
assert any(e.role == "H1" for e in elements)
|
||||
|
||||
def test_join_with_text_items(self):
|
||||
# (page, mcid) joins against extract_text_with_positions to recover
|
||||
# heading text
|
||||
path = fixture_path("firecrawl_docs_tagged.pdf")
|
||||
elements = pdf_inspector.extract_structure_elements(path)
|
||||
items = pdf_inspector.extract_text_with_positions(path)
|
||||
h1_refs = {(e.page, e.mcid) for e in elements if e.role == "H1"}
|
||||
h1_text = "".join(
|
||||
item.text
|
||||
for item in items
|
||||
if item.mcid is not None and (item.page, item.mcid) in h1_refs
|
||||
)
|
||||
assert len(h1_text.strip()) > 0
|
||||
|
||||
def test_with_pages(self):
|
||||
# pages filter is 1-indexed, matching TextItem.page
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("firecrawl_docs_tagged.pdf"), pages=[1]
|
||||
)
|
||||
assert len(elements) > 0
|
||||
assert all(e.page == 1 for e in elements)
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("firecrawl_docs_tagged.pdf")
|
||||
elements = pdf_inspector.extract_structure_elements_bytes(data)
|
||||
assert len(elements) > 0
|
||||
assert any(e.role == "H1" for e in elements)
|
||||
|
||||
def test_untagged_returns_empty(self):
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert elements == []
|
||||
|
||||
def test_repr(self):
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("firecrawl_docs_tagged.pdf")
|
||||
)
|
||||
assert "StructureElement" in repr(elements[0])
|
||||
|
||||
def test_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_structure_elements_bytes(b"not a pdf")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text_in_regions / extract_text_in_regions_bytes
|
||||
|
||||
Generated
+2
-2
@@ -724,7 +724,7 @@ checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.8"
|
||||
version = "0.1.7"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
@@ -740,7 +740,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.4"
|
||||
version = "0.1.3"
|
||||
dependencies = [
|
||||
"console_error_panic_hook",
|
||||
"js-sys",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.4"
|
||||
version = "0.1.3"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
|
||||
Reference in New Issue
Block a user