feat(bindings): expose OCR in Node and Python (#405)
* feat(bindings): expose selective OCR * fix(bindings): address review feedback
This commit is contained in:
@@ -134,10 +134,21 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
toolchain: stable
|
toolchain: stable
|
||||||
|
|
||||||
|
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||||
|
with:
|
||||||
|
python-version: '3.12'
|
||||||
|
|
||||||
|
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||||
|
with:
|
||||||
|
bun-version: latest
|
||||||
|
|
||||||
- name: Cache cargo
|
- name: Cache cargo
|
||||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||||
with:
|
with:
|
||||||
key: ocr-runtime
|
key: ocr-runtime
|
||||||
|
workspaces: |
|
||||||
|
. -> target
|
||||||
|
napi -> target
|
||||||
|
|
||||||
- name: Install PDFium
|
- name: Install PDFium
|
||||||
shell: bash
|
shell: bash
|
||||||
@@ -202,6 +213,50 @@ jobs:
|
|||||||
assert "layout_ms" not in result["pages"][0]["timings"]
|
assert "layout_ms" not in result["pages"][0]["timings"]
|
||||||
PY
|
PY
|
||||||
|
|
||||||
|
- name: Build Node binding
|
||||||
|
working-directory: napi
|
||||||
|
run: |
|
||||||
|
bun install --frozen-lockfile
|
||||||
|
bunx napi build --platform --release
|
||||||
|
|
||||||
|
- name: Run Node OCR binding
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
node --input-type=module - <<'JS'
|
||||||
|
import { readFileSync } from 'node:fs'
|
||||||
|
import { processPdfWithOcr } from './napi/index.js'
|
||||||
|
|
||||||
|
const pdf = readFileSync('tests/fixtures/scan_with_native_header_text.pdf')
|
||||||
|
const result = await processPdfWithOcr(pdf, { offline: true })
|
||||||
|
if (JSON.stringify(result.pagesRoutedToOcr) !== '[1]') throw new Error('unexpected OCR route')
|
||||||
|
if (result.pagesRecommendingHosted.length !== 0) throw new Error('unexpected hosted recommendation')
|
||||||
|
if (!['Ocr', 'Fused'].includes(result.pages[0].provenance.source)) throw new Error('unexpected source')
|
||||||
|
if (!result.pages[0].markdown.trim()) throw new Error('empty OCR markdown')
|
||||||
|
JS
|
||||||
|
|
||||||
|
- name: Build and install Python binding
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
python -m pip install 'maturin>=1,<2'
|
||||||
|
maturin build --release --out "$RUNNER_TEMP/python-wheels"
|
||||||
|
python -m pip install "$RUNNER_TEMP"/python-wheels/*.whl
|
||||||
|
|
||||||
|
- name: Run Python OCR binding
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
python - <<'PY'
|
||||||
|
import pdf_inspector
|
||||||
|
|
||||||
|
result = pdf_inspector.process_pdf_with_ocr(
|
||||||
|
"tests/fixtures/scan_with_native_header_text.pdf",
|
||||||
|
offline=True,
|
||||||
|
)
|
||||||
|
assert result.pages_routed_to_ocr == [1]
|
||||||
|
assert result.pages_recommending_hosted == []
|
||||||
|
assert result.pages[0].provenance.source in {"ocr", "fused"}
|
||||||
|
assert result.pages[0].markdown.strip()
|
||||||
|
PY
|
||||||
|
|
||||||
wasm:
|
wasm:
|
||||||
name: WebAssembly
|
name: WebAssembly
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
|
|||||||
+1
-1
@@ -80,7 +80,7 @@ tempfile = "3.3"
|
|||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = []
|
default = []
|
||||||
python = ["pyo3"]
|
python = ["pyo3", "ocr"]
|
||||||
vision = []
|
vision = []
|
||||||
model-cache = ["vision", "dep:dirs", "dep:fs2", "dep:sha2", "dep:windows-sys"]
|
model-cache = ["vision", "dep:dirs", "dep:fs2", "dep:sha2", "dep:windows-sys"]
|
||||||
model-download = ["model-cache", "dep:ureq"]
|
model-download = ["model-cache", "dep:ureq"]
|
||||||
|
|||||||
@@ -18,10 +18,10 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
|||||||
- **CID font support** — ToUnicode CMap decoding for Type0/Identity-H fonts, UTF-16BE, UTF-8, and Latin-1 encodings.
|
- **CID font support** — ToUnicode CMap decoding for Type0/Identity-H fonts, UTF-16BE, UTF-8, and Latin-1 encodings.
|
||||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||||
- **Optional OCR** — An opt-in Rust and CLI feature selectively renders only pages that need OCR, runs PP-OCRv6 Small locally, and preserves per-page provenance and hosted-fallback recommendations.
|
- **Selective OCR** — Rust, CLI, Python, and Node can render only pages that need OCR, run PP-OCRv6 Small locally, and preserve per-page provenance and hosted-fallback recommendations.
|
||||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||||
- **Lightweight by default** — The default build is pure Rust with no ML models or external services. PDFium, ONNX Runtime, and OCR models are added only when the native `ocr` feature is selected and remain external runtime artifacts.
|
- **Lightweight by default** — The default Rust and browser builds remain pure extraction. Native Python and Node packages include the OCR integration, but PDFium, ONNX Runtime, and model files remain external and are touched only when a page is routed to OCR.
|
||||||
|
|
||||||
## Benchmark
|
## Benchmark
|
||||||
|
|
||||||
@@ -58,6 +58,10 @@ import pdf_inspector
|
|||||||
result = pdf_inspector.process_pdf("document.pdf")
|
result = pdf_inspector.process_pdf("document.pdf")
|
||||||
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
||||||
print(result.markdown) # Markdown string or None
|
print(result.markdown) # Markdown string or None
|
||||||
|
|
||||||
|
# Selective OCR; clean text PDFs do not load the external OCR runtime.
|
||||||
|
ocr = pdf_inspector.process_pdf_with_ocr("document.pdf")
|
||||||
|
print(ocr.pages_routed_to_ocr)
|
||||||
```
|
```
|
||||||
|
|
||||||
> Full API reference: [docs/python.md](docs/python.md)
|
> Full API reference: [docs/python.md](docs/python.md)
|
||||||
@@ -70,11 +74,15 @@ npm install @firecrawl/pdf-inspector
|
|||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { readFileSync } from 'fs';
|
import { readFileSync } from 'fs';
|
||||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector';
|
import { processPdf, processPdfWithOcr } from '@firecrawl/pdf-inspector';
|
||||||
|
|
||||||
const result = processPdf(readFileSync('document.pdf'));
|
const pdf = readFileSync('document.pdf');
|
||||||
|
const result = processPdf(pdf);
|
||||||
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
||||||
console.log(result.markdown); // Markdown string or null
|
console.log(result.markdown); // Markdown string or null
|
||||||
|
|
||||||
|
const ocr = await processPdfWithOcr(pdf); // selective OCR, off the event loop
|
||||||
|
console.log(ocr.pagesRoutedToOcr);
|
||||||
```
|
```
|
||||||
|
|
||||||
> Full API reference: [napi/README.md](napi/README.md)
|
> Full API reference: [napi/README.md](napi/README.md)
|
||||||
@@ -161,7 +169,7 @@ detect-pdf document.pdf --json
|
|||||||
detect-pdf document.pdf --analyze --json
|
detect-pdf document.pdf --analyze --json
|
||||||
```
|
```
|
||||||
|
|
||||||
OCR is a separate native CLI build and does not change the default package:
|
Rust and CLI consumers opt into OCR at build time:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cargo install pdf-inspector --features ocr --bin pdf2md
|
cargo install pdf-inspector --features ocr --bin pdf2md
|
||||||
@@ -171,8 +179,11 @@ PDFIUM_LIB_PATH=/path/to/libpdfium ORT_DYLIB_PATH=/path/to/libonnxruntime \
|
|||||||
|
|
||||||
The OCR JSON envelope is versioned and reports routed pages, per-page source
|
The OCR JSON envelope is versioned and reports routed pages, per-page source
|
||||||
and confidence, warnings, and pages recommended for the hosted document
|
and confidence, warnings, and pages recommended for the hosted document
|
||||||
pipeline. See the [Rust API guide](docs/rust-api.md#complete-ocr-api) for model
|
pipeline. Native Python and Node packages expose the same pipeline without a
|
||||||
cache and offline configuration.
|
source-build feature. All native entry points still require separately
|
||||||
|
installed PDFium and ONNX Runtime libraries only when OCR is routed. See the
|
||||||
|
[Rust API guide](docs/rust-api.md#complete-ocr-api) for model cache and offline
|
||||||
|
configuration.
|
||||||
|
|
||||||
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
||||||
|
|
||||||
|
|||||||
+63
-2
@@ -1,6 +1,6 @@
|
|||||||
# pdf-inspector
|
# pdf-inspector
|
||||||
|
|
||||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
Fast PDF classification, text extraction, and selective OCR. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts clean native and OCR results to Markdown. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
||||||
|
|
||||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||||
|
|
||||||
@@ -10,7 +10,8 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
|||||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||||
- **Lightweight** — native Rust core, no ML models, no external services; ships type stubs.
|
- **Selective OCR** — `auto` routes only pages rejected by native extraction; `force` OCRs every selected page; `off` keeps the result/provenance contract without external runtime work.
|
||||||
|
- **External artifacts** — the wheel embeds no OCR models, PDFium, or ONNX Runtime; clean `auto` requests never load or download them.
|
||||||
|
|
||||||
## Benchmark
|
## Benchmark
|
||||||
|
|
||||||
@@ -39,6 +40,12 @@ pip install maturin
|
|||||||
maturin develop --release
|
maturin develop --release
|
||||||
```
|
```
|
||||||
|
|
||||||
|
OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||||
|
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||||
|
platform library search path. The pinned OCR model set is downloaded and
|
||||||
|
checksum-verified on the first routed page; use `offline=True` with a warm
|
||||||
|
cache or `model_directory` to prohibit network access.
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```python
|
```python
|
||||||
@@ -81,6 +88,19 @@ for page in result.pages:
|
|||||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||||
|
|
||||||
|
# One-call selective OCR. This releases the GIL while processing.
|
||||||
|
ocr = pdf_inspector.process_pdf_with_ocr("document.pdf")
|
||||||
|
for page in ocr.pages:
|
||||||
|
print(page.page_number, page.provenance.source)
|
||||||
|
|
||||||
|
# Restrict OCR processing to 1-indexed PDF pages and prohibit downloads.
|
||||||
|
ocr = pdf_inspector.process_pdf_with_ocr(
|
||||||
|
"document.pdf",
|
||||||
|
page_numbers=[1, 3],
|
||||||
|
model_directory="/opt/models/pp-ocrv6-small",
|
||||||
|
offline=True,
|
||||||
|
)
|
||||||
|
|
||||||
# Structure-tree elements from tagged PDFs (empty list when untagged).
|
# Structure-tree elements from tagged PDFs (empty list when untagged).
|
||||||
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
|
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
|
||||||
# against extract_text_with_positions — e.g. to recover real heading levels:
|
# against extract_text_with_positions — e.g. to recover real heading levels:
|
||||||
@@ -99,6 +119,8 @@ headings = [
|
|||||||
|---|---|
|
|---|---|
|
||||||
| `process_pdf(path, pages=None)` | Full processing (detect + extract + markdown) |
|
| `process_pdf(path, pages=None)` | Full processing (detect + extract + markdown) |
|
||||||
| `process_pdf_bytes(data, pages=None)` | Full processing from bytes |
|
| `process_pdf_bytes(data, pages=None)` | Full processing from bytes |
|
||||||
|
| `process_pdf_with_ocr(path, **options)` | Native extraction + selective OCR with provenance |
|
||||||
|
| `process_pdf_with_ocr_bytes(data, **options)` | Native extraction + selective OCR from bytes |
|
||||||
| `detect_pdf(path)` | Fast detection only (returns PdfResult) |
|
| `detect_pdf(path)` | Fast detection only (returns PdfResult) |
|
||||||
| `detect_pdf_bytes(data)` | Fast detection from bytes |
|
| `detect_pdf_bytes(data)` | Fast detection from bytes |
|
||||||
| `classify_pdf(path)` | Lightweight classification (returns PdfClassification) |
|
| `classify_pdf(path)` | Lightweight classification (returns PdfClassification) |
|
||||||
@@ -137,6 +159,45 @@ class PageOcrReasons: # per-page OCR diagnostics
|
|||||||
page: int # 1-indexed
|
page: int # 1-indexed
|
||||||
reasons: list[str] # machine-readable reason identifiers
|
reasons: list[str] # machine-readable reason identifiers
|
||||||
|
|
||||||
|
class OcrModelIdentity:
|
||||||
|
name: str # model family/name
|
||||||
|
revision: str # immutable artifact-set revision
|
||||||
|
|
||||||
|
class OcrTimings: # per-page processing stages
|
||||||
|
render_ms: int
|
||||||
|
ocr_ms: int
|
||||||
|
assembly_ms: int
|
||||||
|
|
||||||
|
class OcrPageProvenance:
|
||||||
|
page_number: int # 1-indexed
|
||||||
|
source: Literal["native", "ocr", "fused"]
|
||||||
|
ocr_model: OcrModelIdentity | None
|
||||||
|
render_dpi: float | None
|
||||||
|
ocr_confidence: float | None
|
||||||
|
timings: OcrTimings
|
||||||
|
warnings: list[str]
|
||||||
|
hosted_recommended: bool
|
||||||
|
|
||||||
|
class OcrPageResult:
|
||||||
|
page_number: int # 1-indexed
|
||||||
|
markdown: str
|
||||||
|
provenance: OcrPageProvenance
|
||||||
|
|
||||||
|
class OcrPdfResult: # process_pdf_with_ocr / bytes
|
||||||
|
markdown: str
|
||||||
|
pages: list[OcrPageResult]
|
||||||
|
page_count: int
|
||||||
|
pages_recommended_for_ocr: list[int]
|
||||||
|
pages_routed_to_ocr: list[int]
|
||||||
|
pages_recommending_hosted: list[int]
|
||||||
|
ocr_reasons_by_page: list[PageOcrReasons]
|
||||||
|
pages_with_tables: list[int]
|
||||||
|
pages_with_columns: list[int]
|
||||||
|
is_complex: bool
|
||||||
|
processing_time_ms: int
|
||||||
|
render_time_ms: int
|
||||||
|
ocr_time_ms: int
|
||||||
|
|
||||||
class PdfClassification: # classify_pdf
|
class PdfClassification: # classify_pdf
|
||||||
pdf_type: str
|
pdf_type: str
|
||||||
page_count: int
|
page_count: int
|
||||||
|
|||||||
Generated
+2253
-30
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -7,7 +7,7 @@ edition = "2021"
|
|||||||
crate-type = ["cdylib"]
|
crate-type = ["cdylib"]
|
||||||
|
|
||||||
[dependencies]
|
[dependencies]
|
||||||
pdf-inspector = { path = ".." }
|
pdf-inspector = { path = "..", features = ["ocr"] }
|
||||||
napi = { version = "3.0.0", features = ["serde-json"] }
|
napi = { version = "3.0.0", features = ["serde-json"] }
|
||||||
napi-derive = "3.0.0"
|
napi-derive = "3.0.0"
|
||||||
|
|
||||||
|
|||||||
+50
-1
@@ -10,7 +10,8 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
|||||||
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
||||||
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
||||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||||
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~5–6 MB platform binary, TypeScript definitions included.
|
- **Selective OCR** — `Auto` routes only pages rejected by native extraction and returns source/model provenance plus hosted-fallback recommendations.
|
||||||
|
- **External artifacts** — the native package embeds no OCR models, PDFium, or ONNX Runtime; clean `Auto` requests never load or download them.
|
||||||
|
|
||||||
## Benchmark
|
## Benchmark
|
||||||
|
|
||||||
@@ -36,8 +37,40 @@ bun add @firecrawl/pdf-inspector
|
|||||||
|
|
||||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||||
|
|
||||||
|
OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||||
|
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||||
|
platform library search path. The pinned OCR model set is downloaded and
|
||||||
|
checksum-verified on the first routed page; use `offline: true` with a warm
|
||||||
|
cache or `modelDirectory` to prohibit network access.
|
||||||
|
|
||||||
## API
|
## API
|
||||||
|
|
||||||
|
### `processPdfWithOcr(buffer: Buffer, options?: OcrOptions): Promise<OcrPdfResult>`
|
||||||
|
|
||||||
|
Run native extraction first and OCR only the pages selected by its quality
|
||||||
|
signals. The default mode is `Auto`; `Off` returns the same detailed result
|
||||||
|
shape without external runtime work, and `Force` OCRs every selected page.
|
||||||
|
The work runs on the libuv thread pool and never blocks Node's event loop.
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
import { OcrMode, processPdfWithOcr } from '@firecrawl/pdf-inspector'
|
||||||
|
|
||||||
|
const result = await processPdfWithOcr(pdf, {
|
||||||
|
mode: OcrMode.Auto,
|
||||||
|
pageNumbers: [1, 3], // 1-indexed
|
||||||
|
})
|
||||||
|
|
||||||
|
for (const page of result.pages) {
|
||||||
|
console.log(page.pageNumber, page.provenance.source)
|
||||||
|
}
|
||||||
|
console.log(result.pagesRoutedToOcr)
|
||||||
|
console.log(result.pagesRecommendingHosted)
|
||||||
|
```
|
||||||
|
|
||||||
|
For offline deployments, pass `modelDirectory` and `offline: true`. Other
|
||||||
|
controls include `dpi`, `minimumConfidence`,
|
||||||
|
`hostedRecommendationConfidence`, and `password`.
|
||||||
|
|
||||||
### `classifyPdf(buffer: Buffer): PdfClassification`
|
### `classifyPdf(buffer: Buffer): PdfClassification`
|
||||||
|
|
||||||
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
|
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
|
||||||
@@ -124,6 +157,22 @@ interface RegionText {
|
|||||||
needsOcr: boolean // true when text is unreliable
|
needsOcr: boolean // true when text is unreliable
|
||||||
ocrReason?: string // "suspected_garbled_text" when known
|
ocrReason?: string // "suspected_garbled_text" when known
|
||||||
}
|
}
|
||||||
|
|
||||||
|
interface OcrPdfResult {
|
||||||
|
markdown: string
|
||||||
|
pages: OcrPageResult[] // 1-indexed pages + provenance
|
||||||
|
pageCount: number
|
||||||
|
pagesRecommendedForOcr: number[]
|
||||||
|
pagesRoutedToOcr: number[]
|
||||||
|
pagesRecommendingHosted: number[]
|
||||||
|
ocrReasonsByPage: PageOcrReasons[]
|
||||||
|
pagesWithTables: number[]
|
||||||
|
pagesWithColumns: number[]
|
||||||
|
isComplex: boolean
|
||||||
|
processingTimeMs: number
|
||||||
|
renderTimeMs: number
|
||||||
|
ocrTimeMs: number
|
||||||
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
## Platforms
|
## Platforms
|
||||||
|
|||||||
+241
@@ -27,6 +27,26 @@ pub enum ItemType {
|
|||||||
FormField,
|
FormField,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Selects when OCR runs.
|
||||||
|
#[napi(string_enum)]
|
||||||
|
#[derive(Clone, Copy)]
|
||||||
|
pub enum OcrMode {
|
||||||
|
/// Never run OCR; return the native extraction in the OCR result shape.
|
||||||
|
Off,
|
||||||
|
/// Run OCR only on pages selected by the native quality signals.
|
||||||
|
Auto,
|
||||||
|
/// Run OCR on every selected page.
|
||||||
|
Force,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How final page content was sourced.
|
||||||
|
#[napi(string_enum)]
|
||||||
|
pub enum PageContentSource {
|
||||||
|
Native,
|
||||||
|
Ocr,
|
||||||
|
Fused,
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Result types
|
// Result types
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -128,6 +148,84 @@ pub struct VectorGridDetectionJs {
|
|||||||
pub cell_bboxes: Vec<Vec<f64>>,
|
pub cell_bboxes: Vec<Vec<f64>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Options for one-call native extraction with selective OCR.
|
||||||
|
#[napi(object)]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct OcrOptions {
|
||||||
|
/// OCR routing behavior. Defaults to Auto.
|
||||||
|
pub mode: Option<OcrMode>,
|
||||||
|
/// Optional 1-indexed page selection.
|
||||||
|
pub page_numbers: Option<Vec<u32>>,
|
||||||
|
/// Password for an encrypted PDF.
|
||||||
|
pub password: Option<String>,
|
||||||
|
/// Page rasterization resolution. Defaults to 150 DPI.
|
||||||
|
pub dpi: Option<f64>,
|
||||||
|
/// Drop OCR spans below this inclusive 0-1 threshold.
|
||||||
|
pub minimum_confidence: Option<f64>,
|
||||||
|
/// Recommend hosted parsing below this inclusive 0-1 page confidence.
|
||||||
|
pub hosted_recommendation_confidence: Option<f64>,
|
||||||
|
/// Directory containing an offline OCR model set.
|
||||||
|
pub model_directory: Option<String>,
|
||||||
|
/// Disable model downloads and require a model directory or warm cache.
|
||||||
|
pub offline: Option<bool>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Exact OCR model identity retained in page provenance.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct OcrModelIdentity {
|
||||||
|
pub name: String,
|
||||||
|
pub revision: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Per-page OCR processing timings.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct OcrTimings {
|
||||||
|
pub render_ms: u32,
|
||||||
|
pub ocr_ms: u32,
|
||||||
|
pub assembly_ms: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Source, model, confidence, and fallback metadata for one page.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct OcrPageProvenance {
|
||||||
|
/// 1-indexed page number.
|
||||||
|
pub page_number: u32,
|
||||||
|
pub source: PageContentSource,
|
||||||
|
pub ocr_model: Option<OcrModelIdentity>,
|
||||||
|
pub render_dpi: Option<f64>,
|
||||||
|
pub ocr_confidence: Option<f64>,
|
||||||
|
pub timings: OcrTimings,
|
||||||
|
pub warnings: Vec<String>,
|
||||||
|
pub hosted_recommended: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Final Markdown and provenance for one page.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct OcrPageResult {
|
||||||
|
/// 1-indexed page number.
|
||||||
|
pub page_number: u32,
|
||||||
|
pub markdown: String,
|
||||||
|
pub provenance: OcrPageProvenance,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Complete native/OCR Markdown output.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct OcrPdfResult {
|
||||||
|
pub markdown: String,
|
||||||
|
pub pages: Vec<OcrPageResult>,
|
||||||
|
pub page_count: u32,
|
||||||
|
pub pages_recommended_for_ocr: Vec<u32>,
|
||||||
|
pub pages_routed_to_ocr: Vec<u32>,
|
||||||
|
pub pages_recommending_hosted: Vec<u32>,
|
||||||
|
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||||
|
pub pages_with_tables: Vec<u32>,
|
||||||
|
pub pages_with_columns: Vec<u32>,
|
||||||
|
pub is_complex: bool,
|
||||||
|
pub processing_time_ms: u32,
|
||||||
|
pub render_time_ms: u32,
|
||||||
|
pub ocr_time_ms: u32,
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Helpers
|
// Helpers
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -168,6 +266,103 @@ fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<
|
|||||||
.collect()
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn to_core_ocr_options(options: Option<OcrOptions>) -> pdf_inspector::vision::OcrPdfOptions {
|
||||||
|
let mut result = pdf_inspector::vision::OcrPdfOptions::auto();
|
||||||
|
let Some(options) = options else {
|
||||||
|
return result;
|
||||||
|
};
|
||||||
|
|
||||||
|
if let Some(mode) = options.mode {
|
||||||
|
result.ocr.mode = match mode {
|
||||||
|
OcrMode::Off => pdf_inspector::vision::OcrMode::Off,
|
||||||
|
OcrMode::Auto => pdf_inspector::vision::OcrMode::Auto,
|
||||||
|
OcrMode::Force => pdf_inspector::vision::OcrMode::Force,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
if let Some(pages) = options.page_numbers {
|
||||||
|
result = result.page_numbers(pages);
|
||||||
|
}
|
||||||
|
if let Some(password) = options.password {
|
||||||
|
result = result.password(password);
|
||||||
|
}
|
||||||
|
if let Some(dpi) = options.dpi {
|
||||||
|
result.render.dpi = dpi as f32;
|
||||||
|
}
|
||||||
|
if let Some(minimum_confidence) = options.minimum_confidence {
|
||||||
|
result.ocr.minimum_confidence = minimum_confidence as f32;
|
||||||
|
}
|
||||||
|
if let Some(confidence) = options.hosted_recommendation_confidence {
|
||||||
|
result.hosted_recommendation_confidence = confidence as f32;
|
||||||
|
}
|
||||||
|
if let Some(directory) = options.model_directory {
|
||||||
|
result.ocr.model_directory = Some(directory.into());
|
||||||
|
}
|
||||||
|
if options.offline.unwrap_or(false) {
|
||||||
|
result.ocr.model_downloads = pdf_inspector::vision::ModelDownloadPolicy::Offline;
|
||||||
|
}
|
||||||
|
result
|
||||||
|
}
|
||||||
|
|
||||||
|
fn convert_page_content_source(
|
||||||
|
source: pdf_inspector::vision::PageContentSource,
|
||||||
|
) -> PageContentSource {
|
||||||
|
match source {
|
||||||
|
pdf_inspector::vision::PageContentSource::Native => PageContentSource::Native,
|
||||||
|
pdf_inspector::vision::PageContentSource::Ocr => PageContentSource::Ocr,
|
||||||
|
pdf_inspector::vision::PageContentSource::Fused => PageContentSource::Fused,
|
||||||
|
_ => PageContentSource::Native,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn timing_ms(value: u64) -> u32 {
|
||||||
|
u32::try_from(value).unwrap_or(u32::MAX)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn to_napi_ocr_result(result: pdf_inspector::vision::OcrPdfResult) -> OcrPdfResult {
|
||||||
|
OcrPdfResult {
|
||||||
|
markdown: result.markdown,
|
||||||
|
pages: result
|
||||||
|
.pages
|
||||||
|
.into_iter()
|
||||||
|
.map(|page| {
|
||||||
|
let provenance = page.provenance;
|
||||||
|
OcrPageResult {
|
||||||
|
page_number: page.page_number,
|
||||||
|
markdown: page.markdown,
|
||||||
|
provenance: OcrPageProvenance {
|
||||||
|
page_number: provenance.page_number,
|
||||||
|
source: convert_page_content_source(provenance.source),
|
||||||
|
ocr_model: provenance.ocr_model.map(|model| OcrModelIdentity {
|
||||||
|
name: model.name,
|
||||||
|
revision: model.revision,
|
||||||
|
}),
|
||||||
|
render_dpi: provenance.render_dpi.map(f64::from),
|
||||||
|
ocr_confidence: provenance.ocr_confidence.map(f64::from),
|
||||||
|
timings: OcrTimings {
|
||||||
|
render_ms: timing_ms(provenance.timings.render_ms),
|
||||||
|
ocr_ms: timing_ms(provenance.timings.ocr_ms),
|
||||||
|
assembly_ms: timing_ms(provenance.timings.assembly_ms),
|
||||||
|
},
|
||||||
|
warnings: provenance.warnings,
|
||||||
|
hosted_recommended: provenance.hosted_recommended,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
page_count: result.page_count,
|
||||||
|
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
|
||||||
|
pages_routed_to_ocr: result.pages_routed_to_ocr,
|
||||||
|
pages_recommending_hosted: result.pages_recommending_hosted,
|
||||||
|
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||||
|
pages_with_tables: result.pages_with_tables,
|
||||||
|
pages_with_columns: result.pages_with_columns,
|
||||||
|
is_complex: result.is_complex,
|
||||||
|
processing_time_ms: timing_ms(result.processing_time_ms),
|
||||||
|
render_time_ms: timing_ms(result.render_time_ms),
|
||||||
|
ocr_time_ms: timing_ms(result.ocr_time_ms),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||||
match t {
|
match t {
|
||||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||||
@@ -219,6 +414,13 @@ fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult>
|
|||||||
Ok(to_napi_result(result))
|
Ok(to_napi_result(result))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn process_pdf_with_ocr_impl(bytes: &[u8], options: Option<OcrOptions>) -> Result<OcrPdfResult> {
|
||||||
|
let options = to_core_ocr_options(options);
|
||||||
|
let result = pdf_inspector::vision::process_pdf_with_ocr_mem(bytes, options)
|
||||||
|
.map_err(|error| to_napi_err(error, "process_pdf_with_ocr"))?;
|
||||||
|
Ok(to_napi_ocr_result(result))
|
||||||
|
}
|
||||||
|
|
||||||
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||||
let result =
|
let result =
|
||||||
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||||
@@ -818,6 +1020,45 @@ pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<P
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub struct ProcessPdfWithOcrTask {
|
||||||
|
bytes: Vec<u8>,
|
||||||
|
options: Option<OcrOptions>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Task for ProcessPdfWithOcrTask {
|
||||||
|
type Output = OcrPdfResult;
|
||||||
|
type JsValue = OcrPdfResult;
|
||||||
|
|
||||||
|
fn compute(&mut self) -> Result<Self::Output> {
|
||||||
|
let bytes = std::mem::take(&mut self.bytes);
|
||||||
|
let options = self.options.take();
|
||||||
|
catch_panic(
|
||||||
|
"process_pdf_with_ocr",
|
||||||
|
panic::AssertUnwindSafe(move || process_pdf_with_ocr_impl(&bytes, options)),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||||
|
Ok(output)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Process a PDF with selective OCR on the libuv thread pool.
|
||||||
|
///
|
||||||
|
/// OCR defaults to Auto, which only loads PDFium, ONNX Runtime, and the OCR
|
||||||
|
/// model if native extraction routes at least one page. The input buffer is
|
||||||
|
/// copied before the promise is returned and is safe to reuse immediately.
|
||||||
|
#[napi(ts_return_type = "Promise<OcrPdfResult>")]
|
||||||
|
pub fn process_pdf_with_ocr(
|
||||||
|
buffer: Buffer,
|
||||||
|
options: Option<OcrOptions>,
|
||||||
|
) -> AsyncTask<ProcessPdfWithOcrTask> {
|
||||||
|
AsyncTask::new(ProcessPdfWithOcrTask {
|
||||||
|
bytes: buffer.to_vec(),
|
||||||
|
options,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
pub struct ClassifyPdfTask {
|
pub struct ClassifyPdfTask {
|
||||||
bytes: Vec<u8>,
|
bytes: Vec<u8>,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ import { strict as assert } from 'assert';
|
|||||||
import {
|
import {
|
||||||
processPdf,
|
processPdf,
|
||||||
processPdfAsync,
|
processPdfAsync,
|
||||||
|
processPdfWithOcr,
|
||||||
detectPdf,
|
detectPdf,
|
||||||
classifyPdf,
|
classifyPdf,
|
||||||
classifyPdfAsync,
|
classifyPdfAsync,
|
||||||
@@ -220,6 +221,37 @@ const fromMutated = await inFlight;
|
|||||||
assert.equal(fromMutated.markdown, result.markdown);
|
assert.equal(fromMutated.markdown, result.markdown);
|
||||||
console.log(' processPdfAsync input copied at call time: OK');
|
console.log(' processPdfAsync input copied at call time: OK');
|
||||||
|
|
||||||
|
// --- Selective OCR ---
|
||||||
|
console.log('Testing processPdfWithOcr...');
|
||||||
|
|
||||||
|
// Off exercises the complete result/provenance contract without loading
|
||||||
|
// external PDFium, ONNX Runtime, or model artifacts.
|
||||||
|
const ocrOff = await processPdfWithOcr(fixture, { mode: 'Off' });
|
||||||
|
assert.equal(ocrOff.pageCount, 3);
|
||||||
|
assert.equal(ocrOff.pages.length, 3);
|
||||||
|
assert.deepEqual(ocrOff.pagesRoutedToOcr, []);
|
||||||
|
assert.ok(ocrOff.pages.every(page => page.provenance.source === 'Native'));
|
||||||
|
assert.ok(ocrOff.pages.every(page => page.provenance.ocrModel === undefined));
|
||||||
|
assert.ok(ocrOff.markdown.length > 0);
|
||||||
|
|
||||||
|
// Auto must preserve the lightweight path for clean text PDFs.
|
||||||
|
const ocrAuto = await processPdfWithOcr(fixture);
|
||||||
|
assert.deepEqual(ocrAuto.pagesRoutedToOcr, []);
|
||||||
|
assert.equal(ocrAuto.renderTimeMs, 0);
|
||||||
|
assert.equal(ocrAuto.ocrTimeMs, 0);
|
||||||
|
|
||||||
|
const ocrSelected = await processPdfWithOcr(fixture, {
|
||||||
|
mode: 'Off',
|
||||||
|
pageNumbers: [2],
|
||||||
|
});
|
||||||
|
assert.deepEqual(ocrSelected.pages.map(page => page.pageNumber), [2]);
|
||||||
|
|
||||||
|
await assert.rejects(
|
||||||
|
processPdfWithOcr(fixture, { mode: 'Off', pageNumbers: [0] }),
|
||||||
|
/page 0/,
|
||||||
|
);
|
||||||
|
console.log(' processPdfWithOcr: OK');
|
||||||
|
|
||||||
// concurrent async calls all settle
|
// concurrent async calls all settle
|
||||||
const [c1, c2, c3] = await Promise.all([
|
const [c1, c2, c3] = await Promise.all([
|
||||||
processPdfAsync(fixture),
|
processPdfAsync(fixture),
|
||||||
|
|||||||
+81
-1
@@ -1,6 +1,6 @@
|
|||||||
"""Type stubs for pdf_inspector."""
|
"""Type stubs for pdf_inspector."""
|
||||||
|
|
||||||
from typing import Optional
|
from typing import Literal, Optional
|
||||||
|
|
||||||
class PdfResult:
|
class PdfResult:
|
||||||
"""Result of processing a PDF file."""
|
"""Result of processing a PDF file."""
|
||||||
@@ -27,6 +27,53 @@ class PageOcrReasons:
|
|||||||
reasons: list[str]
|
reasons: list[str]
|
||||||
"""Machine-readable OCR reason identifiers."""
|
"""Machine-readable OCR reason identifiers."""
|
||||||
|
|
||||||
|
class OcrModelIdentity:
|
||||||
|
"""Exact OCR model identity retained in page provenance."""
|
||||||
|
name: str
|
||||||
|
revision: str
|
||||||
|
|
||||||
|
class OcrTimings:
|
||||||
|
"""Per-page OCR processing timings."""
|
||||||
|
render_ms: int
|
||||||
|
ocr_ms: int
|
||||||
|
assembly_ms: int
|
||||||
|
|
||||||
|
class OcrPageProvenance:
|
||||||
|
"""Source, model, confidence, and fallback metadata for one page."""
|
||||||
|
page_number: int
|
||||||
|
"""1-indexed page number."""
|
||||||
|
source: Literal["native", "ocr", "fused"]
|
||||||
|
"""'native', 'ocr', or 'fused'."""
|
||||||
|
ocr_model: Optional[OcrModelIdentity]
|
||||||
|
render_dpi: Optional[float]
|
||||||
|
ocr_confidence: Optional[float]
|
||||||
|
timings: OcrTimings
|
||||||
|
warnings: list[str]
|
||||||
|
hosted_recommended: bool
|
||||||
|
|
||||||
|
class OcrPageResult:
|
||||||
|
"""Final Markdown and provenance for one page."""
|
||||||
|
page_number: int
|
||||||
|
"""1-indexed page number."""
|
||||||
|
markdown: str
|
||||||
|
provenance: OcrPageProvenance
|
||||||
|
|
||||||
|
class OcrPdfResult:
|
||||||
|
"""Complete native/OCR Markdown output."""
|
||||||
|
markdown: str
|
||||||
|
pages: list[OcrPageResult]
|
||||||
|
page_count: int
|
||||||
|
pages_recommended_for_ocr: list[int]
|
||||||
|
pages_routed_to_ocr: list[int]
|
||||||
|
pages_recommending_hosted: list[int]
|
||||||
|
ocr_reasons_by_page: list[PageOcrReasons]
|
||||||
|
pages_with_tables: list[int]
|
||||||
|
pages_with_columns: list[int]
|
||||||
|
is_complex: bool
|
||||||
|
processing_time_ms: int
|
||||||
|
render_time_ms: int
|
||||||
|
ocr_time_ms: int
|
||||||
|
|
||||||
class PdfClassification:
|
class PdfClassification:
|
||||||
"""Lightweight PDF classification result."""
|
"""Lightweight PDF classification result."""
|
||||||
pdf_type: str
|
pdf_type: str
|
||||||
@@ -114,6 +161,39 @@ def process_pdf_bytes(data: bytes, pages: Optional[list[int]] = None) -> PdfResu
|
|||||||
"""Process a PDF from bytes in memory."""
|
"""Process a PDF from bytes in memory."""
|
||||||
...
|
...
|
||||||
|
|
||||||
|
def process_pdf_with_ocr(
|
||||||
|
path: str,
|
||||||
|
*,
|
||||||
|
mode: Literal["off", "auto", "force"] = "auto",
|
||||||
|
page_numbers: Optional[list[int]] = None,
|
||||||
|
password: Optional[str] = None,
|
||||||
|
dpi: float = 150.0,
|
||||||
|
minimum_confidence: float = 0.0,
|
||||||
|
hosted_recommendation_confidence: float = 0.5,
|
||||||
|
model_directory: Optional[str] = None,
|
||||||
|
offline: bool = False,
|
||||||
|
) -> OcrPdfResult:
|
||||||
|
"""Process a PDF through native extraction and selective OCR.
|
||||||
|
|
||||||
|
Page numbers are 1-indexed. OCR runs without holding the Python GIL.
|
||||||
|
"""
|
||||||
|
...
|
||||||
|
|
||||||
|
def process_pdf_with_ocr_bytes(
|
||||||
|
data: bytes,
|
||||||
|
*,
|
||||||
|
mode: Literal["off", "auto", "force"] = "auto",
|
||||||
|
page_numbers: Optional[list[int]] = None,
|
||||||
|
password: Optional[str] = None,
|
||||||
|
dpi: float = 150.0,
|
||||||
|
minimum_confidence: float = 0.0,
|
||||||
|
hosted_recommendation_confidence: float = 0.5,
|
||||||
|
model_directory: Optional[str] = None,
|
||||||
|
offline: bool = False,
|
||||||
|
) -> OcrPdfResult:
|
||||||
|
"""Process PDF bytes through native extraction and selective OCR."""
|
||||||
|
...
|
||||||
|
|
||||||
def detect_pdf(path: str) -> PdfResult:
|
def detect_pdf(path: str) -> PdfResult:
|
||||||
"""Fast detection only — no text extraction."""
|
"""Fast detection only — no text extraction."""
|
||||||
...
|
...
|
||||||
|
|||||||
+296
@@ -85,6 +85,107 @@ impl PyPageOcrReasons {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Exact OCR model identity retained in page provenance.
|
||||||
|
#[pyclass(name = "OcrModelIdentity")]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct PyOcrModelIdentity {
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub name: String,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub revision: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Per-page OCR processing timings.
|
||||||
|
#[pyclass(name = "OcrTimings")]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct PyOcrTimings {
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub render_ms: u64,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub ocr_ms: u64,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub assembly_ms: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Source, model, confidence, and fallback metadata for one page.
|
||||||
|
#[pyclass(name = "OcrPageProvenance")]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct PyOcrPageProvenance {
|
||||||
|
/// 1-indexed page number.
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub page_number: u32,
|
||||||
|
/// "native", "ocr", or "fused".
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub source: String,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub ocr_model: Option<PyOcrModelIdentity>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub render_dpi: Option<f32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub ocr_confidence: Option<f32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub timings: PyOcrTimings,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub warnings: Vec<String>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub hosted_recommended: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Final Markdown and provenance for one page.
|
||||||
|
#[pyclass(name = "OcrPageResult")]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct PyOcrPageResult {
|
||||||
|
/// 1-indexed page number.
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub page_number: u32,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub markdown: String,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub provenance: PyOcrPageProvenance,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Complete native/OCR Markdown output.
|
||||||
|
#[pyclass(name = "OcrPdfResult")]
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct PyOcrPdfResult {
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub markdown: String,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub pages: Vec<PyOcrPageResult>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub page_count: u32,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub pages_recommended_for_ocr: Vec<u32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub pages_routed_to_ocr: Vec<u32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub pages_recommending_hosted: Vec<u32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub pages_with_tables: Vec<u32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub pages_with_columns: Vec<u32>,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub is_complex: bool,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub processing_time_ms: u64,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub render_time_ms: u64,
|
||||||
|
#[pyo3(get)]
|
||||||
|
pub ocr_time_ms: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[pymethods]
|
||||||
|
impl PyOcrPdfResult {
|
||||||
|
fn __repr__(&self) -> String {
|
||||||
|
format!(
|
||||||
|
"OcrPdfResult(pages={}, routed_to_ocr={:?}, recommending_hosted={:?})",
|
||||||
|
self.page_count, self.pages_routed_to_ocr, self.pages_recommending_hosted
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Classification wrapper (lightweight)
|
// Classification wrapper (lightweight)
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -362,6 +463,101 @@ fn to_py_err(e: crate::PdfError) -> PyErr {
|
|||||||
PyValueError::new_err(e.to_string())
|
PyValueError::new_err(e.to_string())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
struct PythonOcrOptions {
|
||||||
|
mode: String,
|
||||||
|
page_numbers: Option<Vec<u32>>,
|
||||||
|
password: Option<String>,
|
||||||
|
dpi: f32,
|
||||||
|
minimum_confidence: f32,
|
||||||
|
hosted_recommendation_confidence: f32,
|
||||||
|
model_directory: Option<String>,
|
||||||
|
offline: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build_ocr_options(binding: PythonOcrOptions) -> PyResult<crate::vision::OcrPdfOptions> {
|
||||||
|
let mode = match binding.mode.trim().to_ascii_lowercase().as_str() {
|
||||||
|
"off" => crate::vision::OcrMode::Off,
|
||||||
|
"auto" => crate::vision::OcrMode::Auto,
|
||||||
|
"force" => crate::vision::OcrMode::Force,
|
||||||
|
_ => {
|
||||||
|
return Err(PyValueError::new_err(
|
||||||
|
"mode must be 'off', 'auto', or 'force'",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut options = crate::vision::OcrPdfOptions::new().mode(mode);
|
||||||
|
options.render.dpi = binding.dpi;
|
||||||
|
options.ocr.minimum_confidence = binding.minimum_confidence;
|
||||||
|
options.hosted_recommendation_confidence = binding.hosted_recommendation_confidence;
|
||||||
|
if let Some(pages) = binding.page_numbers {
|
||||||
|
options = options.page_numbers(pages);
|
||||||
|
}
|
||||||
|
if let Some(password) = binding.password {
|
||||||
|
options = options.password(password);
|
||||||
|
}
|
||||||
|
if let Some(directory) = binding.model_directory {
|
||||||
|
options.ocr.model_directory = Some(directory.into());
|
||||||
|
}
|
||||||
|
if binding.offline {
|
||||||
|
options.ocr.model_downloads = crate::vision::ModelDownloadPolicy::Offline;
|
||||||
|
}
|
||||||
|
Ok(options)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn page_content_source_str(source: crate::vision::PageContentSource) -> String {
|
||||||
|
match source {
|
||||||
|
crate::vision::PageContentSource::Native => "native".into(),
|
||||||
|
crate::vision::PageContentSource::Ocr => "ocr".into(),
|
||||||
|
crate::vision::PageContentSource::Fused => "fused".into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn to_py_ocr_result(result: crate::vision::OcrPdfResult) -> PyOcrPdfResult {
|
||||||
|
PyOcrPdfResult {
|
||||||
|
markdown: result.markdown,
|
||||||
|
pages: result
|
||||||
|
.pages
|
||||||
|
.into_iter()
|
||||||
|
.map(|page| {
|
||||||
|
let provenance = page.provenance;
|
||||||
|
PyOcrPageResult {
|
||||||
|
page_number: page.page_number,
|
||||||
|
markdown: page.markdown,
|
||||||
|
provenance: PyOcrPageProvenance {
|
||||||
|
page_number: provenance.page_number,
|
||||||
|
source: page_content_source_str(provenance.source),
|
||||||
|
ocr_model: provenance.ocr_model.map(|model| PyOcrModelIdentity {
|
||||||
|
name: model.name,
|
||||||
|
revision: model.revision,
|
||||||
|
}),
|
||||||
|
render_dpi: provenance.render_dpi,
|
||||||
|
ocr_confidence: provenance.ocr_confidence,
|
||||||
|
timings: PyOcrTimings {
|
||||||
|
render_ms: provenance.timings.render_ms,
|
||||||
|
ocr_ms: provenance.timings.ocr_ms,
|
||||||
|
assembly_ms: provenance.timings.assembly_ms,
|
||||||
|
},
|
||||||
|
warnings: provenance.warnings,
|
||||||
|
hosted_recommended: provenance.hosted_recommended,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
page_count: result.page_count,
|
||||||
|
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
|
||||||
|
pages_routed_to_ocr: result.pages_routed_to_ocr,
|
||||||
|
pages_recommending_hosted: result.pages_recommending_hosted,
|
||||||
|
ocr_reasons_by_page: to_py_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||||
|
pages_with_tables: result.pages_with_tables,
|
||||||
|
pages_with_columns: result.pages_with_columns,
|
||||||
|
is_complex: result.is_complex,
|
||||||
|
processing_time_ms: result.processing_time_ms,
|
||||||
|
render_time_ms: result.render_time_ms,
|
||||||
|
ocr_time_ms: result.ocr_time_ms,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn item_type_str(t: &ItemType) -> String {
|
fn item_type_str(t: &ItemType) -> String {
|
||||||
match t {
|
match t {
|
||||||
ItemType::Text => "text".into(),
|
ItemType::Text => "text".into(),
|
||||||
@@ -502,6 +698,99 @@ fn process_pdf_bytes(data: &[u8], pages: Option<Vec<u32>>) -> PyResult<PyPdfResu
|
|||||||
Ok(to_py_result(result))
|
Ok(to_py_result(result))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Process a PDF file through native extraction and selective OCR.
|
||||||
|
///
|
||||||
|
/// OCR defaults to ``auto`` and only initializes its external runtime and
|
||||||
|
/// model when native quality signals route at least one page. Page numbers
|
||||||
|
/// are 1-indexed. The GIL is released for the complete processing call.
|
||||||
|
#[pyfunction]
|
||||||
|
#[pyo3(signature = (
|
||||||
|
path,
|
||||||
|
*,
|
||||||
|
mode="auto",
|
||||||
|
page_numbers=None,
|
||||||
|
password=None,
|
||||||
|
dpi=150.0,
|
||||||
|
minimum_confidence=0.0,
|
||||||
|
hosted_recommendation_confidence=0.5,
|
||||||
|
model_directory=None,
|
||||||
|
offline=false
|
||||||
|
))]
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn process_pdf_with_ocr(
|
||||||
|
py: Python<'_>,
|
||||||
|
path: String,
|
||||||
|
mode: &str,
|
||||||
|
page_numbers: Option<Vec<u32>>,
|
||||||
|
password: Option<String>,
|
||||||
|
dpi: f32,
|
||||||
|
minimum_confidence: f32,
|
||||||
|
hosted_recommendation_confidence: f32,
|
||||||
|
model_directory: Option<String>,
|
||||||
|
offline: bool,
|
||||||
|
) -> PyResult<PyOcrPdfResult> {
|
||||||
|
let options = build_ocr_options(PythonOcrOptions {
|
||||||
|
mode: mode.to_string(),
|
||||||
|
page_numbers,
|
||||||
|
password,
|
||||||
|
dpi,
|
||||||
|
minimum_confidence,
|
||||||
|
hosted_recommendation_confidence,
|
||||||
|
model_directory,
|
||||||
|
offline,
|
||||||
|
})?;
|
||||||
|
let result = py
|
||||||
|
.allow_threads(move || crate::vision::process_pdf_with_ocr(path, options))
|
||||||
|
.map_err(|error| PyValueError::new_err(error.to_string()))?;
|
||||||
|
Ok(to_py_ocr_result(result))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Process PDF bytes through native extraction and selective OCR.
|
||||||
|
///
|
||||||
|
/// See [`process_pdf_with_ocr`] for options and result semantics.
|
||||||
|
#[pyfunction]
|
||||||
|
#[pyo3(signature = (
|
||||||
|
data,
|
||||||
|
*,
|
||||||
|
mode="auto",
|
||||||
|
page_numbers=None,
|
||||||
|
password=None,
|
||||||
|
dpi=150.0,
|
||||||
|
minimum_confidence=0.0,
|
||||||
|
hosted_recommendation_confidence=0.5,
|
||||||
|
model_directory=None,
|
||||||
|
offline=false
|
||||||
|
))]
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn process_pdf_with_ocr_bytes(
|
||||||
|
py: Python<'_>,
|
||||||
|
data: &[u8],
|
||||||
|
mode: &str,
|
||||||
|
page_numbers: Option<Vec<u32>>,
|
||||||
|
password: Option<String>,
|
||||||
|
dpi: f32,
|
||||||
|
minimum_confidence: f32,
|
||||||
|
hosted_recommendation_confidence: f32,
|
||||||
|
model_directory: Option<String>,
|
||||||
|
offline: bool,
|
||||||
|
) -> PyResult<PyOcrPdfResult> {
|
||||||
|
let options = build_ocr_options(PythonOcrOptions {
|
||||||
|
mode: mode.to_string(),
|
||||||
|
page_numbers,
|
||||||
|
password,
|
||||||
|
dpi,
|
||||||
|
minimum_confidence,
|
||||||
|
hosted_recommendation_confidence,
|
||||||
|
model_directory,
|
||||||
|
offline,
|
||||||
|
})?;
|
||||||
|
let data = data.to_vec();
|
||||||
|
let result = py
|
||||||
|
.allow_threads(move || crate::vision::process_pdf_with_ocr_mem(&data, options))
|
||||||
|
.map_err(|error| PyValueError::new_err(error.to_string()))?;
|
||||||
|
Ok(to_py_ocr_result(result))
|
||||||
|
}
|
||||||
|
|
||||||
/// Fast detection only — no text extraction or markdown.
|
/// Fast detection only — no text extraction or markdown.
|
||||||
#[pyfunction]
|
#[pyfunction]
|
||||||
fn detect_pdf(path: &str) -> PyResult<PyPdfResult> {
|
fn detect_pdf(path: &str) -> PyResult<PyPdfResult> {
|
||||||
@@ -704,6 +993,11 @@ fn extract_structure_elements_bytes(
|
|||||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||||
m.add_class::<PyPdfResult>()?;
|
m.add_class::<PyPdfResult>()?;
|
||||||
m.add_class::<PyPageOcrReasons>()?;
|
m.add_class::<PyPageOcrReasons>()?;
|
||||||
|
m.add_class::<PyOcrModelIdentity>()?;
|
||||||
|
m.add_class::<PyOcrTimings>()?;
|
||||||
|
m.add_class::<PyOcrPageProvenance>()?;
|
||||||
|
m.add_class::<PyOcrPageResult>()?;
|
||||||
|
m.add_class::<PyOcrPdfResult>()?;
|
||||||
m.add_class::<PyPdfClassification>()?;
|
m.add_class::<PyPdfClassification>()?;
|
||||||
m.add_class::<PyTextItem>()?;
|
m.add_class::<PyTextItem>()?;
|
||||||
m.add_class::<PyStructureElement>()?;
|
m.add_class::<PyStructureElement>()?;
|
||||||
@@ -713,6 +1007,8 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
|||||||
m.add_class::<PyPagesExtractionResult>()?;
|
m.add_class::<PyPagesExtractionResult>()?;
|
||||||
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
||||||
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
||||||
|
m.add_function(wrap_pyfunction!(process_pdf_with_ocr, m)?)?;
|
||||||
|
m.add_function(wrap_pyfunction!(process_pdf_with_ocr_bytes, m)?)?;
|
||||||
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
||||||
m.add_function(wrap_pyfunction!(detect_pdf_bytes, m)?)?;
|
m.add_function(wrap_pyfunction!(detect_pdf_bytes, m)?)?;
|
||||||
m.add_function(wrap_pyfunction!(classify_pdf, m)?)?;
|
m.add_function(wrap_pyfunction!(classify_pdf, m)?)?;
|
||||||
|
|||||||
@@ -77,6 +77,50 @@ class TestProcessPdfBytes:
|
|||||||
assert result.markdown is not None
|
assert result.markdown is not None
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# process_pdf_with_ocr / process_pdf_with_ocr_bytes
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
class TestProcessPdfWithOcr:
|
||||||
|
def test_off_mode_has_full_provenance_without_external_runtimes(self):
|
||||||
|
result = pdf_inspector.process_pdf_with_ocr(
|
||||||
|
fixture_path("thermo-freon12.pdf"), mode="off"
|
||||||
|
)
|
||||||
|
assert result.page_count == 3
|
||||||
|
assert len(result.pages) == 3
|
||||||
|
assert result.pages_routed_to_ocr == []
|
||||||
|
assert all(page.provenance.source == "native" for page in result.pages)
|
||||||
|
assert all(page.provenance.ocr_model is None for page in result.pages)
|
||||||
|
assert result.markdown
|
||||||
|
assert "OcrPdfResult" in repr(result)
|
||||||
|
|
||||||
|
def test_auto_mode_skips_external_runtimes_for_clean_text(self):
|
||||||
|
result = pdf_inspector.process_pdf_with_ocr_bytes(
|
||||||
|
fixture_bytes("thermo-freon12.pdf")
|
||||||
|
)
|
||||||
|
assert result.pages_routed_to_ocr == []
|
||||||
|
assert result.render_time_ms == 0
|
||||||
|
assert result.ocr_time_ms == 0
|
||||||
|
|
||||||
|
def test_selected_pages_are_one_indexed(self):
|
||||||
|
result = pdf_inspector.process_pdf_with_ocr(
|
||||||
|
fixture_path("thermo-freon12.pdf"), mode="off", page_numbers=[2]
|
||||||
|
)
|
||||||
|
assert [page.page_number for page in result.pages] == [2]
|
||||||
|
|
||||||
|
def test_rejects_invalid_options(self):
|
||||||
|
with pytest.raises(ValueError, match="mode must be"):
|
||||||
|
pdf_inspector.process_pdf_with_ocr(
|
||||||
|
fixture_path("thermo-freon12.pdf"), mode="sometimes"
|
||||||
|
)
|
||||||
|
with pytest.raises(ValueError, match="page 0"):
|
||||||
|
pdf_inspector.process_pdf_with_ocr(
|
||||||
|
fixture_path("thermo-freon12.pdf"),
|
||||||
|
mode="off",
|
||||||
|
page_numbers=[0],
|
||||||
|
)
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# detect_pdf / detect_pdf_bytes
|
# detect_pdf / detect_pdf_bytes
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
Reference in New Issue
Block a user