Compare commits

..
Author SHA1 Message Date
Abimael Martell 162c5cbf10 refactor(vision): expose OCR API 2026-08-16 21:30:07 -07:00
Abimael Martell 4e4cfdc74c fix(vision): harden OCR API 2026-08-16 21:30:07 -07:00
Abimael Martell 24e66245cb feat(vision): expose OCR pipeline 2026-08-16 21:30:07 -07:00
Abimael Martell cee7b6c381 refactor(vision): use OCR fusion terminology 2026-08-16 21:30:07 -07:00
Abimael Martell b5a63d7036 fix(vision): make OCR fusion conservative 2026-08-16 21:30:07 -07:00
Abimael Martell 6829397bce feat(vision): fuse OCR output 2026-08-16 21:30:07 -07:00
Abimael Martell fc16133a58 refactor(vision): name routed OCR results 2026-08-16 21:30:06 -07:00
Abimael Martell 41a6a67c03 fix(vision): serialize model acquisition 2026-08-16 21:30:06 -07:00
Abimael Martell 58fe5a0224 feat(vision): add OCR routing 2026-08-16 21:30:06 -07:00
Abimael Martell c4c969a562 refactor(vision): use OCR engine terminology 2026-08-16 21:30:06 -07:00
Abimael Martell bf0cd8decb fix(vision): harden OAR runtime loading 2026-08-16 21:30:06 -07:00
Abimael Martell 232b4cdef5 feat(vision): add OAR OCR engine 2026-08-16 21:30:06 -07:00
Abimael Martell e3f5429638 refactor(vision): use OCR terminology 2026-08-16 21:30:06 -07:00
Abimael Martell f6cbe979f6 fix(vision): harden OCR contracts 2026-08-16 21:30:06 -07:00
Abimael Martell 3f43745313 feat(vision): add OCR contracts 2026-08-16 21:30:06 -07:00
Abimael Martell a012cb65a6 docs(render): use OCR terminology 2026-08-16 21:30:06 -07:00
Abimael Martell 6567e1ab2d fix(render): honor PDFium row stride 2026-08-16 21:30:06 -07:00
Abimael Martell 3af409d27f feat(render): add optional PDFium page rendering 2026-08-16 21:30:06 -07:00
42 changed files with 438 additions and 6286 deletions
-2
View File
@@ -1,2 +0,0 @@
*.pdf binary
tests/snapshots/*.md text eol=lf
+1 -184
View File
@@ -9,9 +9,6 @@ on:
env:
CARGO_TERM_COLOR: always
permissions:
contents: read
jobs:
test:
name: Test
@@ -71,12 +68,9 @@ jobs:
with:
key: clippy
- name: Run default clippy
- name: Run clippy
run: cargo clippy -- -D warnings
- name: Run OCR clippy
run: cargo clippy --features ocr -- -D warnings
build:
name: Build
runs-on: ${{ matrix.os }}
@@ -99,183 +93,6 @@ jobs:
- name: Build
run: cargo build --release --verbose
ocr:
name: OCR (${{ matrix.os }})
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, macos-latest, windows-latest]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- name: Install Rust
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
with:
toolchain: stable
- name: Cache cargo
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
with:
key: ocr-${{ matrix.os }}
- name: Test optional OCR feature
run: cargo test --features ocr
ocr-runtime:
name: OCR runtime smoke
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
- name: Install Rust
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
with:
toolchain: stable
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
with:
python-version: '3.12'
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: latest
- name: Cache cargo
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
with:
key: ocr-runtime
workspaces: |
. -> target
napi -> target
- name: Install PDFium
shell: bash
run: |
archive="$RUNNER_TEMP/firecrawl-pdfium-linux-x64.tgz"
directory="$RUNNER_TEMP/firecrawl-pdfium"
curl --fail --location --silent --show-error \
https://github.com/firecrawl/pdfium-rs/releases/download/native-v7988/firecrawl-pdfium-linux-x64.tgz \
--output "$archive"
echo "6248189e07bbc33cdeb31976c539a88614307c8a19f3276dbd018efbe5b4a2a2 $archive" | sha256sum --check
mkdir -p "$directory"
tar -xzf "$archive" -C "$directory"
pdfium_path="$(find "$directory" -type f -name 'libpdfium.so' -print -quit)"
test -n "$pdfium_path"
echo "PDFIUM_LIB_PATH=$pdfium_path" >> "$GITHUB_ENV"
- name: Install ONNX Runtime
shell: bash
run: |
archive="$RUNNER_TEMP/onnxruntime-linux-x64-1.27.0.tgz"
directory="$RUNNER_TEMP/onnxruntime"
curl --fail --location --silent --show-error \
https://github.com/microsoft/onnxruntime/releases/download/v1.27.0/onnxruntime-linux-x64-1.27.0.tgz \
--output "$archive"
echo "547e40a48f1fe73e3f812d7c88a948612c23f896b91e4e2ee1e232d7b468246f $archive" | sha256sum --check
mkdir -p "$directory"
tar -xzf "$archive" -C "$directory"
ort_path="$(find "$directory" -type f -name 'libonnxruntime.so*' -print -quit)"
test -n "$ort_path"
echo "ORT_DYLIB_PATH=$ort_path" >> "$GITHUB_ENV"
- name: Configure isolated model cache
shell: bash
run: echo "PDF_INSPECTOR_MODEL_CACHE=$RUNNER_TEMP/pdf-inspector-models" >> "$GITHUB_ENV"
- name: Build OCR CLI
run: cargo build --features ocr --bin pdf2md
- name: Test PDFium runtime
run: cargo test --features ocr --test local_render_tests
- name: Provision OCR model cache
shell: bash
run: |
target/debug/pdf2md \
tests/fixtures/scan_with_native_header_text.pdf \
--ocr force \
--json > /dev/null
- name: Run OCR CLI
shell: bash
run: |
target/debug/pdf2md \
tests/fixtures/scan_with_native_header_text.pdf \
--ocr auto \
--ocr-offline \
--json > "$RUNNER_TEMP/ocr-result.json"
- name: Validate OCR JSON contract
shell: bash
run: |
python3 - <<'PY'
import json
import os
from pathlib import Path
result = json.loads(
(Path(os.environ["RUNNER_TEMP"]) / "ocr-result.json").read_text()
)
assert result["schema_version"] == 1
assert result["pages_routed_to_ocr"] == [1]
assert result["pages_recommending_hosted"] == []
assert result["pages"][0]["source"] in {"ocr", "fused"}
assert result["pages"][0]["markdown"].strip()
assert "layout_ms" not in result["pages"][0]["timings"]
PY
- name: Run OCR launch smoke set
shell: bash
run: |
export PDF_INSPECTOR_OCR_TEST_MODELS="$PDF_INSPECTOR_MODEL_CACHE/pp-ocrv6-small/oar-ocr-v0.7.0"
cargo test --features ocr --test ocr_tests -- --nocapture
- name: Build Node binding
working-directory: napi
run: |
bun install --frozen-lockfile
bunx napi build --platform --release
- name: Run Node OCR binding
shell: bash
run: |
node --input-type=module - <<'JS'
import { readFileSync } from 'node:fs'
import { processPdfWithOcr } from './napi/index.js'
const pdf = readFileSync('tests/fixtures/scan_with_native_header_text.pdf')
const result = await processPdfWithOcr(pdf, { offline: true })
if (JSON.stringify(result.pagesRoutedToOcr) !== '[1]') throw new Error('unexpected OCR route')
if (result.pagesRecommendingHosted.length !== 0) throw new Error('unexpected hosted recommendation')
if (!['Ocr', 'Fused'].includes(result.pages[0].provenance.source)) throw new Error('unexpected source')
if (!result.pages[0].markdown.trim()) throw new Error('empty OCR markdown')
JS
- name: Build and install Python binding
shell: bash
run: |
python -m pip install 'maturin>=1,<2'
maturin build --release --out "$RUNNER_TEMP/python-wheels"
python -m pip install "$RUNNER_TEMP"/python-wheels/*.whl
- name: Run Python OCR binding
shell: bash
run: |
python - <<'PY'
import pdf_inspector
result = pdf_inspector.process_pdf_with_ocr(
"tests/fixtures/scan_with_native_header_text.pdf",
offline=True,
)
assert result.pages_routed_to_ocr == [1]
assert result.pages_recommending_hosted == []
assert result.pages[0].provenance.source in {"ocr", "fused"}
assert result.pages[0].markdown.strip()
PY
wasm:
name: WebAssembly
runs-on: ubuntu-latest
-5
View File
@@ -86,14 +86,12 @@ jobs:
name: Build ${{ matrix.target }}
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
include:
- os: ubuntu-latest
target: x86_64-unknown-linux-gnu
- os: ubuntu-latest
target: aarch64-unknown-linux-gnu
docker-options: -e CFLAGS_aarch64_unknown_linux_gnu=-D__ARM_ARCH=8
# macos-13 was retired by GitHub; macos-15-intel is the remaining
# Intel runner label (available through 2027).
- os: macos-15-intel
@@ -115,9 +113,6 @@ jobs:
target: ${{ matrix.target }}
args: --release --out dist
manylinux: auto
# The manylinux AArch64 GCC omits this macro while preprocessing
# ring's assembly. AArch64 is ARMv8 by definition.
docker-options: ${{ matrix.docker-options }}
- name: Upload wheel
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
-6
View File
@@ -60,7 +60,6 @@ jobs:
name: Build ${{ matrix.target }}
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
include:
- os: ubuntu-latest
@@ -71,7 +70,6 @@ jobs:
- os: ubuntu-latest
target: aarch64-unknown-linux-gnu
build-flags: --use-napi-cross
cflags: -D__ARM_ARCH=8
- os: ubuntu-latest
target: x86_64-unknown-linux-musl
build-flags: -x
@@ -128,10 +126,6 @@ jobs:
- name: Build native addon
working-directory: napi
env:
# napi-cross's old AArch64 GCC omits this predefined macro while
# preprocessing ring's assembly. AArch64 is ARMv8 by definition.
CFLAGS_aarch64_unknown_linux_gnu: ${{ matrix.cflags }}
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
- name: Upload native binary
+2 -2
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector"
version = "1.15.0"
version = "1.14.2"
edition = "2021"
autobins = false
authors = ["Firecrawl Team"]
@@ -80,7 +80,7 @@ tempfile = "3.3"
[features]
default = []
python = ["pyo3", "ocr"]
python = ["pyo3"]
vision = []
model-cache = ["vision", "dep:dirs", "dep:fs2", "dep:sha2", "dep:windows-sys"]
model-download = ["model-cache", "dep:ureq"]
+6 -31
View File
@@ -5,7 +5,7 @@
[![PyPI](https://img.shields.io/pypi/v/pdf-inspector.svg)](https://pypi.org/project/pdf-inspector/)
[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
Fast Rust library for PDF classification and text extraction. By default it detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown without OCR. Native Rust and CLI consumers can opt into selective OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
@@ -18,10 +18,9 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
- **CID font support** — ToUnicode CMap decoding for Type0/Identity-H fonts, UTF-16BE, UTF-8, and Latin-1 encodings.
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
- **Selective OCR** — Rust, CLI, Python, and Node can render only pages that need OCR, run PP-OCRv6 Small locally, and preserve per-page provenance and hosted-fallback recommendations.
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
- **Lightweight by default** — The default Rust and browser builds remain pure extraction. Native Python and Node packages include the OCR integration, but PDFium, ONNX Runtime, and model files remain external and are touched only when a page is routed to OCR.
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
## Benchmark
@@ -48,7 +47,8 @@ Use the [paired benchmark harness](docs/benchmarking.md) to compare two local bu
### Python
```bash
pip install pdf-inspector
pip install maturin
maturin develop --release
```
```python
@@ -57,10 +57,6 @@ import pdf_inspector
result = pdf_inspector.process_pdf("document.pdf")
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
print(result.markdown) # Markdown string or None
# Selective OCR; clean text PDFs do not load the external OCR runtime.
ocr = pdf_inspector.process_pdf_with_ocr("document.pdf")
print(ocr.pages_routed_to_ocr)
```
> Full API reference: [docs/python.md](docs/python.md)
@@ -73,15 +69,11 @@ npm install @firecrawl/pdf-inspector
```javascript
import { readFileSync } from 'fs';
import { processPdf, processPdfWithOcr } from '@firecrawl/pdf-inspector';
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector';
const pdf = readFileSync('document.pdf');
const result = processPdf(pdf);
const result = processPdf(readFileSync('document.pdf'));
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
console.log(result.markdown); // Markdown string or null
const ocr = await processPdfWithOcr(pdf); // selective OCR, off the event loop
console.log(ocr.pagesRoutedToOcr);
```
> Full API reference: [napi/README.md](napi/README.md)
@@ -168,23 +160,6 @@ detect-pdf document.pdf --json
detect-pdf document.pdf --analyze --json
```
Rust and CLI consumers opt into OCR at build time:
```bash
cargo install pdf-inspector --features ocr --bin pdf2md
PDFIUM_LIB_PATH=/path/to/libpdfium ORT_DYLIB_PATH=/path/to/libonnxruntime \
pdf2md scan.pdf --ocr auto --json
```
The OCR JSON envelope is versioned and reports routed pages, per-page source
and confidence, warnings, and pages recommended for the hosted document
pipeline. Native Python and Node packages expose the same pipeline without a
source-build feature. All native entry points still require separately
installed PDFium and ONNX Runtime libraries only when OCR is routed. See the
[OCR runtime setup guide](docs/ocr-runtime.md) for pinned downloads, platform
support, model-cache behavior, and hosted-fallback integration. See the
[Rust API guide](docs/rust-api.md#complete-ocr-api) for lower-level controls.
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
## Architecture
-97
View File
@@ -1,97 +0,0 @@
# OCR runtime setup
Selective OCR is available from the Rust library and CLI, Python, and Node.js.
Clean native-text documents do not load an OCR dependency or download a model.
When `auto` routes at least one page, the process needs PDFium, ONNX Runtime,
and the pinned PP-OCRv6 Small model set.
## Validated versions
The reproducible runtime path uses these builds:
- [Firecrawl PDFium `native-v7988`](https://github.com/firecrawl/pdfium-rs/releases/tag/native-v7988),
containing PDFium `153.0.7988.0`
- [ONNX Runtime `1.27.0`](https://github.com/microsoft/onnxruntime/releases/tag/v1.27.0)
- PP-OCRv6 Small artifact revision `oar-ocr-v0.7.0`
Use these versions for the reproducible path. Other compatible shared-library
builds may work, but are not part of the release smoke test.
## Install the shared libraries
Download and extract the matching archives:
| Platform | PDFium asset | ONNX Runtime asset |
|---|---|---|
| Linux x64 | `firecrawl-pdfium-linux-x64.tgz` | `onnxruntime-linux-x64-1.27.0.tgz` |
| Linux ARM64 | `firecrawl-pdfium-linux-arm64.tgz` | `onnxruntime-linux-aarch64-1.27.0.tgz` |
| macOS Apple Silicon | `firecrawl-pdfium-mac-arm64.tgz` | `onnxruntime-osx-arm64-1.27.0.tgz` |
| Windows x64 | `firecrawl-pdfium-win-x64.tgz` | `onnxruntime-win-x64-1.27.0.zip` |
The PDFium release publishes `SHA256SUMS`, build provenance, license files,
and an SPDX document for every platform archive. GitHub publishes a SHA-256
digest with each ONNX Runtime asset.
Point pdf-inspector at the extracted shared libraries when they are not on the
platform library search path:
```bash
export PDFIUM_LIB_PATH=/absolute/path/to/libpdfium.so
export ORT_DYLIB_PATH=/absolute/path/to/libonnxruntime.so
pdf2md scan.pdf --ocr auto --json
```
On macOS the filenames end in `.dylib`. On Windows, use PowerShell and point
the variables at `pdfium.dll` and `onnxruntime.dll`:
```powershell
$env:PDFIUM_LIB_PATH = "C:\absolute\path\to\pdfium.dll"
$env:ORT_DYLIB_PATH = "C:\absolute\path\to\onnxruntime.dll"
pdf2md scan.pdf --ocr auto --json
```
The native extraction packages also support platforms without these exact
runtime assets. In particular, the Python package has an Intel macOS wheel,
but ONNX Runtime 1.27.0 does not publish an Intel macOS archive; local OCR on
that target requires a compatible custom ONNX Runtime build.
The full OCR path is exercised end to end on Linux x64 in CI. macOS and
Windows compile and run the feature's platform-independent tests, while their
external-runtime paths should be treated as preview until equivalent smoke
jobs are added.
## Model cache and offline mode
The first routed page downloads and SHA-256-verifies three pinned artifacts:
the detection model, recognition model, and character dictionary. Together
they are about 31 MB. They are stored below the platform cache directory.
Set `PDF_INSPECTOR_MODEL_CACHE` to choose a managed cache root.
For hermetic deployments, populate the model directory ahead of time and use
the language-specific offline option:
- CLI: `--ocr-offline --ocr-model-dir /models/pp-ocrv6-small`
- Rust: `ModelDownloadPolicy::Offline` with `OcrOptions::model_directory`
- Python: `offline=True, model_directory="/models/pp-ocrv6-small"`
- Node.js: `offline: true, modelDirectory: "/models/pp-ocrv6-small"`
The model artifacts come from
[`GreatV/oar-ocr`](https://github.com/GreatV/oar-ocr/releases/tag/v0.7.0),
whose OCR implementation and upstream PaddleOCR project use Apache-2.0
licensing. Models are downloaded at runtime and are not embedded in any
pdf-inspector package.
## Hosted fallback boundary
`pages_recommending_hosted` is available after the local pipeline completes.
It marks pages whose completed OCR result is empty, low-confidence, or still
appears incomplete.
Setup and execution failures happen before that result exists. A missing or
incompatible PDFium/ONNX Runtime library, failed model acquisition, or OCR
execution error is returned as an error. A downstream integration that has a
hosted parser should catch that error and route the document to the hosted
path. This keeps deployment problems distinct from page-quality judgments.
In `auto`, documents with no routed pages return successfully without touching
PDFium, ONNX Runtime, the model cache, or the network.
+2 -65
View File
@@ -1,6 +1,6 @@
# pdf-inspector
Fast PDF classification, text extraction, and selective OCR. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts clean native and OCR results to Markdown. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
@@ -10,8 +10,7 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
- **Selective OCR** — `auto` routes only pages rejected by native extraction; `force` OCRs every selected page; `off` keeps the result/provenance contract without external runtime work.
- **External artifacts** — the wheel embeds no OCR models, PDFium, or ONNX Runtime; clean `auto` requests never load or download them.
- **Lightweight** — native Rust core, no ML models, no external services; ships type stubs.
## Benchmark
@@ -40,14 +39,6 @@ pip install maturin
maturin develop --release
```
OCR calls that route work require compatible PDFium and ONNX Runtime shared
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
platform library search path. The pinned OCR model set is downloaded and
checksum-verified on the first routed page; use `offline=True` with a warm
cache or `model_directory` to prohibit network access. See the
[OCR runtime setup guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md)
for pinned downloads, supported platforms, and hosted-fallback behavior.
## Usage
```python
@@ -90,19 +81,6 @@ for page in result.pages:
# Restrict to specific 0-indexed pages (preserves caller order)
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
# One-call selective OCR. This releases the GIL while processing.
ocr = pdf_inspector.process_pdf_with_ocr("document.pdf")
for page in ocr.pages:
print(page.page_number, page.provenance.source)
# Restrict OCR processing to 1-indexed PDF pages and prohibit downloads.
ocr = pdf_inspector.process_pdf_with_ocr(
"document.pdf",
page_numbers=[1, 3],
model_directory="/opt/models/pp-ocrv6-small",
offline=True,
)
# Structure-tree elements from tagged PDFs (empty list when untagged).
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
# against extract_text_with_positions — e.g. to recover real heading levels:
@@ -121,8 +99,6 @@ headings = [
|---|---|
| `process_pdf(path, pages=None)` | Full processing (detect + extract + markdown) |
| `process_pdf_bytes(data, pages=None)` | Full processing from bytes |
| `process_pdf_with_ocr(path, **options)` | Native extraction + selective OCR with provenance |
| `process_pdf_with_ocr_bytes(data, **options)` | Native extraction + selective OCR from bytes |
| `detect_pdf(path)` | Fast detection only (returns PdfResult) |
| `detect_pdf_bytes(data)` | Fast detection from bytes |
| `classify_pdf(path)` | Lightweight classification (returns PdfClassification) |
@@ -161,45 +137,6 @@ class PageOcrReasons: # per-page OCR diagnostics
page: int # 1-indexed
reasons: list[str] # machine-readable reason identifiers
class OcrModelIdentity:
name: str # model family/name
revision: str # immutable artifact-set revision
class OcrTimings: # per-page processing stages
render_ms: int
ocr_ms: int
assembly_ms: int
class OcrPageProvenance:
page_number: int # 1-indexed
source: Literal["native", "ocr", "fused"]
ocr_model: OcrModelIdentity | None
render_dpi: float | None
ocr_confidence: float | None
timings: OcrTimings
warnings: list[str]
hosted_recommended: bool
class OcrPageResult:
page_number: int # 1-indexed
markdown: str
provenance: OcrPageProvenance
class OcrPdfResult: # process_pdf_with_ocr / bytes
markdown: str
pages: list[OcrPageResult]
page_count: int
pages_recommended_for_ocr: list[int]
pages_routed_to_ocr: list[int]
pages_recommending_hosted: list[int]
ocr_reasons_by_page: list[PageOcrReasons]
pages_with_tables: list[int]
pages_with_columns: list[int]
is_complex: bool
processing_time_ms: int
render_time_ms: int
ocr_time_ms: int
class PdfClassification: # classify_pdf
pdf_type: str
page_count: int
+23 -65
View File
@@ -1,6 +1,6 @@
# pdf-inspector
Fast PDF classification and text extraction. The default build detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown without OCR. It is pure Rust, has no ML models or external services, and uses [lopdf](https://crates.io/crates/lopdf) for PDF parsing. Native Rust and CLI consumers can opt into selective OCR. Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector/).
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. The default build is pure Rust, has no ML models or external services, and uses [lopdf](https://crates.io/crates/lopdf) for PDF parsing. Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector/).
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
@@ -123,10 +123,10 @@ The native-only `vision` feature exposes the stable seam used by OCR
integrations without selecting or embedding an inference runtime. The
separate `model-cache` feature adds pinned artifact management:
- `PageRenderer` and `OcrEngine` traits;
- `PageRenderer`, `OcrEngine`, and `LayoutEngine` traits;
- renderer-neutral owned page buffers and affine pixel↔PDF transforms;
- `OcrOptions` and opt-in `Off`/`Auto`/`Force` routing modes;
- positioned OCR results and per-page provenance types; and
- positioned OCR/layout results and per-page provenance types; and
- a versioned PP-OCRv6 Small manifest with checksum-verified, locked, atomic
model-cache installation and explicit offline-directory overrides.
@@ -135,9 +135,9 @@ separate `model-cache` feature adds pinned artifact management:
pdf-inspector = { version = "1", features = ["vision", "model-cache"] }
```
The OCR contracts preserve existing behavior by default: OCR is `Off` and
model resolution is never reached. `ModelStore` itself does not access the
network. The optional `model-download` feature provides an
The OCR contracts preserve existing behavior by default: OCR is `Off`, learned
layout is disabled, and model resolution is never reached. `ModelStore` itself
does not access the network. The optional `model-download` feature provides an
HTTPS downloader that streams pinned artifacts into the checksum-verified
cache only after routing has selected OCR work. Offline consumers set an
explicit model directory and `ModelDownloadPolicy::Offline`. Renderer-only
@@ -171,10 +171,9 @@ and `PdfiumRenderer` implements the renderer-neutral `PageRenderer` trait.
pdf-inspector = { version = "1", features = ["render-pdfium"] }
```
PDFium is loaded at runtime and is not bundled into the crate. Set
`PDFIUM_LIB_PATH` to the platform shared library, place that library next to
the executable, or use another discovery route supported by
`firecrawl-pdfium`. A load failure reports this prerequisite directly.
PDFium is loaded at runtime. Set `PDFIUM_LIB_PATH`, place its shared library
next to the executable, or use another discovery route supported by
`firecrawl-pdfium`.
```rust
use pdf_inspector::vision::{PdfiumRenderer, RenderOptions};
@@ -206,11 +205,9 @@ The native-only `ocr-oar` feature adds a CPU PP-OCRv6 Small implementation of
not enable model auto-download, ONNX Runtime download, or PDF rendering. Model
files remain external, must match the pinned manifest, and are opened only
after `ModelStore` verifies their exact size and SHA-256 digest. Install an
ONNX Runtime shared library separately and set `ORT_DYLIB_PATH` to its full
path when it is not available through the platform library search path. The
runtime is resolved only when an OCR engine is first constructed; clean
`Auto` requests do not require it. The feature currently requires Rust 1.95
or newer, matching OAR 0.9.1's MSRV.
ONNX Runtime shared library separately and set `ORT_DYLIB_PATH` when it is not
available through the platform library search path. The feature currently
requires Rust 1.95 or newer, matching OAR 0.9.1's MSRV.
```toml
[dependencies]
@@ -332,10 +329,7 @@ let fused = fuse_ocr_pages(
for page in &fused.pages {
println!("{}", page.markdown);
if page.provenance.hosted_recommended {
eprintln!(
"page {} needs the hosted document pipeline",
page.page_number,
);
eprintln!("page {} needs the hosted document pipeline", page.page + 1);
}
}
```
@@ -360,11 +354,15 @@ pdf-inspector = { version = "1", features = ["ocr"] }
```
```rust
use pdf_inspector::vision::{process_pdf_with_ocr, OcrPdfOptions};
use pdf_inspector::vision::{
process_pdf_with_ocr, OcrMode, OcrPdfOptions,
};
let result = process_pdf_with_ocr(
"document.pdf",
OcrPdfOptions::auto().page_numbers([1, 2, 3]),
OcrPdfOptions::new()
.mode(OcrMode::Auto)
.pages([1, 2, 3]),
)?;
println!("{}", result.markdown);
@@ -379,51 +377,12 @@ Native extraction always runs first. In `Auto`, a clean PDF returns before
PDFium loading, model-cache access, HTTP, or OAR initialization. Model files
remain external and the default crate feature set remains unchanged. `Off`
provides the same native-only behavior through the OCR result/provenance
shape; `Force` renders every selected page. OCR uses the existing deterministic
table, column, reading-order, and Markdown assembly path; no learned layout
model is included.
The [OCR runtime setup guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md)
lists the pinned PDFium and ONNX Runtime builds, environment variables, model
cache behavior, and the error boundary downstream hosted fallbacks should use.
For ambiguous mixed pages, `Auto` privately retains clean native fragments
instead of discarding them when OCR is selected. After recognition it compares
script-agnostic text quality, OCR confidence, character overlap, and material
new coverage. Exact native text wins over a duplicate or weak OCR hypothesis;
complementary image-backed text is fused; and pages where both candidates are
weak recommend the hosted document pipeline. A page routed because native
coverage appeared incomplete also recommends hosted processing when confident
OCR only duplicates the retained fragment: the agreement preserves trustworthy
text, but neither hypothesis proves full-page coverage. Public native-only
extraction continues to suppress pages marked unreliable, and clean text
documents pay no renderer or model-initialization cost.
In `Auto`, pages routed only for suspicious font encoding or vectorized text
first get a bounded positioned-text probe through PDFium. A credible recovered
text layer with sufficient geometric page coverage skips rasterization and
model loading for that page; garbled, partial, or insubstantial recovery
continues through OCR. Recovered tables are reflected in the same document
metadata as tables found by the primary extractor.
The one-call API keeps the most recently used verified OCR engine in process.
Long-lived workers therefore verify the pinned artifacts and build the ONNX
sessions once, then reuse those loaded sessions across documents. The cache is
bounded to one model configuration and keyed by normalized model/runtime paths
plus the pinned manifest revision and artifact digests; switching the model
directory, runtime library, or compiled manifest replaces it. An active engine
owns the model data it already verified, so mutating artifacts in place does
not hot-reload a running process; restart the process when intentionally
replacing files at the same paths. CPU inference uses at most four intra-op
threads per ONNX session so a single small page does not oversubscribe larger
hosts, and recognizes variable-width line crops individually to avoid
padding-heavy CPU batches. The high-level pipeline renders and fuses at most
four routed pages at a time, bounding bitmap memory on long documents.
shape; `Force` renders every selected page. Learned layout intentionally
returns an explicit unsupported error in this lightweight pipeline.
Build the CLI with the same opt-in feature:
```bash
cargo install pdf-inspector --features ocr --bin pdf2md
cargo build --release --features ocr --bin pdf2md
pdf2md document.pdf --ocr auto --raw
pdf2md document.pdf --ocr auto --json
@@ -432,11 +391,10 @@ pdf2md document.pdf --ocr auto --ocr-offline --ocr-model-dir /opt/models/pp-ocrv
CLI controls include `--ocr-dpi`, `--ocr-min-confidence`,
`--ocr-hosted-threshold`, `--select-pages`, and the existing encrypted-PDF
`--password` option. JSON output has `schema_version: 1` and includes per-page Markdown, source/model
`--password` option. JSON output includes per-page Markdown, source/model
provenance, confidence, timings, warnings, routed pages, and hosted-fallback
recommendations. Page numbers in `OcrPdfResult` and its per-page provenance
are 1-indexed, matching the PDF page numbers accepted by
`OcrPdfOptions::page_numbers`.
are 1-indexed, matching the PDF page numbers accepted by `OcrPdfOptions::pages`.
Extract per-page Markdown (one string per page, plus document-wide layout
metadata):
+32 -2255
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -1,13 +1,13 @@
[package]
name = "pdf-inspector-napi"
version = "1.15.0"
version = "1.14.2"
edition = "2021"
[lib]
crate-type = ["cdylib"]
[dependencies]
pdf-inspector = { path = "..", features = ["ocr"] }
pdf-inspector = { path = ".." }
napi = { version = "3.0.0", features = ["serde-json"] }
napi-derive = "3.0.0"
+1 -52
View File
@@ -10,8 +10,7 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
- **Selective OCR** — `Auto` routes only pages rejected by native extraction and returns source/model provenance plus hosted-fallback recommendations.
- **External artifacts** — the native package embeds no OCR models, PDFium, or ONNX Runtime; clean `Auto` requests never load or download them.
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~56 MB platform binary, TypeScript definitions included.
## Benchmark
@@ -37,42 +36,8 @@ bun add @firecrawl/pdf-inspector
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
OCR calls that route work require compatible PDFium and ONNX Runtime shared
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
platform library search path. The pinned OCR model set is downloaded and
checksum-verified on the first routed page; use `offline: true` with a warm
cache or `modelDirectory` to prohibit network access. See the
[OCR runtime setup guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md)
for pinned downloads, supported platforms, and hosted-fallback behavior.
## API
### `processPdfWithOcr(buffer: Buffer, options?: OcrOptions): Promise<OcrPdfResult>`
Run native extraction first and OCR only the pages selected by its quality
signals. The default mode is `Auto`; `Off` returns the same detailed result
shape without external runtime work, and `Force` OCRs every selected page.
The work runs on the libuv thread pool and never blocks Node's event loop.
```typescript
import { OcrMode, processPdfWithOcr } from '@firecrawl/pdf-inspector'
const result = await processPdfWithOcr(pdf, {
mode: OcrMode.Auto,
pageNumbers: [1, 3], // 1-indexed
})
for (const page of result.pages) {
console.log(page.pageNumber, page.provenance.source)
}
console.log(result.pagesRoutedToOcr)
console.log(result.pagesRecommendingHosted)
```
For offline deployments, pass `modelDirectory` and `offline: true`. Other
controls include `dpi`, `minimumConfidence`,
`hostedRecommendationConfidence`, and `password`.
### `classifyPdf(buffer: Buffer): PdfClassification`
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
@@ -159,22 +124,6 @@ interface RegionText {
needsOcr: boolean // true when text is unreliable
ocrReason?: string // "suspected_garbled_text" when known
}
interface OcrPdfResult {
markdown: string
pages: OcrPageResult[] // 1-indexed pages + provenance
pageCount: number
pagesRecommendedForOcr: number[]
pagesRoutedToOcr: number[]
pagesRecommendingHosted: number[]
ocrReasonsByPage: PageOcrReasons[]
pagesWithTables: number[]
pagesWithColumns: number[]
isComplex: boolean
processingTimeMs: number
renderTimeMs: number
ocrTimeMs: number
}
```
## Platforms
+6 -6
View File
@@ -8,12 +8,12 @@
"@napi-rs/cli": "^3.4.1",
},
"optionalDependencies": {
"@firecrawl/pdf-inspector-darwin-arm64": "1.15.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.15.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.15.0",
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.15.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.15.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.15.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.2",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.2",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.2",
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.2",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.2",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.2",
},
},
},
+7 -7
View File
@@ -1,6 +1,6 @@
{
"name": "@firecrawl/pdf-inspector",
"version": "1.15.0",
"version": "1.14.2",
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
"main": "index.js",
"types": "index.d.ts",
@@ -52,11 +52,11 @@
"@napi-rs/cli": "^3.4.1"
},
"optionalDependencies": {
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.15.0",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.15.0",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.15.0",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.15.0",
"@firecrawl/pdf-inspector-darwin-arm64": "1.15.0",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.15.0"
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.2",
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.2",
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.2",
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.2",
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.2",
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.2"
}
}
-241
View File
@@ -27,26 +27,6 @@ pub enum ItemType {
FormField,
}
/// Selects when OCR runs.
#[napi(string_enum)]
#[derive(Clone, Copy)]
pub enum OcrMode {
/// Never run OCR; return the native extraction in the OCR result shape.
Off,
/// Run OCR only on pages selected by the native quality signals.
Auto,
/// Run OCR on every selected page.
Force,
}
/// How final page content was sourced.
#[napi(string_enum)]
pub enum PageContentSource {
Native,
Ocr,
Fused,
}
// ---------------------------------------------------------------------------
// Result types
// ---------------------------------------------------------------------------
@@ -148,84 +128,6 @@ pub struct VectorGridDetectionJs {
pub cell_bboxes: Vec<Vec<f64>>,
}
/// Options for one-call native extraction with selective OCR.
#[napi(object)]
#[derive(Clone)]
pub struct OcrOptions {
/// OCR routing behavior. Defaults to Auto.
pub mode: Option<OcrMode>,
/// Optional 1-indexed page selection.
pub page_numbers: Option<Vec<u32>>,
/// Password for an encrypted PDF.
pub password: Option<String>,
/// Page rasterization resolution. Defaults to 150 DPI.
pub dpi: Option<f64>,
/// Drop OCR spans below this inclusive 0-1 threshold.
pub minimum_confidence: Option<f64>,
/// Recommend hosted parsing below this inclusive 0-1 page confidence.
pub hosted_recommendation_confidence: Option<f64>,
/// Directory containing an offline OCR model set.
pub model_directory: Option<String>,
/// Disable model downloads and require a model directory or warm cache.
pub offline: Option<bool>,
}
/// Exact OCR model identity retained in page provenance.
#[napi(object)]
pub struct OcrModelIdentity {
pub name: String,
pub revision: String,
}
/// Per-page OCR processing timings.
#[napi(object)]
pub struct OcrTimings {
pub render_ms: u32,
pub ocr_ms: u32,
pub assembly_ms: u32,
}
/// Source, model, confidence, and fallback metadata for one page.
#[napi(object)]
pub struct OcrPageProvenance {
/// 1-indexed page number.
pub page_number: u32,
pub source: PageContentSource,
pub ocr_model: Option<OcrModelIdentity>,
pub render_dpi: Option<f64>,
pub ocr_confidence: Option<f64>,
pub timings: OcrTimings,
pub warnings: Vec<String>,
pub hosted_recommended: bool,
}
/// Final Markdown and provenance for one page.
#[napi(object)]
pub struct OcrPageResult {
/// 1-indexed page number.
pub page_number: u32,
pub markdown: String,
pub provenance: OcrPageProvenance,
}
/// Complete native/OCR Markdown output.
#[napi(object)]
pub struct OcrPdfResult {
pub markdown: String,
pub pages: Vec<OcrPageResult>,
pub page_count: u32,
pub pages_recommended_for_ocr: Vec<u32>,
pub pages_routed_to_ocr: Vec<u32>,
pub pages_recommending_hosted: Vec<u32>,
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
pub pages_with_tables: Vec<u32>,
pub pages_with_columns: Vec<u32>,
pub is_complex: bool,
pub processing_time_ms: u32,
pub render_time_ms: u32,
pub ocr_time_ms: u32,
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
@@ -266,103 +168,6 @@ fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<
.collect()
}
fn to_core_ocr_options(options: Option<OcrOptions>) -> pdf_inspector::vision::OcrPdfOptions {
let mut result = pdf_inspector::vision::OcrPdfOptions::auto();
let Some(options) = options else {
return result;
};
if let Some(mode) = options.mode {
result.ocr.mode = match mode {
OcrMode::Off => pdf_inspector::vision::OcrMode::Off,
OcrMode::Auto => pdf_inspector::vision::OcrMode::Auto,
OcrMode::Force => pdf_inspector::vision::OcrMode::Force,
};
}
if let Some(pages) = options.page_numbers {
result = result.page_numbers(pages);
}
if let Some(password) = options.password {
result = result.password(password);
}
if let Some(dpi) = options.dpi {
result.render.dpi = dpi as f32;
}
if let Some(minimum_confidence) = options.minimum_confidence {
result.ocr.minimum_confidence = minimum_confidence as f32;
}
if let Some(confidence) = options.hosted_recommendation_confidence {
result.hosted_recommendation_confidence = confidence as f32;
}
if let Some(directory) = options.model_directory {
result.ocr.model_directory = Some(directory.into());
}
if options.offline.unwrap_or(false) {
result.ocr.model_downloads = pdf_inspector::vision::ModelDownloadPolicy::Offline;
}
result
}
fn convert_page_content_source(
source: pdf_inspector::vision::PageContentSource,
) -> PageContentSource {
match source {
pdf_inspector::vision::PageContentSource::Native => PageContentSource::Native,
pdf_inspector::vision::PageContentSource::Ocr => PageContentSource::Ocr,
pdf_inspector::vision::PageContentSource::Fused => PageContentSource::Fused,
_ => PageContentSource::Native,
}
}
fn timing_ms(value: u64) -> u32 {
u32::try_from(value).unwrap_or(u32::MAX)
}
fn to_napi_ocr_result(result: pdf_inspector::vision::OcrPdfResult) -> OcrPdfResult {
OcrPdfResult {
markdown: result.markdown,
pages: result
.pages
.into_iter()
.map(|page| {
let provenance = page.provenance;
OcrPageResult {
page_number: page.page_number,
markdown: page.markdown,
provenance: OcrPageProvenance {
page_number: provenance.page_number,
source: convert_page_content_source(provenance.source),
ocr_model: provenance.ocr_model.map(|model| OcrModelIdentity {
name: model.name,
revision: model.revision,
}),
render_dpi: provenance.render_dpi.map(f64::from),
ocr_confidence: provenance.ocr_confidence.map(f64::from),
timings: OcrTimings {
render_ms: timing_ms(provenance.timings.render_ms),
ocr_ms: timing_ms(provenance.timings.ocr_ms),
assembly_ms: timing_ms(provenance.timings.assembly_ms),
},
warnings: provenance.warnings,
hosted_recommended: provenance.hosted_recommended,
},
}
})
.collect(),
page_count: result.page_count,
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
pages_routed_to_ocr: result.pages_routed_to_ocr,
pages_recommending_hosted: result.pages_recommending_hosted,
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
pages_with_tables: result.pages_with_tables,
pages_with_columns: result.pages_with_columns,
is_complex: result.is_complex,
processing_time_ms: timing_ms(result.processing_time_ms),
render_time_ms: timing_ms(result.render_time_ms),
ocr_time_ms: timing_ms(result.ocr_time_ms),
}
}
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
match t {
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
@@ -414,13 +219,6 @@ fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult>
Ok(to_napi_result(result))
}
fn process_pdf_with_ocr_impl(bytes: &[u8], options: Option<OcrOptions>) -> Result<OcrPdfResult> {
let options = to_core_ocr_options(options);
let result = pdf_inspector::vision::process_pdf_with_ocr_mem(bytes, options)
.map_err(|error| to_napi_err(error, "process_pdf_with_ocr"))?;
Ok(to_napi_ocr_result(result))
}
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
let result =
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
@@ -1020,45 +818,6 @@ pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<P
})
}
pub struct ProcessPdfWithOcrTask {
bytes: Vec<u8>,
options: Option<OcrOptions>,
}
impl Task for ProcessPdfWithOcrTask {
type Output = OcrPdfResult;
type JsValue = OcrPdfResult;
fn compute(&mut self) -> Result<Self::Output> {
let bytes = std::mem::take(&mut self.bytes);
let options = self.options.take();
catch_panic(
"process_pdf_with_ocr",
panic::AssertUnwindSafe(move || process_pdf_with_ocr_impl(&bytes, options)),
)
}
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
Ok(output)
}
}
/// Process a PDF with selective OCR on the libuv thread pool.
///
/// OCR defaults to Auto, which only loads PDFium, ONNX Runtime, and the OCR
/// model if native extraction routes at least one page. The input buffer is
/// copied before the promise is returned and is safe to reuse immediately.
#[napi(ts_return_type = "Promise<OcrPdfResult>")]
pub fn process_pdf_with_ocr(
buffer: Buffer,
options: Option<OcrOptions>,
) -> AsyncTask<ProcessPdfWithOcrTask> {
AsyncTask::new(ProcessPdfWithOcrTask {
bytes: buffer.to_vec(),
options,
})
}
pub struct ClassifyPdfTask {
bytes: Vec<u8>,
}
-32
View File
@@ -3,7 +3,6 @@ import { strict as assert } from 'assert';
import {
processPdf,
processPdfAsync,
processPdfWithOcr,
detectPdf,
classifyPdf,
classifyPdfAsync,
@@ -221,37 +220,6 @@ const fromMutated = await inFlight;
assert.equal(fromMutated.markdown, result.markdown);
console.log(' processPdfAsync input copied at call time: OK');
// --- Selective OCR ---
console.log('Testing processPdfWithOcr...');
// Off exercises the complete result/provenance contract without loading
// external PDFium, ONNX Runtime, or model artifacts.
const ocrOff = await processPdfWithOcr(fixture, { mode: 'Off' });
assert.equal(ocrOff.pageCount, 3);
assert.equal(ocrOff.pages.length, 3);
assert.deepEqual(ocrOff.pagesRoutedToOcr, []);
assert.ok(ocrOff.pages.every(page => page.provenance.source === 'Native'));
assert.ok(ocrOff.pages.every(page => page.provenance.ocrModel === undefined));
assert.ok(ocrOff.markdown.length > 0);
// Auto must preserve the lightweight path for clean text PDFs.
const ocrAuto = await processPdfWithOcr(fixture);
assert.deepEqual(ocrAuto.pagesRoutedToOcr, []);
assert.equal(ocrAuto.renderTimeMs, 0);
assert.equal(ocrAuto.ocrTimeMs, 0);
const ocrSelected = await processPdfWithOcr(fixture, {
mode: 'Off',
pageNumbers: [2],
});
assert.deepEqual(ocrSelected.pages.map(page => page.pageNumber), [2]);
await assert.rejects(
processPdfWithOcr(fixture, { mode: 'Off', pageNumbers: [0] }),
/page 0/,
);
console.log(' processPdfWithOcr: OK');
// concurrent async calls all settle
const [c1, c2, c3] = await Promise.all([
processPdfAsync(fixture),
+1 -81
View File
@@ -1,6 +1,6 @@
"""Type stubs for pdf_inspector."""
from typing import Literal, Optional
from typing import Optional
class PdfResult:
"""Result of processing a PDF file."""
@@ -27,53 +27,6 @@ class PageOcrReasons:
reasons: list[str]
"""Machine-readable OCR reason identifiers."""
class OcrModelIdentity:
"""Exact OCR model identity retained in page provenance."""
name: str
revision: str
class OcrTimings:
"""Per-page OCR processing timings."""
render_ms: int
ocr_ms: int
assembly_ms: int
class OcrPageProvenance:
"""Source, model, confidence, and fallback metadata for one page."""
page_number: int
"""1-indexed page number."""
source: Literal["native", "ocr", "fused"]
"""'native', 'ocr', or 'fused'."""
ocr_model: Optional[OcrModelIdentity]
render_dpi: Optional[float]
ocr_confidence: Optional[float]
timings: OcrTimings
warnings: list[str]
hosted_recommended: bool
class OcrPageResult:
"""Final Markdown and provenance for one page."""
page_number: int
"""1-indexed page number."""
markdown: str
provenance: OcrPageProvenance
class OcrPdfResult:
"""Complete native/OCR Markdown output."""
markdown: str
pages: list[OcrPageResult]
page_count: int
pages_recommended_for_ocr: list[int]
pages_routed_to_ocr: list[int]
pages_recommending_hosted: list[int]
ocr_reasons_by_page: list[PageOcrReasons]
pages_with_tables: list[int]
pages_with_columns: list[int]
is_complex: bool
processing_time_ms: int
render_time_ms: int
ocr_time_ms: int
class PdfClassification:
"""Lightweight PDF classification result."""
pdf_type: str
@@ -161,39 +114,6 @@ def process_pdf_bytes(data: bytes, pages: Optional[list[int]] = None) -> PdfResu
"""Process a PDF from bytes in memory."""
...
def process_pdf_with_ocr(
path: str,
*,
mode: Literal["off", "auto", "force"] = "auto",
page_numbers: Optional[list[int]] = None,
password: Optional[str] = None,
dpi: float = 150.0,
minimum_confidence: float = 0.0,
hosted_recommendation_confidence: float = 0.5,
model_directory: Optional[str] = None,
offline: bool = False,
) -> OcrPdfResult:
"""Process a PDF through native extraction and selective OCR.
Page numbers are 1-indexed. OCR runs without holding the Python GIL.
"""
...
def process_pdf_with_ocr_bytes(
data: bytes,
*,
mode: Literal["off", "auto", "force"] = "auto",
page_numbers: Optional[list[int]] = None,
password: Optional[str] = None,
dpi: float = 150.0,
minimum_confidence: float = 0.0,
hosted_recommendation_confidence: float = 0.5,
model_directory: Optional[str] = None,
offline: bool = False,
) -> OcrPdfResult:
"""Process PDF bytes through native extraction and selective OCR."""
...
def detect_pdf(path: str) -> PdfResult:
"""Fast detection only — no text extraction."""
...
+1 -1
View File
@@ -6,7 +6,7 @@ build-backend = "maturin"
name = "pdf-inspector"
# Keep package versions in sync with `python3 scripts/version.py <version>`.
# CI publishes automatically when the synchronized change lands on main.
version = "1.15.0"
version = "1.14.2"
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
readme = "docs/python.md"
license = { text = "MIT" }
+1 -1
View File
@@ -975,7 +975,7 @@ result = pdf_inspector.<span class="fn">process_pdf</span>(<span class="str">"do
<script>
(() => {
const MAX_FILE_SIZE = 25 * 1024 * 1024;
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@1.15.0/pdf_inspector_wasm.js";
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@1.14.2/pdf_inspector_wasm.js";
const input = document.querySelector("#pdf-input");
const dropZone = document.querySelector("#drop-zone");
const filePanel = document.querySelector("#demo-file");
+29 -61
View File
@@ -165,8 +165,8 @@ fn format_ocr_json(result: &OcrPdfResult) -> String {
.collect::<Vec<_>>()
.join(",");
format!(
r#"{{"page":{},"source":"{}","markdown":"{}","ocr_model":{},"render_dpi":{},"ocr_confidence":{},"hosted_recommended":{},"timings":{{"render_ms":{},"ocr_ms":{},"assembly_ms":{}}},"warnings":[{}]}}"#,
provenance.page_number,
r#"{{"page":{},"source":"{}","markdown":"{}","ocr_model":{},"render_dpi":{},"ocr_confidence":{},"hosted_recommended":{},"timings":{{"render_ms":{},"ocr_ms":{},"layout_ms":{},"assembly_ms":{}}},"warnings":[{}]}}"#,
provenance.page,
source,
json_escape(&page.markdown),
model,
@@ -175,6 +175,7 @@ fn format_ocr_json(result: &OcrPdfResult) -> String {
provenance.hosted_recommended,
provenance.timings.render_ms,
provenance.timings.ocr_ms,
provenance.timings.layout_ms,
provenance.timings.assembly_ms,
warnings,
)
@@ -195,7 +196,7 @@ fn format_ocr_json(result: &OcrPdfResult) -> String {
.join(",");
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
format!(
r#"{{"schema_version":1,"page_count":{},"processing_time_ms":{},"render_time_ms":{},"ocr_time_ms":{},"pages_recommended_for_ocr":[{}],"pages_routed_to_ocr":[{}],"pages_recommending_hosted":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"pages":[{}],"markdown":"{}"}}"#,
r#"{{"page_count":{},"processing_time_ms":{},"render_time_ms":{},"ocr_time_ms":{},"pages_recommended_for_ocr":[{}],"pages_routed_to_ocr":[{}],"pages_recommending_hosted":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"pages":[{}],"markdown":"{}"}}"#,
result.page_count,
result.processing_time_ms,
result.render_time_ms,
@@ -223,19 +224,6 @@ fn argument_value<'a>(args: &'a [String], name: &str) -> Result<Option<&'a str>,
.transpose()
}
fn format_ocr_error_json(error: &str) -> String {
format!(r#"{{"schema_version":1,"error":"{}"}}"#, json_escape(error))
}
fn exit_ocr_error(error: &str, json_output: bool) -> ! {
if json_output {
println!("{}", format_ocr_error_json(error));
} else {
eprintln!("Error: {error}");
}
process::exit(1);
}
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
fn float_argument(args: &[String], name: &str, default: f32) -> Result<f32, String> {
argument_value(args, name)?
@@ -259,9 +247,7 @@ fn extract_items_json(
#[cfg(test)]
mod tests {
use super::{extract_items_json, format_items_json, format_ocr_error_json};
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
use super::{format_ocr_json, process_pdf_with_ocr, OcrPdfOptions};
use super::{extract_items_json, format_items_json};
use pdf_inspector::extractor::ItemType;
use pdf_inspector::TextItem;
@@ -311,27 +297,6 @@ mod tests {
"decrypted item JSON should contain fixture text, got {json}"
);
}
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
#[test]
fn ocr_json_has_a_versioned_stable_envelope() {
let result =
process_pdf_with_ocr("tests/fixtures/thermo-freon12.pdf", OcrPdfOptions::new())
.unwrap();
let json = format_ocr_json(&result);
assert!(json.starts_with(r#"{"schema_version":1,"page_count":3,"#));
assert!(json.contains(r#""page":1,"source":"native""#));
assert!(!json.contains("layout_ms"));
}
#[test]
fn ocr_json_errors_use_the_same_versioned_envelope() {
assert_eq!(
format_ocr_error_json("bad \"value\""),
r#"{"schema_version":1,"error":"bad \"value\""}"#
);
}
}
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
@@ -476,27 +441,23 @@ fn main() {
.iter()
.any(|option| args.iter().any(|argument| argument == option));
if ocr_mode_argument.is_none() && has_ocr_only_option {
exit_ocr_error(
"OCR options require --ocr off, --ocr auto, or --ocr force",
json_output,
);
eprintln!("Error: OCR options require --ocr off, --ocr auto, or --ocr force");
process::exit(1);
}
if let Some(mode) = ocr_mode_argument {
if items_json_output || detect_only || analyze {
exit_ocr_error(
"--ocr cannot be combined with --items-json, --detect-only, or --analyze",
json_output,
eprintln!(
"Error: --ocr cannot be combined with --items-json, --detect-only, or --analyze"
);
process::exit(1);
}
#[cfg(not(all(feature = "ocr", not(target_arch = "wasm32"))))]
{
let _ = mode;
exit_ocr_error(
"this pdf2md build does not include OCR; rebuild with --features ocr",
json_output,
);
eprintln!("Error: this pdf2md build does not include OCR; rebuild with --features ocr");
process::exit(1);
}
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
@@ -506,26 +467,28 @@ fn main() {
"auto" => OcrMode::Auto,
"force" => OcrMode::Force,
value => {
exit_ocr_error(
&format!("invalid --ocr mode {value:?}; expected off, auto, or force"),
json_output,
);
eprintln!("Error: invalid --ocr mode {value:?}; expected off, auto, or force");
process::exit(1);
}
};
let dpi = float_argument(&args, "--ocr-dpi", 150.0).unwrap_or_else(|error| {
exit_ocr_error(&error, json_output);
eprintln!("Error: {error}");
process::exit(1);
});
let minimum_confidence = float_argument(&args, "--ocr-min-confidence", 0.0)
.unwrap_or_else(|error| {
exit_ocr_error(&error, json_output);
eprintln!("Error: {error}");
process::exit(1);
});
let hosted_threshold = float_argument(&args, "--ocr-hosted-threshold", 0.5)
.unwrap_or_else(|error| {
exit_ocr_error(&error, json_output);
eprintln!("Error: {error}");
process::exit(1);
});
let model_directory =
argument_value(&args, "--ocr-model-dir").unwrap_or_else(|error| {
exit_ocr_error(&error, json_output);
eprintln!("Error: {error}");
process::exit(1);
});
let mut ocr = OcrOptions::new()
@@ -548,7 +511,7 @@ fn main() {
.markdown(markdown)
.hosted_recommendation_confidence(hosted_threshold);
if let Some(pages) = page_filter.clone() {
pdf_options = pdf_options.page_numbers(pages);
pdf_options = pdf_options.pages(pages);
}
if let Some(password) = password.clone() {
pdf_options = pdf_options.password(password);
@@ -586,7 +549,12 @@ fn main() {
}
}
Err(error) => {
exit_ocr_error(&error.to_string(), json_output);
if json_output {
println!(r#"{{"error":"{}"}}"#, json_escape(&error.to_string()));
} else {
eprintln!("Error: {error}");
}
process::exit(1);
}
}
return;
+12 -150
View File
@@ -403,15 +403,8 @@ pub(crate) fn detect_from_document(
&& !(analysis.has_decodable_text_fonts && analysis.text_operator_count >= 10);
let looks_like_scan =
analysis.image_count <= 1 && analysis.text_operator_count < 50 && alphanum_low;
// A template-image page below the `pages_with_text` floor is
// a scan with incidental chrome (masthead, stamp, date line)
// even when that chrome is diverse, decodable text — keep
// this in sync with `page_ocr_signals`.
let sparse_text_over_scan = analysis.has_template_image
&& analysis.text_operator_count < config.min_text_ops_per_page.max(10);
if (analysis.has_template_image && looks_like_scan)
|| analysis.has_vector_text
|| sparse_text_over_scan
|| (analysis.text_operator_count < config.min_text_ops_per_page
&& analysis.has_images)
{
@@ -1802,24 +1795,17 @@ pub(crate) fn analyze_page_images(doc: &Document, page_id: ObjectId) -> (bool, u
/// low alphanumeric diversity in raw string operands (unless decodable
/// CID/ToUnicode fonts explain that away) — the gate used for
/// `pages_with_template_images` and Mixed-type per-page routing.
/// 2. Insufficient real text volume, using the same `effective_min_ops`
/// floor (`min_text_ops_per_page.max(10)`) that `pages_with_text`
/// applies to image-bearing pages. That floor is a per-page judgment,
/// not part of the cross-page aggregate: classification counts a
/// template-image page with fewer ops as textless and routes it to OCR,
/// so this function must agree. The lower bare threshold (3) let a
/// full-page scan carrying a small native masthead — a newspaper
/// header, stamp, or date line of ~4 diverse, decodable text ops —
/// extract as "a text page" here while whole-document classification
/// called the same page scanned, silently dropping the page body from
/// OCR routing. `alphanum_low` can't catch that case: masthead chrome
/// is real text, so its byte diversity is high.
///
/// This function always evaluates against `DetectionConfig::default()` —
/// it has no config parameter, and the per-page extraction path that calls
/// it never carries one. A caller passing a custom `min_text_ops_per_page`
/// to `detect_from_document` affects whole-document detection only; the
/// two paths agree under the default configuration.
/// 2. Insufficient real text volume, using `DetectionConfig::default()`'s
/// `min_text_ops_per_page` (3) — the same threshold Mixed-type per-page
/// routing applies via `text_operator_count < config.min_text_ops_per_page
/// && has_images` (simplified here since a template image implies
/// `has_images`). Deliberately *not* the higher `effective_min_ops`
/// floor (`min_text_ops_per_page.max(10)`) that whole-document
/// `PdfType::ImageBased`/`Scanned` classification uses for
/// `pages_with_text` — that's a cross-page aggregate decision this
/// per-page function has no way to replicate exactly, and the lower
/// per-page threshold is the one a single page's own signals can
/// actually agree with.
///
/// `has_vector_text` is true when a page has vector-outlined text (glyphs
/// drawn as paths rather than shown via text-showing operators) —
@@ -1841,7 +1827,7 @@ pub(crate) fn page_ocr_signals(doc: &Document, page_id: ObjectId) -> (bool, bool
let looks_like_scan =
analysis.image_count <= 1 && analysis.text_operator_count < 50 && alphanum_low;
let insufficient_text =
analysis.text_operator_count < DetectionConfig::default().min_text_ops_per_page.max(10);
analysis.text_operator_count < DetectionConfig::default().min_text_ops_per_page;
looks_like_scan || insufficient_text
};
@@ -3009,130 +2995,6 @@ mod tests {
);
}
// ---------- masthead-over-scan tests: template image + sparse chrome ----------
/// Builds a page whose only image is a full-page scan inside a Form
/// XObject, plus `masthead_ops` native text-show ops of diverse,
/// decodable chrome (newspaper masthead / date line style).
fn masthead_scan_page(masthead_lines: &[&str]) -> (Document, ObjectId) {
use lopdf::dictionary;
let mut doc = Document::with_version("1.4");
let pages_id = doc.new_object_id();
let page_id = doc.new_object_id();
let image_id = doc.add_object(Object::Stream(lopdf::Stream::new(
dictionary! {
"Type" => "XObject",
"Subtype" => Object::Name(b"Image".to_vec()),
"Width" => Object::Integer(1500),
"Height" => Object::Integer(2383),
},
Vec::new(),
)));
let form_id = doc.add_object(Object::Stream(lopdf::Stream::new(
dictionary! {
"Type" => "XObject",
"Subtype" => Object::Name(b"Form".to_vec()),
"Resources" => dictionary! {
"XObject" => dictionary! {
"Im0" => Object::Reference(image_id),
},
},
},
b"1500 0 0 2383 0 0 cm /Im0 Do".to_vec(),
)));
let font_id = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => Object::Name(b"Type1".to_vec()),
"BaseFont" => Object::Name(b"Helvetica".to_vec()),
});
let mut content = b"q /Fm0 Do Q BT /F1 12 Tf ".to_vec();
for line in masthead_lines {
content.extend_from_slice(format!("({line}) Tj ").as_bytes());
}
content.extend_from_slice(b"ET");
let content_id =
doc.add_object(Object::Stream(lopdf::Stream::new(dictionary! {}, content)));
doc.objects.insert(
page_id,
Object::Dictionary(dictionary! {
"Type" => "Page",
"Parent" => Object::Reference(pages_id),
"MediaBox" => vec![0.into(), 0.into(), 1500.into(), 2383.into()],
"Resources" => dictionary! {
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
"XObject" => dictionary! { "Fm0" => Object::Reference(form_id) },
},
"Contents" => Object::Reference(content_id),
}),
);
doc.objects.insert(
pages_id,
Object::Dictionary(dictionary! {
"Type" => "Pages",
"Kids" => vec![Object::Reference(page_id)],
"Count" => Object::Integer(1),
}),
);
(doc, page_id)
}
#[test]
fn test_masthead_over_form_wrapped_scan_needs_ocr() {
// A full-page scan wrapped in a Form XObject with ~4 ops of real,
// diverse masthead text. `alphanum_low` can't flag it (the chrome is
// genuine text), so the sparse-text floor must: without OCR the page
// body is silently lost while classification calls the page scanned.
let (doc, page_id) = masthead_scan_page(&[
"18",
"FINANCIAL EXPRESS",
"WWW.FINANCIALEXPRESS.COM",
"FRIDAY, DECEMBER 13, 2024",
]);
let analysis = analyze_page_content(&doc, page_id);
assert!(
analysis.has_template_image,
"sanity: full-page image inside the form must be found"
);
assert!(
analysis.unique_alphanum_chars >= 10,
"sanity: masthead text is diverse, alphanum_low cannot fire"
);
let (needs_ocr, _) = page_ocr_signals(&doc, page_id);
assert!(
needs_ocr,
"template image + text below the pages_with_text floor is a scan"
);
}
#[test]
fn test_text_page_over_background_image_stays_native() {
// Counterpart: a real text page over a full-page background image
// (letterhead/watermark) has enough text ops to clear the
// `pages_with_text` floor and must NOT be routed to OCR.
let lines: Vec<String> = (0..12)
.map(|i| format!("Paragraph line {i} with ordinary body text"))
.collect();
let refs: Vec<&str> = lines.iter().map(String::as_str).collect();
let (doc, page_id) = masthead_scan_page(&refs);
let analysis = analyze_page_content(&doc, page_id);
assert!(
analysis.has_template_image,
"sanity: background image found"
);
assert!(
analysis.text_operator_count >= 10,
"sanity: body text clears the floor"
);
let (needs_ocr, _) = page_ocr_signals(&doc, page_id);
assert!(
!needs_ocr,
"a text page with a background image must stay native"
);
}
// ---------- P2 tests: Form XObject font traversal ----------
#[test]
+4 -20
View File
@@ -561,11 +561,7 @@ pub(crate) fn extract_page_text_items(
y,
width,
height: rendered_size,
font: crate::extractor::fonts::item_font_name(
&current_font,
base_font,
)
.to_string(),
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
@@ -749,11 +745,7 @@ pub(crate) fn extract_page_text_items(
y,
width,
height: rendered_size,
font: crate::extractor::fonts::item_font_name(
&current_font,
base_font,
)
.to_string(),
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
@@ -860,11 +852,7 @@ pub(crate) fn extract_page_text_items(
y,
width,
height: rendered_size,
font: crate::extractor::fonts::item_font_name(
&current_font,
base_font,
)
.to_string(),
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
@@ -1017,11 +1005,7 @@ pub(crate) fn extract_page_text_items(
y,
width,
height: rendered_size,
font: crate::extractor::fonts::item_font_name(
&current_font,
base_font,
)
.to_string(),
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
-31
View File
@@ -225,26 +225,6 @@ pub(crate) fn build_type3_scales(
scales
}
/// The name a `TextItem` carries for its font: the `/BaseFont` family name
/// ("ABCDEF+CMMI10"), which identifies the actual face, rather than the
/// arbitrary per-page resource tag ("F2").
///
/// Exception: resource names using Distiller's CID convention (`C2_0`,
/// `C0_1`) are kept as-is — `text_utils::is_cid_font` keys on that prefix
/// for micro-gap joining, and the family name carries no CID marker to
/// replace it. This is a known, deliberate wart: `TextItem::font` is the
/// face name except for this one producer convention. The clean fix is an
/// explicit CID flag on `TextItem`, which touches its ~29 construction
/// sites; do that migration when `TextItem` next changes shape, and delete
/// this carve-out with it.
pub(crate) fn item_font_name<'a>(resource_name: &'a str, base_font: &'a str) -> &'a str {
if crate::text_utils::is_cid_font(resource_name) {
resource_name
} else {
base_font
}
}
/// Parse font widths from a font dictionary, dispatching by Subtype
pub(crate) fn parse_font_widths(
doc: &Document,
@@ -1684,17 +1664,6 @@ fn score_text(text: &str) -> i32 {
#[cfg(test)]
mod tests {
#[test]
fn item_font_name_prefers_family_over_resource_tag() {
use super::item_font_name;
assert_eq!(item_font_name("F2", "ABCDEF+CMMI10"), "ABCDEF+CMMI10");
assert_eq!(item_font_name("T22", "Times-Roman"), "Times-Roman");
// Distiller CID-convention resources keep the resource name:
// is_cid_font keys on the C2_/C0_ prefix for micro-gap joining.
assert_eq!(item_font_name("C2_0", "ABCDEE+SimSun"), "C2_0");
assert_eq!(item_font_name("C0_1", "ABCDEE+MSMincho"), "C0_1");
}
#[test]
fn type3_scale_resolves_indirect_matrix_and_bbox_numbers() {
use lopdf::{dictionary, Document, Object};
+2 -10
View File
@@ -620,11 +620,7 @@ fn extract_form_xobject_text_inner(
y,
width,
height: rendered_size,
font: crate::extractor::fonts::item_font_name(
&current_font,
base_font,
)
.to_string(),
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
@@ -779,11 +775,7 @@ fn extract_form_xobject_text_inner(
y,
width,
height: rendered_size,
font: crate::extractor::fonts::item_font_name(
&current_font,
base_font,
)
.to_string(),
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
+3 -20
View File
@@ -459,15 +459,8 @@ pub fn extract_pages_markdown_mem(
buffer: &[u8],
pages: Option<&[u32]>,
) -> Result<PagesExtractionResult, PdfError> {
extract_pages_markdown_mem_impl(
buffer,
pages,
None,
&MarkdownOptions::default(),
false,
false,
)
.map(|(result, _)| result)
extract_pages_markdown_mem_impl(buffer, pages, None, &MarkdownOptions::default(), false)
.map(|(result, _)| result)
}
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
@@ -483,7 +476,6 @@ pub(crate) fn extract_pages_markdown_mem_for_ocr(
password,
markdown_options,
markdown_options.strip_headers_footers,
true,
)
}
@@ -493,7 +485,6 @@ fn extract_pages_markdown_mem_impl(
password: Option<&str>,
markdown_options: &MarkdownOptions,
strip_repeated_headers_footers: bool,
preserve_ocr_candidates: bool,
) -> Result<(PagesExtractionResult, u32), PdfError> {
validate_pdf_bytes(buffer)?;
let (doc, page_count) = load_document_from_mem_with_password(buffer, password)?;
@@ -672,15 +663,7 @@ fn extract_pages_markdown_mem_impl(
results.push(PageMarkdown {
page: page_0idx,
// The public native extractor continues to suppress unreliable
// text. The OCR orchestrator retains clean partial text
// internally so it can compare/fuse it with OCR before deciding
// what is safe to return.
markdown: if needs_ocr && !preserve_ocr_candidates {
String::new()
} else {
md
},
markdown: if needs_ocr { String::new() } else { md },
needs_ocr,
ocr_reason,
});
-44
View File
@@ -203,42 +203,9 @@ pub(crate) fn is_code_like(text: &str) -> bool {
false
}
/// True when a line's text is essentially all monospace (≥90% by character
/// count). Code lines are wholly monospace; anything less is prose carrying
/// mono-styled fragments — a URL sidebar, or a sentence quoting an inline
/// code literal — and fencing it would split paragraphs mid-sentence.
/// Any-item matching was safe only while items carried opaque font resource
/// names that never matched the monospace patterns; items now carry real
/// family names.
pub(crate) fn line_is_monospace(line: &crate::types::TextLine) -> bool {
let mut monospace_chars = 0usize;
let mut total_chars = 0usize;
for item in &line.items {
let text = item.text.trim();
let chars = text.chars().count();
total_chars += chars;
// Hyperlinks and underlined text set in a mono face are link
// styling, not code — a URL sidebar must not fence lyric lines.
let looks_like_link = item.is_underline
|| matches!(item.item_type, crate::types::ItemType::Link(_))
|| text.contains("://")
|| text.starts_with("www.");
if is_monospace_font(&item.font) && !looks_like_link {
monospace_chars += chars;
}
}
total_chars > 0 && monospace_chars * 10 >= total_chars * 9
}
/// Check if font name indicates monospace
pub(crate) fn is_monospace_font(font_name: &str) -> bool {
let lower = font_name.to_lowercase();
// "Monotype" is a foundry prefix on proportional faces (Monotype
// Corsiva, Monotype Garamond) — it must not satisfy the generic "mono"
// token below.
if lower.contains("monotype") {
return false;
}
let patterns = [
"courier",
"consolas",
@@ -263,17 +230,6 @@ pub(crate) fn is_monospace_font(font_name: &str) -> bool {
mod tests {
use super::*;
#[test]
fn monotype_foundry_faces_are_not_monospace() {
// "Monotype" is a foundry prefix on proportional faces; the generic
// "mono" token must not classify them as code fonts.
assert!(!is_monospace_font("MonotypeCorsiva"));
assert!(!is_monospace_font("ABCDEF+Monotype-Garamond"));
assert!(is_monospace_font("RobotoMono-Regular"));
assert!(is_monospace_font("PTMono"));
assert!(is_monospace_font("Courier"));
}
#[test]
fn format_list_item_plain_bullet() {
assert_eq!(format_list_item("● Item"), "- Item");
+28 -52
View File
@@ -11,7 +11,9 @@ use super::analysis::{
detect_header_level, font_size_rarity, has_dot_leaders, is_heading_fragment, is_toc_entry_line,
is_toc_marker_heading,
};
use super::classify::{format_list_item, is_caption_line, is_list_item, starts_with_bullet_marker};
use super::classify::{
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
};
use super::heading::classify_heading_sequences;
use super::postprocess::clean_markdown;
use super::preprocess::{merge_drop_caps, merge_heading_lines};
@@ -769,27 +771,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
let mut in_list = false;
let mut in_paragraph = false;
let mut last_list_x: Option<f32> = None;
// Code lines accumulate here and the fence is emitted only when the
// block flushes with content — an empty ``` ``` pair can never appear.
fn flush_code_block(output: &mut String, pending_code: &mut String) {
let trimmed = pending_code.trim();
// A fragment too short to be code — a lone ® or stray glyph set in
// a mono face — reads better as plain text than as a fenced block.
if trimmed.chars().count() < 3 {
if !trimmed.is_empty() {
output.push_str(trimmed);
output.push_str("\n\n");
}
} else {
output.push_str("```\n");
output.push_str(pending_code);
output.push_str("```\n");
}
pending_code.clear();
}
let mut in_code_block = false;
let mut pending_code = String::new();
let mut prev_had_dot_leaders = false;
let mut paragraph_in_wrapped_bold_run = false;
let mut toc_suppress_page: Option<u32> = None;
@@ -823,7 +805,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// Flush current page's remaining tables and images
if current_page > 0 {
if in_code_block {
flush_code_block(&mut output, &mut pending_code);
output.push_str("```\n");
in_code_block = false;
}
flush_page_tables_and_images(
@@ -885,14 +867,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
PositionedBlockKind::Image => inserted_images.contains(&(current_page, idx)),
};
if positioned_block_precedes_line(block, line) && !already_inserted {
// Code lines buffer until their block closes; flush them
// first so this block cannot jump ahead of code that
// precedes it in reading order. A code line after the
// block reopens a new fence naturally.
if in_code_block {
flush_code_block(&mut output, &mut pending_code);
in_code_block = false;
}
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
@@ -963,22 +937,15 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// These should be on their own line followed by a paragraph break
let struct_role = struct_roles.and_then(|roles| resolve_line_struct_role(line, roles));
// Determine if this line is code (struct-tree or font-based) for
// block accumulation. Font-based detection only opens a block at a
// paragraph boundary: a mono-set line that continues an open prose
// paragraph is the producer smearing an inline code literal's style
// across a wrapped line (HTML-to-PDF exports do this), and fencing
// it would cut the sentence in three.
// Determine if this line is code (struct-tree or font-based) for block accumulation
let is_code_line = struct_role
.as_ref()
.is_some_and(|r| matches!(r, StructRole::Code))
|| (options.detect_code
&& (in_code_block || !in_paragraph)
&& super::classify::line_is_monospace(line));
|| (options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)));
// Close code block when transitioning to non-code
if in_code_block && !is_code_line {
flush_code_block(&mut output, &mut pending_code);
output.push_str("```\n");
in_code_block = false;
}
@@ -1212,9 +1179,12 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
in_code_block = true;
pending_code.push_str(plain_trimmed);
pending_code.push('\n');
if !in_code_block {
output.push_str("```\n");
in_code_block = true;
}
output.push_str(plain_trimmed);
output.push('\n');
continue;
}
@@ -1239,7 +1209,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// Close any trailing code block
if in_code_block {
flush_code_block(&mut output, &mut pending_code);
output.push_str("```\n");
}
// Flush current page and any remaining pages with tables/images
@@ -1400,7 +1370,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
&& !is_toc_entry_line(plain_trimmed)
&& !is_heading_fragment(plain_trimmed)
&& toc_suppress_page != Some(line.page)
&& !(options.detect_code && super::classify::line_is_monospace(line))
&& !(options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)))
{
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
if let Some(header_level) = detect_header_level(
@@ -1501,13 +1471,19 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
}
}
// Detect code blocks by font. Only at a paragraph boundary — a
// mono-set line continuing an open prose paragraph is an inline
// code literal's style smeared across a wrapped line, not code.
if options.detect_code && !in_paragraph && super::classify::line_is_monospace(line) {
// Use plain text for code blocks
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
continue;
// Detect code blocks by font
if options.detect_code {
let is_mono = line.items.iter().any(|i| is_monospace_font(&i.font));
if is_mono {
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Use plain text for code blocks
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
continue;
}
}
// Regular text - join lines within same paragraph with space
-296
View File
@@ -85,107 +85,6 @@ impl PyPageOcrReasons {
}
}
/// Exact OCR model identity retained in page provenance.
#[pyclass(name = "OcrModelIdentity")]
#[derive(Clone)]
pub struct PyOcrModelIdentity {
#[pyo3(get)]
pub name: String,
#[pyo3(get)]
pub revision: String,
}
/// Per-page OCR processing timings.
#[pyclass(name = "OcrTimings")]
#[derive(Clone)]
pub struct PyOcrTimings {
#[pyo3(get)]
pub render_ms: u64,
#[pyo3(get)]
pub ocr_ms: u64,
#[pyo3(get)]
pub assembly_ms: u64,
}
/// Source, model, confidence, and fallback metadata for one page.
#[pyclass(name = "OcrPageProvenance")]
#[derive(Clone)]
pub struct PyOcrPageProvenance {
/// 1-indexed page number.
#[pyo3(get)]
pub page_number: u32,
/// "native", "ocr", or "fused".
#[pyo3(get)]
pub source: String,
#[pyo3(get)]
pub ocr_model: Option<PyOcrModelIdentity>,
#[pyo3(get)]
pub render_dpi: Option<f32>,
#[pyo3(get)]
pub ocr_confidence: Option<f32>,
#[pyo3(get)]
pub timings: PyOcrTimings,
#[pyo3(get)]
pub warnings: Vec<String>,
#[pyo3(get)]
pub hosted_recommended: bool,
}
/// Final Markdown and provenance for one page.
#[pyclass(name = "OcrPageResult")]
#[derive(Clone)]
pub struct PyOcrPageResult {
/// 1-indexed page number.
#[pyo3(get)]
pub page_number: u32,
#[pyo3(get)]
pub markdown: String,
#[pyo3(get)]
pub provenance: PyOcrPageProvenance,
}
/// Complete native/OCR Markdown output.
#[pyclass(name = "OcrPdfResult")]
#[derive(Clone)]
pub struct PyOcrPdfResult {
#[pyo3(get)]
pub markdown: String,
#[pyo3(get)]
pub pages: Vec<PyOcrPageResult>,
#[pyo3(get)]
pub page_count: u32,
#[pyo3(get)]
pub pages_recommended_for_ocr: Vec<u32>,
#[pyo3(get)]
pub pages_routed_to_ocr: Vec<u32>,
#[pyo3(get)]
pub pages_recommending_hosted: Vec<u32>,
#[pyo3(get)]
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
#[pyo3(get)]
pub pages_with_tables: Vec<u32>,
#[pyo3(get)]
pub pages_with_columns: Vec<u32>,
#[pyo3(get)]
pub is_complex: bool,
#[pyo3(get)]
pub processing_time_ms: u64,
#[pyo3(get)]
pub render_time_ms: u64,
#[pyo3(get)]
pub ocr_time_ms: u64,
}
#[pymethods]
impl PyOcrPdfResult {
fn __repr__(&self) -> String {
format!(
"OcrPdfResult(pages={}, routed_to_ocr={:?}, recommending_hosted={:?})",
self.page_count, self.pages_routed_to_ocr, self.pages_recommending_hosted
)
}
}
// ---------------------------------------------------------------------------
// Classification wrapper (lightweight)
// ---------------------------------------------------------------------------
@@ -463,101 +362,6 @@ fn to_py_err(e: crate::PdfError) -> PyErr {
PyValueError::new_err(e.to_string())
}
struct PythonOcrOptions {
mode: String,
page_numbers: Option<Vec<u32>>,
password: Option<String>,
dpi: f32,
minimum_confidence: f32,
hosted_recommendation_confidence: f32,
model_directory: Option<String>,
offline: bool,
}
fn build_ocr_options(binding: PythonOcrOptions) -> PyResult<crate::vision::OcrPdfOptions> {
let mode = match binding.mode.trim().to_ascii_lowercase().as_str() {
"off" => crate::vision::OcrMode::Off,
"auto" => crate::vision::OcrMode::Auto,
"force" => crate::vision::OcrMode::Force,
_ => {
return Err(PyValueError::new_err(
"mode must be 'off', 'auto', or 'force'",
));
}
};
let mut options = crate::vision::OcrPdfOptions::new().mode(mode);
options.render.dpi = binding.dpi;
options.ocr.minimum_confidence = binding.minimum_confidence;
options.hosted_recommendation_confidence = binding.hosted_recommendation_confidence;
if let Some(pages) = binding.page_numbers {
options = options.page_numbers(pages);
}
if let Some(password) = binding.password {
options = options.password(password);
}
if let Some(directory) = binding.model_directory {
options.ocr.model_directory = Some(directory.into());
}
if binding.offline {
options.ocr.model_downloads = crate::vision::ModelDownloadPolicy::Offline;
}
Ok(options)
}
fn page_content_source_str(source: crate::vision::PageContentSource) -> String {
match source {
crate::vision::PageContentSource::Native => "native".into(),
crate::vision::PageContentSource::Ocr => "ocr".into(),
crate::vision::PageContentSource::Fused => "fused".into(),
}
}
fn to_py_ocr_result(result: crate::vision::OcrPdfResult) -> PyOcrPdfResult {
PyOcrPdfResult {
markdown: result.markdown,
pages: result
.pages
.into_iter()
.map(|page| {
let provenance = page.provenance;
PyOcrPageResult {
page_number: page.page_number,
markdown: page.markdown,
provenance: PyOcrPageProvenance {
page_number: provenance.page_number,
source: page_content_source_str(provenance.source),
ocr_model: provenance.ocr_model.map(|model| PyOcrModelIdentity {
name: model.name,
revision: model.revision,
}),
render_dpi: provenance.render_dpi,
ocr_confidence: provenance.ocr_confidence,
timings: PyOcrTimings {
render_ms: provenance.timings.render_ms,
ocr_ms: provenance.timings.ocr_ms,
assembly_ms: provenance.timings.assembly_ms,
},
warnings: provenance.warnings,
hosted_recommended: provenance.hosted_recommended,
},
}
})
.collect(),
page_count: result.page_count,
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
pages_routed_to_ocr: result.pages_routed_to_ocr,
pages_recommending_hosted: result.pages_recommending_hosted,
ocr_reasons_by_page: to_py_page_ocr_reasons(result.ocr_reasons_by_page),
pages_with_tables: result.pages_with_tables,
pages_with_columns: result.pages_with_columns,
is_complex: result.is_complex,
processing_time_ms: result.processing_time_ms,
render_time_ms: result.render_time_ms,
ocr_time_ms: result.ocr_time_ms,
}
}
fn item_type_str(t: &ItemType) -> String {
match t {
ItemType::Text => "text".into(),
@@ -698,99 +502,6 @@ fn process_pdf_bytes(data: &[u8], pages: Option<Vec<u32>>) -> PyResult<PyPdfResu
Ok(to_py_result(result))
}
/// Process a PDF file through native extraction and selective OCR.
///
/// OCR defaults to ``auto`` and only initializes its external runtime and
/// model when native quality signals route at least one page. Page numbers
/// are 1-indexed. The GIL is released for the complete processing call.
#[pyfunction]
#[pyo3(signature = (
path,
*,
mode="auto",
page_numbers=None,
password=None,
dpi=150.0,
minimum_confidence=0.0,
hosted_recommendation_confidence=0.5,
model_directory=None,
offline=false
))]
#[allow(clippy::too_many_arguments)]
fn process_pdf_with_ocr(
py: Python<'_>,
path: String,
mode: &str,
page_numbers: Option<Vec<u32>>,
password: Option<String>,
dpi: f32,
minimum_confidence: f32,
hosted_recommendation_confidence: f32,
model_directory: Option<String>,
offline: bool,
) -> PyResult<PyOcrPdfResult> {
let options = build_ocr_options(PythonOcrOptions {
mode: mode.to_string(),
page_numbers,
password,
dpi,
minimum_confidence,
hosted_recommendation_confidence,
model_directory,
offline,
})?;
let result = py
.allow_threads(move || crate::vision::process_pdf_with_ocr(path, options))
.map_err(|error| PyValueError::new_err(error.to_string()))?;
Ok(to_py_ocr_result(result))
}
/// Process PDF bytes through native extraction and selective OCR.
///
/// See [`process_pdf_with_ocr`] for options and result semantics.
#[pyfunction]
#[pyo3(signature = (
data,
*,
mode="auto",
page_numbers=None,
password=None,
dpi=150.0,
minimum_confidence=0.0,
hosted_recommendation_confidence=0.5,
model_directory=None,
offline=false
))]
#[allow(clippy::too_many_arguments)]
fn process_pdf_with_ocr_bytes(
py: Python<'_>,
data: &[u8],
mode: &str,
page_numbers: Option<Vec<u32>>,
password: Option<String>,
dpi: f32,
minimum_confidence: f32,
hosted_recommendation_confidence: f32,
model_directory: Option<String>,
offline: bool,
) -> PyResult<PyOcrPdfResult> {
let options = build_ocr_options(PythonOcrOptions {
mode: mode.to_string(),
page_numbers,
password,
dpi,
minimum_confidence,
hosted_recommendation_confidence,
model_directory,
offline,
})?;
let data = data.to_vec();
let result = py
.allow_threads(move || crate::vision::process_pdf_with_ocr_mem(&data, options))
.map_err(|error| PyValueError::new_err(error.to_string()))?;
Ok(to_py_ocr_result(result))
}
/// Fast detection only — no text extraction or markdown.
#[pyfunction]
fn detect_pdf(path: &str) -> PyResult<PyPdfResult> {
@@ -993,11 +704,6 @@ fn extract_structure_elements_bytes(
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<PyPdfResult>()?;
m.add_class::<PyPageOcrReasons>()?;
m.add_class::<PyOcrModelIdentity>()?;
m.add_class::<PyOcrTimings>()?;
m.add_class::<PyOcrPageProvenance>()?;
m.add_class::<PyOcrPageResult>()?;
m.add_class::<PyOcrPdfResult>()?;
m.add_class::<PyPdfClassification>()?;
m.add_class::<PyTextItem>()?;
m.add_class::<PyStructureElement>()?;
@@ -1007,8 +713,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<PyPagesExtractionResult>()?;
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
m.add_function(wrap_pyfunction!(process_pdf_with_ocr, m)?)?;
m.add_function(wrap_pyfunction!(process_pdf_with_ocr_bytes, m)?)?;
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
m.add_function(wrap_pyfunction!(detect_pdf_bytes, m)?)?;
m.add_function(wrap_pyfunction!(classify_pdf, m)?)?;
-17
View File
@@ -442,16 +442,6 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
fn is_footnote_row(text: &str) -> bool {
let trimmed = text.trim();
// Japanese documents commonly use the reference mark followed by an
// ASCII or full-width number (for example `※1` / `※1`). These rows often
// sit immediately below a wide table and must not be merged into its last
// data row as wrapped first-column content.
if let Some(rest) = trimmed.strip_prefix('※') {
return rest.chars().next().is_some_and(|character| {
character.is_ascii_digit() || (''..='').contains(&character)
});
}
// Check for common footnote patterns
// (1), (2), etc.
if trimmed.starts_with('(') && trimmed.len() >= 2 {
@@ -513,13 +503,6 @@ mod tests {
assert!(is_footnote_row("NOTES: uppercase"));
}
#[test]
fn test_is_footnote_row_reference_mark_number() {
assert!(is_footnote_row("※1 explanation"));
assert!(is_footnote_row("※1 説明"));
assert!(!is_footnote_row("※ general marker"));
}
#[test]
fn test_is_footnote_row_plain_text_false() {
assert!(!is_footnote_row("Regular cell text"));
+151 -9
View File
@@ -1,4 +1,4 @@
//! Public contracts between rendering, OCR, and orchestration.
//! Public contracts between rendering, OCR, layout, and orchestration.
use std::error::Error;
use std::path::PathBuf;
@@ -18,6 +18,19 @@ pub enum OcrMode {
Force,
}
/// Resource/quality profile for the OCR engine.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
#[non_exhaustive]
pub enum OcrProfile {
/// Lowest latency and memory footprint.
Edge,
/// OCR-oriented balance of quality and CPU cost.
#[default]
Balanced,
/// Highest quality within the lightweight model family.
Quality,
}
/// Controls whether missing model artifacts may be fetched.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
#[non_exhaustive]
@@ -34,8 +47,12 @@ pub enum ModelDownloadPolicy {
pub struct OcrOptions {
/// Page-level routing behavior.
pub mode: OcrMode,
/// Local quality/resource profile.
pub profile: OcrProfile,
/// Drop recognition spans below this confidence threshold.
pub minimum_confidence: f32,
/// Optional language hints understood by the selected engine.
pub languages: Vec<String>,
/// Optional directory containing an offline model set.
pub model_directory: Option<PathBuf>,
/// Whether a missing pinned artifact may be downloaded.
@@ -46,7 +63,9 @@ impl Default for OcrOptions {
fn default() -> Self {
Self {
mode: OcrMode::Off,
profile: OcrProfile::Balanced,
minimum_confidence: 0.0,
languages: Vec::new(),
model_directory: None,
model_downloads: ModelDownloadPolicy::IfMissing,
}
@@ -65,12 +84,24 @@ impl OcrOptions {
self
}
/// Sets the local resource/quality profile.
pub fn profile(mut self, profile: OcrProfile) -> Self {
self.profile = profile;
self
}
/// Sets the minimum accepted recognition confidence.
pub fn minimum_confidence(mut self, minimum_confidence: f32) -> Self {
self.minimum_confidence = minimum_confidence;
self
}
/// Replaces the language hints passed to the OCR engine.
pub fn languages(mut self, languages: impl IntoIterator<Item = impl Into<String>>) -> Self {
self.languages = languages.into_iter().map(Into::into).collect();
self
}
/// Uses an explicit model directory, suitable for offline packaging.
pub fn model_directory(mut self, directory: impl Into<PathBuf>) -> Self {
self.model_directory = Some(directory.into());
@@ -84,6 +115,55 @@ impl OcrOptions {
}
}
/// Configuration for an optional learned layout engine.
///
/// Layout inference is disabled by default. Existing deterministic layout,
/// table, and Markdown logic remains the assembly path when this is disabled.
#[derive(Debug, Clone, PartialEq)]
pub struct LayoutOptions {
/// Whether the learned layout extension may run.
pub enabled: bool,
/// Drop layout regions below this confidence threshold.
pub minimum_confidence: f32,
/// Optional directory containing an offline layout model set.
pub model_directory: Option<PathBuf>,
}
impl Default for LayoutOptions {
fn default() -> Self {
Self {
enabled: false,
minimum_confidence: 0.0,
model_directory: None,
}
}
}
impl LayoutOptions {
/// Creates layout options with learned layout disabled.
pub fn new() -> Self {
Self::default()
}
/// Enables or disables learned layout inference.
pub fn enabled(mut self, enabled: bool) -> Self {
self.enabled = enabled;
self
}
/// Sets the minimum accepted region confidence.
pub fn minimum_confidence(mut self, minimum_confidence: f32) -> Self {
self.minimum_confidence = minimum_confidence;
self
}
/// Uses an explicit layout model directory.
pub fn model_directory(mut self, directory: impl Into<PathBuf>) -> Self {
self.model_directory = Some(directory.into());
self
}
}
/// A point in bitmap space, measured from the top-left in pixels.
#[derive(Debug, Clone, Copy, Default, PartialEq)]
pub struct ImagePoint {
@@ -150,7 +230,7 @@ pub struct OcrSpan {
#[derive(Debug, Clone, PartialEq)]
pub struct OcrPage {
/// 1-indexed PDF page number.
pub page_number: u32,
pub page: u32,
/// Positioned recognition spans.
pub spans: Vec<OcrSpan>,
/// Mean confidence across accepted spans, when available.
@@ -163,6 +243,54 @@ pub struct OcrPage {
pub warnings: Vec<String>,
}
/// Normalized semantic class emitted by a learned layout engine.
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub enum LayoutRegionKind {
/// Body or other prose text.
Text,
/// Document heading or title.
Heading,
/// Table region.
Table,
/// Figure/image region.
Figure,
/// Figure or table caption.
Caption,
/// Header/footer/page furniture.
Furniture,
/// Model-specific class retained without changing the common taxonomy.
Other(String),
}
/// One learned layout region in bitmap coordinates.
#[derive(Debug, Clone, PartialEq)]
pub struct LayoutRegion {
/// Normalized semantic class.
pub kind: LayoutRegionKind,
/// Region polygon in the original rendered page's pixel space.
pub polygon: ImageQuad,
/// Model confidence in the inclusive range 01.
pub confidence: f32,
/// Optional model-provided reading-order position.
pub reading_order: Option<u32>,
}
/// Learned layout output for one 1-indexed page.
#[derive(Debug, Clone, PartialEq)]
pub struct LayoutPage {
/// 1-indexed PDF page number.
pub page: u32,
/// Semantic regions.
pub regions: Vec<LayoutRegion>,
/// Exact model identity used for this result.
pub model: ModelIdentity,
/// Layout inference wall time for this page.
pub processing_time_ms: u64,
/// Non-fatal engine warnings.
pub warnings: Vec<String>,
}
/// How final page content was sourced.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[non_exhaustive]
@@ -182,6 +310,8 @@ pub struct VisionTimings {
pub render_ms: u64,
/// OCR wall time.
pub ocr_ms: u64,
/// Optional learned layout wall time.
pub layout_ms: u64,
/// Native/OCR fusion and assembly wall time.
pub assembly_ms: u64,
}
@@ -190,11 +320,13 @@ pub struct VisionTimings {
#[derive(Debug, Clone, PartialEq)]
pub struct PageProvenance {
/// 1-indexed PDF page number.
pub page_number: u32,
pub page: u32,
/// Final page-content source.
pub source: PageContentSource,
/// OCR model, when OCR ran.
pub ocr_model: Option<ModelIdentity>,
/// Learned layout model, when layout inference ran.
pub layout_model: Option<ModelIdentity>,
/// Render resolution used for local vision.
pub render_dpi: Option<f32>,
/// Mean accepted OCR confidence, when available.
@@ -237,13 +369,23 @@ pub trait OcrEngine: Send + Sync {
pages: &[RenderedPage],
options: &OcrOptions,
) -> Result<Vec<OcrPage>, Self::Error>;
}
/// Number of pages this engine can process concurrently in one
/// `recognize` call. The pipeline sizes its page batches from this so a
/// parallel engine is not starved by small chunks; `1` means sequential.
fn preferred_page_concurrency(&self) -> usize {
1
}
/// Optional learned semantic layout extension.
pub trait LayoutEngine: Send + Sync {
/// Engine-specific failure type.
type Error: Error + Send + Sync + 'static;
/// Exact model identity used by this engine instance.
fn model(&self) -> &ModelIdentity;
/// Analyzes rendered pages, optionally using their OCR spans.
fn analyze(
&self,
pages: &[RenderedPage],
ocr: &[OcrPage],
options: &LayoutOptions,
) -> Result<Vec<LayoutPage>, Self::Error>;
}
#[cfg(test)]
+16 -659
View File
@@ -6,7 +6,6 @@ use std::time::Instant;
use thiserror::Error;
use crate::markdown::{to_markdown_from_items_with_rects_and_page_count, MarkdownOptions};
use crate::text_quality::{detect_encoding_issues, is_cid_garbage, is_garbage_text};
use crate::types::{ItemType, TextItem};
use crate::PageMarkdown;
@@ -74,7 +73,7 @@ impl OcrFusionOptions {
#[derive(Debug, Clone, PartialEq)]
pub struct FusedPageMarkdown {
/// 1-indexed document page number, matching OCR and provenance fields.
pub page_number: u32,
pub page: u32,
/// Final page Markdown.
pub markdown: String,
/// Native/OCR source, model, timing, and fallback metadata.
@@ -92,73 +91,6 @@ pub struct FusedPages {
pub ocr_time_ms: u64,
}
/// Origin of a trustworthy native-text candidate retained for adaptive OCR.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum NativeCandidateOrigin {
/// Text produced by pdf-inspector's normal native extractor.
Extractor,
/// Positioned text independently recovered through PDFium.
Pdfium,
}
impl NativeCandidateOrigin {
fn description(self) -> &'static str {
match self {
Self::Extractor => "native extraction",
Self::Pdfium => "PDFium native recovery",
}
}
}
#[derive(Debug, Clone, Copy)]
struct TextCandidateQuality {
alphanumeric_chars: usize,
score: f32,
}
/// Clean native text retained while an ambiguous page is compared with OCR.
#[derive(Debug, Clone)]
pub(crate) struct NativeFallbackCandidate {
markdown: String,
quality: TextCandidateQuality,
origin: NativeCandidateOrigin,
}
impl NativeFallbackCandidate {
/// True when an independent native recovery is substantial enough to
/// cancel OCR for recoverable font/vector routing reasons.
pub(crate) fn is_complete_recovery(&self) -> bool {
self.quality.alphanumeric_chars >= 40 && self.quality.score >= 0.68
}
pub(crate) fn markdown(&self) -> &str {
&self.markdown
}
pub(crate) fn is_stronger_than(&self, other: &Self) -> bool {
self.quality.alphanumeric_chars > other.quality.alphanumeric_chars
|| (self.quality.alphanumeric_chars == other.quality.alphanumeric_chars
&& self.quality.score > other.quality.score)
}
}
/// Validates and scores native Markdown for possible post-OCR comparison.
///
/// This intentionally uses script-agnostic evidence. A native candidate only
/// needs to be trustworthy, not necessarily complete: a clean native header
/// can still be fused with an image-backed OCR body.
pub(crate) fn assess_native_candidate(
markdown: String,
origin: NativeCandidateOrigin,
) -> Option<NativeFallbackCandidate> {
let quality = assess_text_candidate(&markdown)?;
(quality.alphanumeric_chars >= 8).then_some(NativeFallbackCandidate {
markdown,
quality,
origin,
})
}
/// Converts positioned OCR spans to Markdown through pdf-inspector's existing
/// deterministic geometry, reading-order, table, and Markdown pipeline.
///
@@ -170,13 +102,12 @@ pub fn ocr_page_to_markdown(
options: &MarkdownOptions,
) -> String {
let (items, _) = ocr_text_items(page);
let markdown = to_markdown_from_items_with_rects_and_page_count(
to_markdown_from_items_with_rects_and_page_count(
items,
options.clone(),
&[],
document_page_count,
);
preserve_ocr_line_breaks(&markdown, page)
)
}
/// Fuses a selective OCR run into per-page native Markdown.
@@ -184,47 +115,13 @@ pub fn ocr_page_to_markdown(
/// OCR replaces pages whose native extraction was already rejected. On clean
/// native pages (for example in `Force` mode), normalized duplicate OCR blocks
/// are removed and only genuinely additional blocks are appended. Pages that
/// needed OCR but still have no credible OCR result, or whose confident OCR
/// only repeats an incomplete native fragment, recommend the hosted document
/// pipeline instead of silently presenting partial content as final.
/// needed OCR but still have no credible local result recommend the hosted
/// document pipeline instead of silently presenting an empty result as final.
pub fn fuse_ocr_pages(
native_pages: &[PageMarkdown],
ocr_run: &OcrRun,
document_page_count: u32,
options: &OcrFusionOptions,
) -> Result<FusedPages, OcrFusionError> {
fuse_ocr_pages_impl(
native_pages,
ocr_run,
document_page_count,
options,
&BTreeMap::new(),
)
}
/// OCR-pipeline fusion with trustworthy partial native candidates.
pub(crate) fn fuse_ocr_pages_adaptive(
native_pages: &[PageMarkdown],
ocr_run: &OcrRun,
document_page_count: u32,
options: &OcrFusionOptions,
native_candidates: &BTreeMap<u32, NativeFallbackCandidate>,
) -> Result<FusedPages, OcrFusionError> {
fuse_ocr_pages_impl(
native_pages,
ocr_run,
document_page_count,
options,
native_candidates,
)
}
fn fuse_ocr_pages_impl(
native_pages: &[PageMarkdown],
ocr_run: &OcrRun,
document_page_count: u32,
options: &OcrFusionOptions,
native_candidates: &BTreeMap<u32, NativeFallbackCandidate>,
) -> Result<FusedPages, OcrFusionError> {
validate_options(options)?;
@@ -279,32 +176,16 @@ fn fuse_ocr_pages_impl(
&[],
document_page_count,
);
let ocr_markdown = preserve_ocr_line_breaks(&ocr_markdown, local);
let (markdown, source, adaptive_recommends_hosted) = if let Some(candidate) =
native_candidates
.get(&page_number)
.filter(|_| native.needs_ocr)
{
let choice = choose_adaptive_content(
candidate,
&ocr_markdown,
local.ocr.mean_confidence,
options.hosted_recommendation_confidence,
);
warnings.push(choice.warning);
(choice.markdown, choice.source, choice.recommend_hosted)
} else if native.markdown.trim().is_empty() || native.needs_ocr {
(ocr_markdown, PageContentSource::Ocr, false)
let (markdown, source) = if native.markdown.trim().is_empty() || native.needs_ocr {
(ocr_markdown, PageContentSource::Ocr)
} else {
let (markdown, source) = merge_native_and_ocr(&native.markdown, &ocr_markdown);
(markdown, source, false)
merge_native_and_ocr(&native.markdown, &ocr_markdown)
};
let weak_ocr = local
.ocr
.mean_confidence
.is_none_or(|confidence| confidence < options.hosted_recommendation_confidence);
let recommend_hosted = native.needs_ocr
&& (markdown.trim().is_empty() || weak_ocr || adaptive_recommends_hosted);
let recommend_hosted = native.needs_ocr && (markdown.trim().is_empty() || weak_ocr);
if native.needs_ocr && markdown.trim().is_empty() {
warnings.push("OCR produced no usable text".to_string());
}
@@ -331,12 +212,13 @@ fn fuse_ocr_pages_impl(
};
pages.push(FusedPageMarkdown {
page_number,
page: page_number,
markdown,
provenance: PageProvenance {
page_number,
page: page_number,
source,
ocr_model,
layout_model: None,
render_dpi: ocr_by_page
.contains_key(&page_number)
.then_some(options.render_dpi),
@@ -344,6 +226,7 @@ fn fuse_ocr_pages_impl(
timings: VisionTimings {
render_ms: render_by_page.get(&page_number).copied().unwrap_or(0),
ocr_ms,
layout_ms: 0,
assembly_ms: elapsed_ms(assembly_started),
},
warnings,
@@ -359,159 +242,6 @@ fn fuse_ocr_pages_impl(
})
}
struct AdaptiveContentChoice {
markdown: String,
source: PageContentSource,
warning: String,
recommend_hosted: bool,
}
fn choose_adaptive_content(
native: &NativeFallbackCandidate,
ocr: &str,
ocr_confidence: Option<f32>,
weak_ocr_threshold: f32,
) -> AdaptiveContentChoice {
let ocr_quality = assess_text_candidate(ocr);
let weak_ocr = ocr_confidence.is_none_or(|confidence| confidence < weak_ocr_threshold);
if weak_ocr || ocr_quality.is_none() {
return AdaptiveContentChoice {
markdown: ensure_trailing_newline(native.markdown()),
source: PageContentSource::Native,
warning: format!(
"kept trustworthy {} because OCR was weak or unusable",
native.origin.description()
),
recommend_hosted: true,
};
}
let ocr_quality = ocr_quality.expect("checked above");
let overlap = content_overlap(native.markdown(), ocr);
let ocr_novel_chars = ocr_quality
.alphanumeric_chars
.saturating_sub(overlap.shared_chars);
let material_novelty =
ocr_novel_chars >= 2 && ocr_novel_chars * 8 >= ocr_quality.alphanumeric_chars.max(1);
let native_substantially_covered =
overlap.shared_chars * 4 >= native.quality.alphanumeric_chars.max(1) * 3;
if native_substantially_covered && !material_novelty {
return AdaptiveContentChoice {
markdown: ensure_trailing_newline(native.markdown()),
source: PageContentSource::Native,
warning: format!(
"kept trustworthy {} because OCR added no material coverage",
native.origin.description()
),
// The candidate exists only because this page was routed with
// incomplete native coverage. Agreement between two partial
// hypotheses preserves trustworthy text, but does not prove that
// the rest of the page was recovered.
recommend_hosted: true,
};
}
// A shorter, materially lower-quality OCR hypothesis should not displace
// exact native text even when its mean confidence happens to be high.
if ocr_quality.score + 0.12 < native.quality.score
&& ocr_quality.alphanumeric_chars * 10
<= native.quality.alphanumeric_chars.saturating_mul(11)
{
return AdaptiveContentChoice {
markdown: ensure_trailing_newline(native.markdown()),
source: PageContentSource::Native,
warning: format!(
"kept higher-quality {} after comparing OCR",
native.origin.description()
),
recommend_hosted: false,
};
}
let (markdown, source) = merge_native_and_ocr(native.markdown(), ocr);
let warning = match source {
PageContentSource::Native => format!(
"kept trustworthy {} because OCR duplicated its content",
native.origin.description()
),
PageContentSource::Fused => format!(
"fused trustworthy {} with complementary OCR",
native.origin.description()
),
PageContentSource::Ocr => unreachable!("merge never returns OCR-only content"),
};
AdaptiveContentChoice {
markdown,
source,
warning,
recommend_hosted: false,
}
}
#[derive(Debug, Clone, Copy)]
struct ContentOverlap {
shared_chars: usize,
}
fn content_overlap(first: &str, second: &str) -> ContentOverlap {
let mut first_counts = BTreeMap::<char, usize>::new();
for character in normalized_content_chars(first) {
*first_counts.entry(character).or_insert(0) += 1;
}
let mut second_counts = BTreeMap::<char, usize>::new();
for character in normalized_content_chars(second) {
*second_counts.entry(character).or_insert(0) += 1;
}
let shared_chars = first_counts
.iter()
.map(|(character, count)| (*count).min(second_counts.get(character).copied().unwrap_or(0)))
.sum();
ContentOverlap { shared_chars }
}
fn normalized_content_chars(text: &str) -> impl Iterator<Item = char> + '_ {
text.chars()
.flat_map(char::to_lowercase)
.filter(|character| character.is_alphanumeric())
}
fn assess_text_candidate(markdown: &str) -> Option<TextCandidateQuality> {
if markdown.trim().is_empty()
|| is_garbage_text(markdown)
|| is_cid_garbage(markdown)
|| detect_encoding_issues(markdown)
{
return None;
}
let alphanumeric_chars = markdown
.chars()
.filter(|character| character.is_alphanumeric())
.count();
if alphanumeric_chars == 0 {
return None;
}
let visible_chars = markdown
.chars()
.filter(|character| !character.is_whitespace())
.count()
.max(1);
let density = alphanumeric_chars as f32 / visible_chars as f32;
let length_score = (alphanumeric_chars as f32 / 160.0).min(1.0);
let nonempty_lines = markdown
.lines()
.filter(|line| !line.trim().is_empty())
.count()
.max(1);
let line_score = (alphanumeric_chars as f32 / nonempty_lines as f32 / 12.0).min(1.0);
let score = (0.45 + length_score * 0.25 + density * 0.20 + line_score * 0.10).min(1.0);
Some(TextCandidateQuality {
alphanumeric_chars,
score,
})
}
/// Converts recognized line polygons to ordinary PDF-space text items.
fn ocr_text_items(page: &RoutedOcrPage) -> (Vec<TextItem>, usize) {
let mut discarded = 0usize;
@@ -599,122 +329,6 @@ fn image_quad_bounds(
(right > left && bottom > top).then_some((left, top, right, bottom))
}
fn preserve_ocr_line_breaks(markdown: &str, page: &RoutedOcrPage) -> String {
let spans: Vec<(&str, f32, f32, f32, f32)> = page
.ocr
.spans
.iter()
.filter_map(|span| {
let (left, top, right, bottom) = image_quad_bounds(
&span.polygon.points,
page.rendered.width(),
page.rendered.height(),
)?;
(!span.text.trim().is_empty()).then_some((span.text.trim(), left, top, right, bottom))
})
.collect();
if spans.len() < 2 {
return markdown.to_string();
}
let mut line_heights: Vec<f32> = spans
.iter()
.map(|(_, _, top, _, bottom)| bottom - top)
.filter(|height| height.is_finite() && *height > 0.0)
.collect();
if line_heights.is_empty() {
return markdown.to_string();
}
line_heights.sort_by(f32::total_cmp);
let median_height = line_heights[line_heights.len() / 2];
// The Markdown converter owns reading order and may normalize syntax such
// as list markers. Match every span back to its unique output occurrence,
// then use Markdown order rather than imposing a second geometry sort.
// If the mapping is incomplete or ambiguous, leave the converter output
// untouched instead of risking a break at the wrong duplicate text.
let mut mapped = Vec::with_capacity(spans.len());
for (text, left, top, right, bottom) in spans {
let Some((start, end)) = unique_markdown_span(markdown, text) else {
return markdown.to_string();
};
mapped.push((start, end, left, top, right, bottom));
}
mapped.sort_by_key(|span| span.0);
if mapped
.windows(2)
.any(|pair| pair[0].1 > pair[1].0 || pair[0].0 == pair[1].0)
{
return markdown.to_string();
}
let mut replacements = Vec::new();
for pair in mapped.windows(2) {
let (_, current_end, current_left, _, current_right, current_bottom) = pair[0];
let (next_start, _, next_left, next_top, next_right, _) = pair[1];
let overlap = (current_right.min(next_right) - current_left.max(next_left)).max(0.0);
let narrowest_width = (current_right - current_left).min(next_right - next_left);
let same_text_flow = narrowest_width > 0.0 && overlap >= narrowest_width * 0.2;
let separated = next_top - current_bottom >= median_height * 0.65;
let between = &markdown[current_end..next_start];
if same_text_flow
&& separated
&& between.chars().all(char::is_whitespace)
&& !markdown_line_at(markdown, current_end).is_some_and(is_markdown_table_line)
&& !markdown_line_at(markdown, next_start).is_some_and(is_markdown_table_line)
{
replacements.push((current_end, next_start));
}
}
let mut output = markdown.to_string();
for (start, end) in replacements.into_iter().rev() {
output.replace_range(start..end, "\n\n");
}
output
}
fn unique_markdown_span(markdown: &str, span_text: &str) -> Option<(usize, usize)> {
let exact: Vec<_> = markdown.match_indices(span_text).collect();
match exact.as_slice() {
[(start, matched)] => return Some((*start, *start + matched.len())),
[] => {}
_ => return None,
}
let normalized = strip_list_marker(span_text)?;
let matches: Vec<_> = markdown.match_indices(normalized).collect();
match matches.as_slice() {
[(start, matched)] => Some((*start, *start + matched.len())),
_ => None,
}
}
fn strip_list_marker(text: &str) -> Option<&str> {
const BULLETS: &[char] = &['•', '●', '○', '◦', '▪', '', '—'];
let trimmed = text.trim_start();
let remainder = trimmed
.strip_prefix(BULLETS)?
.trim_start_matches(char::is_whitespace);
(!remainder.is_empty()).then_some(remainder)
}
fn markdown_line_at(markdown: &str, offset: usize) -> Option<&str> {
if offset > markdown.len() || !markdown.is_char_boundary(offset) {
return None;
}
let start = markdown[..offset].rfind('\n').map_or(0, |index| index + 1);
let end = markdown[offset..]
.find('\n')
.map_or(markdown.len(), |index| offset + index);
markdown.get(start..end)
}
fn is_markdown_table_line(line: &str) -> bool {
let trimmed = line.trim();
trimmed.starts_with('|') && trimmed.ends_with('|') && trimmed.matches('|').count() >= 2
}
fn merge_native_and_ocr(native: &str, ocr: &str) -> (String, PageContentSource) {
let native_keys = comparison_units(native);
let mut addition_keys = Vec::new();
@@ -1014,7 +628,7 @@ mod tests {
RoutedOcrPage {
rendered: rendered_page(page),
ocr: OcrPage {
page_number: page,
page,
spans,
mean_confidence: confidence,
model: ModelIdentity::new("test-ocr", "v1"),
@@ -1024,20 +638,6 @@ mod tests {
}
}
fn positioned_span(text: &str, left: f32, top: f32, right: f32, bottom: f32) -> OcrSpan {
OcrSpan {
text: text.to_string(),
polygon: ImageQuad::new([
ImagePoint::new(left, top),
ImagePoint::new(right, top),
ImagePoint::new(right, bottom),
ImagePoint::new(left, bottom),
]),
confidence: 0.9,
orientation_degrees: None,
}
}
fn run(pages: Vec<RoutedOcrPage>) -> OcrRun {
OcrRun {
pages,
@@ -1046,10 +646,6 @@ mod tests {
}
}
fn native_candidate(markdown: &str) -> NativeFallbackCandidate {
assess_native_candidate(markdown.to_string(), NativeCandidateOrigin::Extractor).unwrap()
}
#[test]
fn scanned_page_uses_geometry_ordered_ocr_and_provenance() {
let native = [native(0, "", true)];
@@ -1069,11 +665,8 @@ mod tests {
< result.pages[0].markdown.find("Second").unwrap()
);
assert_eq!(result.pages[0].provenance.source, PageContentSource::Ocr);
assert_eq!(result.pages[0].page_number, 1);
assert_eq!(
result.pages[0].page_number,
result.pages[0].provenance.page_number
);
assert_eq!(result.pages[0].page, 1);
assert_eq!(result.pages[0].page, result.pages[0].provenance.page);
assert_eq!(
result.pages[0].provenance.ocr_model.as_ref().unwrap().name,
"test-ocr"
@@ -1081,115 +674,6 @@ mod tests {
assert!(!result.pages[0].provenance.hosted_recommended);
}
#[test]
fn ocr_assembly_preserves_well_separated_detected_rows() {
let page = routed_page(
1,
vec![
positioned_span("Column A Column B", 10.0, 10.0, 190.0, 20.0),
positioned_span("First row value", 10.0, 35.0, 190.0, 45.0),
positioned_span("Second row value", 10.0, 60.0, 190.0, 70.0),
],
Some(0.9),
);
let markdown = ocr_page_to_markdown(&page, 1, &MarkdownOptions::default());
assert!(markdown.contains("Column A Column B\n\n"), "{markdown:?}");
assert!(markdown.contains("First row value\n\n"), "{markdown:?}");
}
#[test]
fn line_break_recovery_uses_markdown_reading_order_for_columns() {
let page = routed_page(
1,
vec![
positioned_span("Left top", 10.0, 10.0, 90.0, 20.0),
positioned_span("Right top", 110.0, 10.0, 190.0, 20.0),
positioned_span("Left bottom", 10.0, 40.0, 90.0, 50.0),
positioned_span("Right bottom", 110.0, 40.0, 190.0, 50.0),
],
Some(0.9),
);
let markdown = "Left top Left bottom\n\nRight top Right bottom";
let recovered = preserve_ocr_line_breaks(markdown, &page);
assert_eq!(
recovered,
"Left top\n\nLeft bottom\n\nRight top\n\nRight bottom"
);
}
#[test]
fn line_break_recovery_preserves_tables_but_handles_page_prose() {
let page = routed_page(
1,
vec![
positioned_span("A", 10.0, 10.0, 90.0, 20.0),
positioned_span("B", 110.0, 10.0, 190.0, 20.0),
positioned_span("x", 10.0, 30.0, 90.0, 40.0),
positioned_span("y", 110.0, 30.0, 190.0, 40.0),
positioned_span("First prose", 10.0, 60.0, 190.0, 70.0),
positioned_span("Second prose", 10.0, 90.0, 190.0, 100.0),
],
Some(0.9),
);
let markdown = "| A | B |\n|---|---|\n| x | y |\n\nFirst prose Second prose";
let recovered = preserve_ocr_line_breaks(markdown, &page);
assert!(recovered.starts_with("| A | B |\n|---|---|\n| x | y |"));
assert!(recovered.ends_with("First prose\n\nSecond prose"));
}
#[test]
fn line_break_recovery_accepts_normalized_list_markers() {
let page = routed_page(
1,
vec![
positioned_span("• First item", 10.0, 10.0, 190.0, 20.0),
positioned_span("Next paragraph", 10.0, 40.0, 190.0, 50.0),
],
Some(0.9),
);
let recovered = preserve_ocr_line_breaks("- First item Next paragraph", &page);
assert_eq!(recovered, "- First item\n\nNext paragraph");
}
#[test]
fn line_break_recovery_accepts_white_bullet_list_markers() {
let page = routed_page(
1,
vec![
positioned_span("◦ First item", 10.0, 10.0, 190.0, 20.0),
positioned_span("Next paragraph", 10.0, 40.0, 190.0, 50.0),
],
Some(0.9),
);
let recovered = preserve_ocr_line_breaks("- First item Next paragraph", &page);
assert_eq!(recovered, "- First item\n\nNext paragraph");
}
#[test]
fn line_break_recovery_leaves_ambiguous_duplicates_unchanged() {
let page = routed_page(
1,
vec![
positioned_span("Repeated", 10.0, 10.0, 190.0, 20.0),
positioned_span("Repeated", 10.0, 40.0, 190.0, 50.0),
],
Some(0.9),
);
let markdown = "Repeated Repeated";
assert_eq!(preserve_ocr_line_breaks(markdown, &page), markdown);
}
#[test]
fn force_mode_deduplicates_native_content() {
let native = [native(0, "Hello, world!\n", false)];
@@ -1205,133 +689,6 @@ mod tests {
assert_eq!(result.pages[0].provenance.source, PageContentSource::Native);
}
#[test]
fn adaptive_fallback_keeps_native_text_when_ocr_is_weak() {
let native = [native(0, "", true)];
let run = run(vec![routed_page(
1,
vec![span("Inv0ice total uncertain", 10.0, 0.3)],
Some(0.3),
)]);
let candidates = BTreeMap::from([(
1,
native_candidate("Invoice total: $420.00\nPayment received\n"),
)]);
let result =
fuse_ocr_pages_adaptive(&native, &run, 1, &OcrFusionOptions::new(), &candidates)
.unwrap();
assert_eq!(result.pages[0].provenance.source, PageContentSource::Native);
assert_eq!(
result.pages[0].markdown,
"Invoice total: $420.00\nPayment received\n"
);
assert!(result.pages[0].provenance.hosted_recommended);
assert!(result.pages[0].provenance.warnings[0].contains("OCR was weak"));
}
#[test]
fn adaptive_fallback_prefers_exact_native_text_over_duplicate_ocr() {
let native = [native(0, "", true)];
let run = run(vec![routed_page(
1,
vec![span("Invoice total 420.00 Payment received", 10.0, 0.98)],
Some(0.98),
)]);
let candidates = BTreeMap::from([(
1,
native_candidate("Invoice total: $420.00\nPayment received\n"),
)]);
let result =
fuse_ocr_pages_adaptive(&native, &run, 1, &OcrFusionOptions::new(), &candidates)
.unwrap();
assert_eq!(result.pages[0].provenance.source, PageContentSource::Native);
assert!(result.pages[0].provenance.hosted_recommended);
assert_eq!(result.pages[0].markdown.matches("Invoice").count(), 1);
assert!(result.pages[0].provenance.warnings[0].contains("no material coverage"));
}
#[test]
fn adaptive_fallback_fuses_short_novel_ocr_content() {
let native = [native(0, "", true)];
let run = run(vec![routed_page(
1,
vec![span("Status ready 42", 10.0, 0.98)],
Some(0.98),
)]);
let candidates = BTreeMap::from([(1, native_candidate("Status ready\n"))]);
let result =
fuse_ocr_pages_adaptive(&native, &run, 1, &OcrFusionOptions::new(), &candidates)
.unwrap();
assert_eq!(result.pages[0].provenance.source, PageContentSource::Fused);
assert!(result.pages[0].markdown.contains("42"));
assert!(!result.pages[0].provenance.hosted_recommended);
}
#[test]
fn adaptive_fallback_recommends_hosted_for_high_confidence_garbage_ocr() {
let native = [native(0, "", true)];
let garbage = "a@@b%%c&&d==e~~".repeat(12);
let run = run(vec![routed_page(
1,
vec![span(&garbage, 10.0, 0.99)],
Some(0.99),
)]);
let candidates = BTreeMap::from([(1, native_candidate("Invoice total 420\n"))]);
let result =
fuse_ocr_pages_adaptive(&native, &run, 1, &OcrFusionOptions::new(), &candidates)
.unwrap();
assert_eq!(result.pages[0].provenance.source, PageContentSource::Native);
assert!(result.pages[0].provenance.hosted_recommended);
assert!(!result.pages[0].markdown.contains("@@"));
}
#[test]
fn adaptive_fallback_fuses_native_header_with_scanned_body() {
let native = [native(0, "", true)];
let run = run(vec![routed_page(
1,
vec![
span("Quarterly account report", 10.0, 0.96),
span("March revenue 420 units", 30.0, 0.96),
span("April revenue 510 units", 50.0, 0.96),
],
Some(0.96),
)]);
let candidates = BTreeMap::from([(1, native_candidate("# Quarterly account report\n"))]);
let result =
fuse_ocr_pages_adaptive(&native, &run, 1, &OcrFusionOptions::new(), &candidates)
.unwrap();
assert_eq!(result.pages[0].provenance.source, PageContentSource::Fused);
assert_eq!(result.pages[0].markdown.matches("Quarterly").count(), 1);
assert!(result.pages[0].markdown.contains("March revenue 420 units"));
assert!(result.pages[0].markdown.contains("April revenue 510 units"));
assert!(result.pages[0].provenance.warnings[0].contains("complementary"));
}
#[test]
fn native_candidate_scoring_is_multilingual_and_rejects_garbage() {
assert!(assess_native_candidate(
"請求書 合計金額 4200円\n支払済み\n".to_string(),
NativeCandidateOrigin::Extractor,
)
.is_some());
assert!(assess_native_candidate(
"a@@b%%c&&d==e~~".repeat(12),
NativeCandidateOrigin::Extractor,
)
.is_none());
}
#[test]
fn force_mode_appends_only_additional_ocr_blocks() {
let native = [native(0, "Native title\n", false)];
+3 -2
View File
@@ -30,8 +30,9 @@ mod pdfium;
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
pub use contracts::{
ImagePoint, ImageQuad, ModelDownloadPolicy, ModelIdentity, OcrEngine, OcrMode, OcrOptions,
OcrPage, OcrSpan, PageContentSource, PageProvenance, PageRenderer, VisionTimings,
ImagePoint, ImageQuad, LayoutEngine, LayoutOptions, LayoutPage, LayoutRegion, LayoutRegionKind,
ModelDownloadPolicy, ModelIdentity, OcrEngine, OcrMode, OcrOptions, OcrPage, OcrProfile,
OcrSpan, PageContentSource, PageProvenance, PageRenderer, VisionTimings,
};
#[cfg(all(feature = "model-download", not(target_arch = "wasm32")))]
pub use download::{HttpModelDownloadError, HttpModelDownloader, DEFAULT_MODEL_DOWNLOAD_TIMEOUT};
+7 -8
View File
@@ -190,17 +190,16 @@ impl ModelStore {
&self.cache_root
}
/// Effective directory containing one manifest's artifacts.
pub(crate) fn model_root(&self, manifest: &ModelManifest) -> PathBuf {
self.override_root
.clone()
.unwrap_or_else(|| self.manifest_cache_root(manifest))
}
/// Validates and resolves every required artifact.
pub fn resolve(&self, manifest: &ModelManifest) -> Result<ModelPaths, ModelStoreError> {
validate_manifest(manifest)?;
let root = self.model_root(manifest);
let managed_root;
let root = if let Some(root) = self.override_root.as_deref() {
root
} else {
managed_root = self.manifest_cache_root(manifest);
managed_root.as_path()
};
let mut artifacts = BTreeMap::new();
for artifact in manifest.artifacts {
+34 -454
View File
@@ -1,14 +1,10 @@
//! PP-OCRv6 Small implementation backed by OAR and ONNX Runtime.
use std::path::PathBuf;
use std::sync::Arc;
use std::time::Instant;
use image::RgbImage;
use oar_ocr::core::config::onnx::OrtSessionConfig;
use oar_ocr::domain::tasks::TextDetectionConfig;
use oar_ocr::oarocr::{EdgeProcessor, TextCroppingProcessor};
use oar_ocr::predictors::{TextDetectionPredictor, TextRecognitionPredictor};
use oar_ocr::oarocr::{OAROCRBuilder, OAROCR};
use oar_ocr::processors::BoundingBox;
use thiserror::Error;
@@ -52,9 +48,7 @@ pub enum OarOcrError {
page: u32,
},
/// The external ONNX Runtime shared library could not be loaded.
#[error(
"failed to load ONNX Runtime from {path}; install a compatible ONNX Runtime shared library or set ORT_DYLIB_PATH to its path: {source}"
)]
#[error("failed to load ONNX Runtime from {path}: {source}")]
OnnxRuntimeLoad {
/// Requested shared-library path or platform library name.
path: PathBuf,
@@ -73,170 +67,17 @@ pub enum OarOcrError {
Backend(#[from] oar_ocr::core::OCRError),
}
/// Standard detection input cap. PP-OCR detection resizes each page so its
/// longest side fits this before inference; it is the PaddleOCR default and
/// is sufficient for ordinary body text at 150 DPI.
const DETECTION_LIMIT_STANDARD: u32 = 960;
/// Escalated detection input cap for dense fine-print pages. Beyond this the
/// measured recall plateaus while inference cost keeps growing.
const DETECTION_LIMIT_ESCALATED: u32 = 2560;
/// Hard ceiling protecting detection from out-of-memory on giant renders.
const DETECTION_MAXIMUM_SIDE: u32 = 4000;
/// Escalate only for pages dense with small text: at least this many detected
/// regions in the standard pass...
const ESCALATION_MINIMUM_REGIONS: usize = 80;
/// ...whose median height, at detection scale, is below this. Calibrated at
/// `unclip_ratio` 2.0 (the expansion inflates measured heights, so this
/// constant is coupled to [`detection_config`]): dense fine-print pages that
/// gain from escalation measure 12.014.2 px with 144+ regions; the nearest
/// non-gaining page above the region gate (an engineering drawing) measures
/// 15.7 px, and prose/typewriter pages measure 14.5 px+ with too few
/// regions to qualify at all.
const ESCALATION_MAXIMUM_MEDIAN_HEIGHT: f32 = 15.0;
/// One worker's model sessions: a standard-limit detector plus a recognizer,
/// and that worker's own lazily built escalated-limit detector.
/// Staged (detect, crop, recognize as separate calls) rather than OAROCR's
/// combined `predict` so an escalated page replaces only its detection pass —
/// recognition runs exactly once, on the final region set.
struct OcrWorker {
detector: TextDetectionPredictor,
recognizer: TextRecognitionPredictor,
/// Built on this worker's first dense fine-print page. `None` inside the
/// cell records a failed build so it is not retried per page.
escalated: std::sync::OnceLock<Option<TextDetectionPredictor>>,
}
/// CPU PP-OCRv6 Small engine using OAR's detection and recognition components.
/// CPU PP-OCRv6 Small engine using OAR's detection and recognition pipeline.
///
/// Construction accepts only [`ModelPaths`] that have already passed
/// pdf-inspector's manifest size and SHA-256 verification. OAR's independent
/// model auto-download feature is deliberately not enabled.
#[derive(Debug)]
pub struct OarOcrEngine {
workers: Vec<OcrWorker>,
detection_path: PathBuf,
intra_threads: usize,
/// Present only when more than one worker exists; sized to match.
pool: Option<rayon::ThreadPool>,
pipeline: OAROCR,
model: ModelIdentity,
}
impl std::fmt::Debug for OarOcrEngine {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.debug_struct("OarOcrEngine")
.field("workers", &self.workers.len())
.field("parallel", &self.pool.is_some())
.field("model", &self.model)
.finish()
}
}
/// Pages processed concurrently: one OAROCR pipeline (and its ONNX sessions)
/// per worker, because oar-ocr serializes each session behind a mutex.
/// Measured on CPU: workers beyond 3 stop scaling (memory-bandwidth bound)
/// and each worker is fastest with 2 intra-op threads.
fn pipeline_concurrency() -> usize {
let cores = std::thread::available_parallelism()
.map(std::num::NonZeroUsize::get)
.unwrap_or(1);
(cores / 4).clamp(1, 3)
}
fn intra_threads_per_pipeline(concurrency: usize) -> usize {
let cores = std::thread::available_parallelism()
.map(std::num::NonZeroUsize::get)
.unwrap_or(1);
if concurrency > 1 {
2
} else {
cores.min(4)
}
}
/// True when a standard-limit detection pass over a downscaled page shows
/// dense, small text: the page deserves a second pass at the escalated limit.
fn should_escalate_detection(
median_detection_height: f32,
region_count: usize,
downscale: f32,
) -> bool {
downscale < 1.0
&& region_count >= ESCALATION_MINIMUM_REGIONS
&& median_detection_height < ESCALATION_MAXIMUM_MEDIAN_HEIGHT
}
/// Median detected-region height in detection-input pixels: original-image
/// heights multiplied by the downscale detection applied.
fn median_detection_height(heights: &mut [f32], downscale: f32) -> f32 {
if heights.is_empty() {
return f32::MAX;
}
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
let middle = heights.len() / 2;
let median = if heights.len().is_multiple_of(2) {
(heights[middle - 1] + heights[middle]) / 2.0
} else {
heights[middle]
};
median * downscale
}
/// Detection preprocessing config at a given input cap.
///
/// Supplying an explicit config suppresses OAROCR's "general" text-type
/// overrides, so every field the override would have set must be pinned
/// here to match what the combined pipeline ran with before the staged
/// split: score 0.3 and box 0.6 (equal to [`TextDetectionConfig`]'s
/// defaults) and unclip 2.0 (the default is 1.5 — leaving it would
/// silently shrink detection-box expansion and risk clipping edge glyphs).
fn detection_config(detection_limit: u32) -> TextDetectionConfig {
TextDetectionConfig {
limit_side_len: Some(detection_limit),
limit_type: Some(oar_ocr::processors::LimitType::Max),
max_side_len: Some(DETECTION_MAXIMUM_SIDE),
unclip_ratio: 2.0,
..Default::default()
}
}
fn build_detector(
detection: &std::path::Path,
detection_limit: u32,
intra_threads: usize,
) -> Result<TextDetectionPredictor, OarOcrError> {
Ok(TextDetectionPredictor::builder()
.with_config(detection_config(detection_limit))
.with_ort_config(ocr_session_config(intra_threads))
.build(detection)?)
}
fn build_workers(
detection: &std::path::Path,
recognition: &std::path::Path,
dictionary: &std::path::Path,
count: usize,
intra_threads: usize,
) -> Result<Vec<OcrWorker>, OarOcrError> {
let mut workers = Vec::with_capacity(count);
for _ in 0..count {
let detector = build_detector(detection, DETECTION_LIMIT_STANDARD, intra_threads)?;
let recognizer = TextRecognitionPredictor::builder()
.dict_path(dictionary)
.with_ort_config(ocr_session_config(intra_threads))
.build(recognition)?;
workers.push(OcrWorker {
detector,
recognizer,
escalated: std::sync::OnceLock::new(),
});
}
Ok(workers)
}
impl OarOcrEngine {
/// Loads PP-OCRv6 Small from a resolved, verified model set.
pub fn from_models(models: &ModelPaths) -> Result<Self, OarOcrError> {
@@ -245,169 +86,30 @@ impl OarOcrEngine {
let recognition = required_model(models, ModelArtifactKind::TextRecognition)?;
let dictionary = required_model(models, ModelArtifactKind::CharacterDictionary)?;
let concurrency = pipeline_concurrency();
let intra_threads = intra_threads_per_pipeline(concurrency);
let workers = build_workers(
detection,
recognition,
dictionary,
concurrency,
intra_threads,
)?;
let pool = if concurrency > 1 {
rayon::ThreadPoolBuilder::new()
.num_threads(concurrency)
.build()
.ok()
} else {
None
};
let pipeline = OAROCRBuilder::new(detection, recognition, dictionary).build()?;
let model = ModelIdentity::new(models.manifest_id(), models.revision());
Ok(Self {
workers,
detection_path: detection.to_path_buf(),
intra_threads,
pool,
model,
})
}
/// This worker's escalated-limit detector, built on first use.
fn escalated_detector<'w>(&self, worker: &'w OcrWorker) -> Option<&'w TextDetectionPredictor> {
worker
.escalated
.get_or_init(|| {
match build_detector(
&self.detection_path,
DETECTION_LIMIT_ESCALATED,
self.intra_threads,
) {
Ok(detector) => Some(detector),
Err(error) => {
log::warn!(
"escalated OCR detection unavailable, keeping standard pass: {error}"
);
None
}
}
})
.as_ref()
}
/// Detects text regions for one page: a standard-limit pass first, then —
/// for pages the standard limit demonstrably under-resolves — a second
/// pass at the escalated limit whose boxes replace the first. Pages whose
/// render dwarfs even the escalated limit skip the standard pass outright.
fn detect_boxes(
&self,
page: &RenderedPage,
image: &Arc<RgbImage>,
worker: &OcrWorker,
) -> Result<Vec<BoundingBox>, OarOcrError> {
let longest_side = page.width().max(page.height()) as f32;
// A page more than twice the standard limit loses over half its
// resolution before detection even runs; go straight to the escalated
// detector instead of paying a doomed standard pass.
if longest_side > (DETECTION_LIMIT_STANDARD * 2) as f32 {
if let Some(escalated) = self.escalated_detector(worker) {
log::debug!(
"page {}: direct escalated detection (render {longest_side}px)",
page.page(),
);
match detect_with(escalated, image, page.page()) {
Ok(boxes) => return Ok(boxes),
Err(error) => {
// Same degradation as the adaptive branch below: a
// failing escalated pass falls back to standard
// detection instead of failing the page outright.
// Return the standard boxes directly — the adaptive
// trigger would only re-invoke the detector that
// just failed (repeating an OOM on a dense page).
log::warn!(
"page {}: direct escalated detection failed, using standard pass: {error}",
page.page()
);
return detect_with(&worker.detector, image, page.page());
}
}
}
}
let detections = detect_with(&worker.detector, image, page.page())?;
// Dense fine-print pages (broadsheets, pricing sheets) lose most of
// their text when detection downscales them to the standard limit.
// When the standard pass shows many regions of tiny detection-scale
// height, rerun detection at the escalated limit.
let downscale = (DETECTION_LIMIT_STANDARD as f32 / longest_side).min(1.0);
let mut heights: Vec<f32> = detections.iter().map(polygon_height).collect();
let median = median_detection_height(&mut heights, downscale);
log::trace!(
"page {}: standard pass {} regions, median height {:.1}px at detection scale",
page.page(),
detections.len(),
median
);
if should_escalate_detection(median, detections.len(), downscale) {
log::debug!(
"page {}: escalating detection ({} regions, median height {:.1}px at detection scale)",
page.page(),
detections.len(),
median
);
if let Some(escalated) = self.escalated_detector(worker) {
match detect_with(escalated, image, page.page()) {
Ok(escalated_boxes) => return Ok(escalated_boxes),
Err(error) => {
log::warn!(
"page {}: escalated detection failed, keeping standard pass: {error}",
page.page()
);
}
}
}
}
Ok(detections)
Ok(Self { pipeline, model })
}
fn recognize_page(
&self,
page: &RenderedPage,
options: &OcrOptions,
worker: usize,
) -> Result<OcrPage, OarOcrError> {
let started = Instant::now();
let worker = &self.workers[worker % self.workers.len()];
let image = Arc::new(rendered_page_to_rgb(page)?);
let boxes = self.detect_boxes(page, &image, worker)?;
// Reading order, matching what the combined pipeline produced.
let boxes = oar_ocr::processors::sort_quad_boxes(&boxes);
let image = rendered_page_to_rgb(page)?;
let result = self
.pipeline
.predict(vec![image])?
.into_iter()
.next()
.ok_or(OarOcrError::MissingPageResult { page: page.page() })?;
// Same rotation-aware cropping the combined pipeline uses.
let crops =
TextCroppingProcessor::new(true).process((Arc::clone(&image), boxes.clone()))?;
drop(image);
let recognizer = &worker.recognizer;
let mut spans = Vec::with_capacity(boxes.len());
let mut spans = Vec::with_capacity(result.text_regions.len());
let mut invalid_geometry = 0usize;
let mut missing_recognition = 0usize;
for (bounding_box, crop) in boxes.iter().zip(crops) {
let Some(crop) = crop else {
invalid_geometry += 1;
continue;
};
// One crop per call: document line crops often have very
// different widths, and batching pads every crop to the widest
// line. Measured on CPU, batched recognition (even width-sorted)
// is 23× slower than per-crop calls.
let crop = Arc::try_unwrap(crop).unwrap_or_else(|shared| (*shared).clone());
let recognized = recognizer.predict(vec![crop])?;
let (Some(text), Some(confidence)) = (
recognized.texts.into_iter().next(),
recognized.scores.into_iter().next(),
) else {
for region in result.text_regions {
let (Some(text), Some(confidence)) = (region.text, region.confidence) else {
missing_recognition += 1;
continue;
};
@@ -420,25 +122,24 @@ impl OarOcrEngine {
continue;
}
let Some(polygon) = bounding_box_to_quad(bounding_box, page.width(), page.height())
else {
let polygon = region.dt_poly.as_ref().unwrap_or(&region.bounding_box);
let Some(polygon) = bounding_box_to_quad(polygon, page.width(), page.height()) else {
invalid_geometry += 1;
continue;
};
spans.push(OcrSpan {
text,
text: text.to_string(),
polygon,
confidence,
// The combined pipeline's orientation_angle came from the
// text-line-orientation classifier, a model this engine has
// never loaded — it was structurally None before the staged
// split too (the staged/combined A/B was byte-identical).
// Region rotation is still carried by the polygon itself.
orientation_degrees: None,
orientation_degrees: region.orientation_angle,
});
}
let mut warnings = Vec::new();
if !options.languages.is_empty() {
warnings
.push("language hints are not used by the PP-OCRv6 Small OAR backend".to_string());
}
if missing_recognition > 0 {
warnings.push(format!(
"discarded {missing_recognition} regions without usable recognition output"
@@ -458,7 +159,7 @@ impl OarOcrEngine {
let processing_time_ms = u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX);
Ok(OcrPage {
page_number: page.page(),
page: page.page(),
spans,
mean_confidence,
model: self.model.clone(),
@@ -468,48 +169,11 @@ impl OarOcrEngine {
}
}
/// Runs one detector over one page image and returns its region polygons.
fn detect_with(
detector: &TextDetectionPredictor,
image: &Arc<RgbImage>,
page_number: u32,
) -> Result<Vec<BoundingBox>, OarOcrError> {
let mut result = detector.predict(vec![(**image).clone()])?;
if result.detections.is_empty() {
return Err(OarOcrError::MissingPageResult { page: page_number });
}
Ok(result
.detections
.swap_remove(0)
.into_iter()
.map(|detection| detection.bbox)
.collect())
}
/// Vertical extent of a detection polygon in original-image pixels.
fn polygon_height(polygon: &BoundingBox) -> f32 {
let mut min_y = f32::MAX;
let mut max_y = f32::MIN;
for point in &polygon.points {
min_y = min_y.min(point.y);
max_y = max_y.max(point.y);
}
if max_y > min_y {
max_y - min_y
} else {
0.0
}
}
fn ocr_session_config(intra_threads: usize) -> OrtSessionConfig {
OrtSessionConfig::new()
.with_intra_threads(intra_threads.max(1))
.with_inter_threads(1)
.with_parallel_execution(false)
}
fn load_onnx_runtime() -> Result<(), OarOcrError> {
let path = onnx_runtime_library_path();
let path = std::env::var_os(ONNX_RUNTIME_LIBRARY_ENV)
.filter(|path| !path.is_empty())
.map(PathBuf::from)
.unwrap_or_else(default_onnx_runtime_library);
drop(
ort::init_from(&path).map_err(|source| OarOcrError::OnnxRuntimeLoad {
path: path.clone(),
@@ -519,13 +183,6 @@ fn load_onnx_runtime() -> Result<(), OarOcrError> {
Ok(())
}
pub(crate) fn onnx_runtime_library_path() -> PathBuf {
std::env::var_os(ONNX_RUNTIME_LIBRARY_ENV)
.filter(|path| !path.is_empty())
.map(PathBuf::from)
.unwrap_or_else(default_onnx_runtime_library)
}
fn default_onnx_runtime_library() -> PathBuf {
#[cfg(target_os = "windows")]
const NAME: &str = "onnxruntime.dll";
@@ -550,33 +207,10 @@ impl OcrEngine for OarOcrEngine {
) -> Result<Vec<OcrPage>, Self::Error> {
validate_options(options)?;
let Some(pool) = self.pool.as_ref().filter(|_| pages.len() > 1) else {
return pages
.iter()
.map(|page| self.recognize_page(page, options, 0))
.collect();
};
pool.install(|| {
use rayon::prelude::*;
pages
.par_iter()
.map(|page| {
let worker = rayon::current_thread_index().unwrap_or(0);
self.recognize_page(page, options, worker)
})
.collect()
})
}
fn preferred_page_concurrency(&self) -> usize {
// Without a pool, recognition runs sequentially regardless of worker
// count — report that honestly so the pipeline doesn't render
// oversized page batches for parallelism that isn't there.
if self.pool.is_some() {
self.workers.len()
} else {
1
}
pages
.iter()
.map(|page| self.recognize_page(page, options))
.collect()
}
}
@@ -721,17 +355,6 @@ mod tests {
use super::*;
use crate::vision::PageTransform;
#[test]
fn cpu_session_budget_is_bounded_for_small_ocr_models() {
let concurrency = pipeline_concurrency();
let config = ocr_session_config(intra_threads_per_pipeline(concurrency));
assert!((1..=4).contains(&config.intra_threads.unwrap()));
assert_eq!(config.inter_threads, Some(1));
assert_eq!(config.parallel_execution, Some(false));
// Zero requests are clamped so a session always has a thread.
assert_eq!(ocr_session_config(0).intra_threads, Some(1));
}
fn page(format: RenderPixelFormat, stride: usize, pixels: Vec<u8>) -> RenderedPage {
let transform =
PageTransform::from_corners(2, 2, (0.0, 2.0), (2.0, 2.0), (0.0, 0.0)).unwrap();
@@ -843,47 +466,4 @@ mod tests {
)
.is_ok());
}
#[test]
fn escalation_fires_for_dense_fine_print_pages() {
// Measured cases (at unclip 2.0) that gain from escalation: dense
// tiled ad pages (12.314.2px, 158286 regions) and a dense pricing
// sheet (12.0px, 144 regions), all downscaled by the standard limit.
assert!(should_escalate_detection(14.2, 186, 0.55));
assert!(should_escalate_detection(13.1, 286, 0.55));
assert!(should_escalate_detection(12.3, 158, 0.55));
assert!(should_escalate_detection(12.0, 144, 0.55));
}
#[test]
fn escalation_skips_ordinary_pages() {
// Academic prose: too few regions (and tall enough at unclip 2.0).
assert!(!should_escalate_detection(14.5, 47, 0.58));
// Engineering drawing: many regions but tall enough text.
assert!(!should_escalate_detection(15.7, 205, 0.58));
// Typewriter scan: tall text, few regions.
assert!(!should_escalate_detection(17.5, 77, 0.55));
// Page not downscaled at all: escalation cannot add pixels.
assert!(!should_escalate_detection(9.0, 300, 1.0));
}
#[test]
fn median_detection_height_scales_and_handles_empty() {
let mut heights = vec![30.0, 10.0, 20.0];
assert_eq!(median_detection_height(&mut heights, 0.5), 10.0);
// Even counts average the two middle values instead of picking the
// upper one, so borderline pages don't skew away from escalation.
let mut even = vec![10.0, 12.0, 14.0, 30.0];
assert_eq!(median_detection_height(&mut even, 1.0), 13.0);
let mut empty: Vec<f32> = Vec::new();
assert_eq!(median_detection_height(&mut empty, 0.5), f32::MAX);
}
#[test]
fn concurrency_derivations_stay_in_bounds() {
let concurrency = pipeline_concurrency();
assert!((1..=3).contains(&concurrency));
assert!(intra_threads_per_pipeline(2) == 2);
assert!((1..=4).contains(&intra_threads_per_pipeline(1)));
}
}
+3 -215
View File
@@ -2,11 +2,9 @@
use std::path::Path;
use firecrawl_pdfium::{PageChar, Pdfium, PixelFormat, PixelPoint, RenderConfig};
use firecrawl_pdfium::{Pdfium, PixelFormat, PixelPoint, RenderConfig};
use thiserror::Error;
use crate::types::{ItemType, TextItem};
use super::{
PageRenderer, PageTransform, RenderBufferError, RenderOptions, RenderPixelFormat, RenderedPage,
};
@@ -49,15 +47,6 @@ pub enum RenderError {
/// Number of pages in the document.
page_count: usize,
},
/// The PDFium shared library could not be discovered or loaded.
#[error(
"failed to load PDFium; install a compatible PDFium shared library or set PDFIUM_LIB_PATH to its path"
)]
PdfiumLoad {
/// Dynamic loading failure.
#[source]
source: firecrawl_pdfium::Error,
},
/// PDFium loading, document parsing, form setup, or rendering failed.
#[error(transparent)]
Pdfium(#[from] firecrawl_pdfium::Error),
@@ -76,28 +65,18 @@ pub struct PdfiumRenderer {
pdfium: Pdfium,
}
/// Positioned native text recovered from one selected PDF page.
#[derive(Debug)]
pub(crate) struct PdfiumTextPage {
pub(crate) page: u32,
pub(crate) page_width: f32,
pub(crate) page_height: f32,
pub(crate) items: Vec<TextItem>,
}
impl PdfiumRenderer {
/// Loads PDFium using `firecrawl-pdfium`'s documented discovery chain.
pub fn load() -> Result<Self, RenderError> {
Ok(Self {
pdfium: Pdfium::load().map_err(|source| RenderError::PdfiumLoad { source })?,
pdfium: Pdfium::load()?,
})
}
/// Loads PDFium from an explicit native library path.
pub fn load_from_path(path: impl AsRef<Path>) -> Result<Self, RenderError> {
Ok(Self {
pdfium: Pdfium::load_from_path(path)
.map_err(|source| RenderError::PdfiumLoad { source })?,
pdfium: Pdfium::load_from_path(path)?,
})
}
@@ -121,56 +100,6 @@ impl PdfiumRenderer {
self.render_pages_impl(pdf_bytes, pages, password, options)
}
/// Extracts positioned native text from selected 1-indexed pages.
///
/// This is deliberately separate from rendering: callers can probe a
/// suspicious embedded text layer before paying for rasterization and
/// OCR. A page-level text failure is treated as an unavailable recovery
/// candidate so the caller can continue to its normal OCR fallback.
pub(crate) fn extract_text_pages(
&self,
pdf_bytes: &[u8],
pages: &[u32],
password: Option<&str>,
) -> Result<Vec<PdfiumTextPage>, RenderError> {
const MAX_TEXT_CHARS_PER_PAGE: usize = 250_000;
if pages.is_empty() {
return Ok(Vec::new());
}
if pages.contains(&0) {
return Err(RenderError::InvalidPageNumber);
}
let document = self.pdfium.load_document(pdf_bytes.to_vec(), password)?;
let page_count = document.page_count();
if let Some(&page) = pages.iter().find(|&&page| page as usize > page_count) {
return Err(RenderError::PageOutOfBounds { page, page_count });
}
let mut recovered = Vec::with_capacity(pages.len());
for &page_number in pages {
let page = document.page(page_number as usize - 1)?;
let page_size = page.size();
let text = match page.text_with_limit(MAX_TEXT_CHARS_PER_PAGE) {
Ok(text) => text,
Err(error) => {
log::debug!(
"page {page_number}: positioned native text recovery unavailable: {error}"
);
continue;
}
};
recovered.push(PdfiumTextPage {
page: page_number,
page_width: page_size.width,
page_height: page_size.height,
items: text_chars_to_items(text.chars(), page_number),
});
}
Ok(recovered)
}
fn render_pages_impl(
&self,
pdf_bytes: &[u8],
@@ -216,106 +145,6 @@ impl PdfiumRenderer {
}
}
fn text_chars_to_items(chars: &[PageChar], page: u32) -> Vec<TextItem> {
#[derive(Debug, Clone, Copy)]
struct Bounds {
left: f64,
bottom: f64,
right: f64,
top: f64,
}
fn flush(items: &mut Vec<TextItem>, text: &mut String, bounds: &mut Option<Bounds>, page: u32) {
let Some(bounds) = bounds.take() else {
text.clear();
return;
};
if text.is_empty() {
return;
}
let width = (bounds.right - bounds.left) as f32;
let height = (bounds.top - bounds.bottom) as f32;
let x = bounds.left as f32;
let y = bounds.bottom as f32;
if !x.is_finite()
|| !y.is_finite()
|| !width.is_finite()
|| !height.is_finite()
|| width <= 0.0
|| height <= 0.0
{
text.clear();
return;
}
items.push(TextItem {
text: std::mem::take(text),
x,
y,
width,
height,
font: "PDFium native text".to_string(),
font_size: height.max(1.0),
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
});
}
let mut items = Vec::new();
let mut text = String::new();
let mut bounds: Option<Bounds> = None;
for character in chars {
let Some(value) = character.unicode else {
flush(&mut items, &mut text, &mut bounds, page);
continue;
};
if value.is_whitespace() {
flush(&mut items, &mut text, &mut bounds, page);
continue;
}
let rect = character.loose_bounds.normalized();
if !rect.left.is_finite()
|| !rect.bottom.is_finite()
|| !rect.right.is_finite()
|| !rect.top.is_finite()
|| rect.width() <= 0.0
|| rect.height() <= 0.0
{
flush(&mut items, &mut text, &mut bounds, page);
continue;
}
text.push(value);
bounds = Some(match bounds {
Some(bounds) => Bounds {
left: bounds.left.min(rect.left),
bottom: bounds.bottom.min(rect.bottom),
right: bounds.right.max(rect.right),
top: bounds.top.max(rect.top),
},
None => Bounds {
left: rect.left,
bottom: rect.bottom,
right: rect.right,
top: rect.top,
},
});
}
flush(&mut items, &mut text, &mut bounds, page);
items.sort_by(|first, second| {
first
.page
.cmp(&second.page)
.then(second.y.total_cmp(&first.y))
.then(first.x.total_cmp(&second.x))
});
items
}
impl PageRenderer for PdfiumRenderer {
type Error = RenderError;
@@ -408,17 +237,6 @@ fn bgr_to_rgb_in_place(
#[cfg(test)]
mod tests {
use super::*;
use firecrawl_pdfium::{PagePoint, PageRect};
fn page_char(value: char, bounds: PageRect) -> PageChar {
PageChar {
unicode: Some(value),
code: value as u32,
bounds,
loose_bounds: bounds,
origin: PagePoint::new(bounds.left, bounds.bottom),
}
}
#[test]
fn bgr_pixels_are_converted_to_rgb_in_place() {
@@ -445,34 +263,4 @@ mod tests {
Err(RenderBufferError::InvalidBufferLength { .. })
));
}
#[test]
fn invalid_character_geometry_splits_text_runs() {
let chars = [
page_char('A', PageRect::new(0.0, 0.0, 8.0, 10.0)),
page_char('X', PageRect::new(10.0, 0.0, 10.0, 10.0)),
page_char('B', PageRect::new(20.0, 0.0, 28.0, 10.0)),
];
let items = text_chars_to_items(&chars, 1);
assert_eq!(
items
.iter()
.map(|item| item.text.as_str())
.collect::<Vec<_>>(),
["A", "B"]
);
}
#[test]
fn coordinates_that_overflow_f32_are_discarded() {
let left = f64::from(f32::MAX) * 2.0;
let chars = [page_char(
'A',
PageRect::new(left, 0.0, left + 1.0e30, 10.0),
)];
assert!(text_chars_to_items(&chars, 1).is_empty());
}
}
+46 -987
View File
File diff suppressed because it is too large Load Diff
+2 -6
View File
@@ -106,11 +106,7 @@ where
source: Box::new(source),
})?;
let ocr_time_ms = elapsed_ms(ocr_started);
validate_page_order(
"OCR engine",
pages,
recognized.iter().map(|page| page.page_number),
)?;
validate_page_order("OCR engine", pages, recognized.iter().map(|page| page.page))?;
Ok(OcrRun {
pages: rendered
@@ -272,7 +268,7 @@ mod tests {
Ok(pages
.iter()
.map(|page| OcrPage {
page_number: page.page(),
page: page.page(),
spans: vec![OcrSpan {
text: format!("page {}", page.page()),
polygon: ImageQuad::new([
+3 -1
View File
@@ -5,7 +5,9 @@ use pdf_inspector::vision::{PdfiumRenderer, RenderError, RenderOptions, RenderPi
fn load_renderer() -> Option<PdfiumRenderer> {
match PdfiumRenderer::load() {
Ok(renderer) => Some(renderer),
Err(RenderError::PdfiumLoad { .. }) => {
Err(RenderError::Pdfium(firecrawl_pdfium::Error::Load(
firecrawl_pdfium::LoadError::LibraryNotFound { .. },
))) => {
eprintln!("skipping PDFium runtime test because no native library is installed");
None
}
+7 -62
View File
@@ -2,8 +2,7 @@
#[cfg(feature = "ocr")]
use pdf_inspector::vision::{
process_pdf_with_ocr_mem, ModelDownloadPolicy, OcrPdfOptions, OcrPipelineError,
PageContentSource,
process_pdf_with_ocr_mem, ModelDownloadPolicy, OcrPdfOptions, PageContentSource,
};
use pdf_inspector::vision::{
ModelStore, OarOcrEngine, OcrEngine, OcrMode, OcrOptions, PageTransform, RenderPixelFormat,
@@ -20,7 +19,9 @@ const EXPECTED_TEXT_ENV: &str = "PDF_INSPECTOR_OCR_TEST_EXPECTED";
fn load_renderer() -> Option<PdfiumRenderer> {
match PdfiumRenderer::load() {
Ok(renderer) => Some(renderer),
Err(RenderError::PdfiumLoad { .. }) => {
Err(RenderError::Pdfium(firecrawl_pdfium::Error::Load(
firecrawl_pdfium::LoadError::LibraryNotFound { .. },
))) => {
eprintln!("skipping OCR runtime test because no native PDFium library is installed");
None
}
@@ -120,71 +121,15 @@ fn complete_ocr_pipeline_routes_and_assembles_a_scanned_fixture() {
.minimum_confidence(0.3)
.model_directory(model_directory)
.model_downloads(ModelDownloadPolicy::Offline);
let options = OcrPdfOptions::new().ocr(ocr);
let result = process_pdf_with_ocr_mem(&bytes, options.clone()).unwrap();
let repeated = process_pdf_with_ocr_mem(&bytes, options).unwrap();
let result = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().ocr(ocr)).unwrap();
assert_eq!(result.pages_routed_to_ocr, vec![1]);
assert!(!result.markdown.trim().is_empty());
assert!(result
.markdown
.contains("Order Date Item Code Description Status Unit Cost\n\n03/14/2024"));
assert!(result.markdown.contains("$482,110.40\n\n05/02/2024"));
assert_eq!(result.pages[0].provenance.source, PageContentSource::Fused);
assert!(result.pages[0]
.provenance
.warnings
.iter()
.any(|warning| warning.contains("complementary OCR")));
assert_eq!(result.pages[0].provenance.source, PageContentSource::Ocr);
assert_eq!(
result.pages[0].provenance.ocr_model.as_ref().unwrap().name,
PP_OCR_V6_SMALL.id
);
assert_eq!(repeated.markdown, result.markdown);
}
#[cfg(all(feature = "ocr", feature = "render-pdfium"))]
#[test]
fn auto_recovers_credible_native_text_before_loading_ocr_models() {
let Some(_renderer) = load_renderer() else {
return;
};
let bytes = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
let ocr = OcrOptions::new()
.mode(OcrMode::Auto)
.model_directory("/models/must-not-be-read")
.model_downloads(ModelDownloadPolicy::Offline);
let result = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().ocr(ocr)).unwrap();
assert_eq!(result.pages_recommended_for_ocr, vec![1]);
assert!(result.pages_routed_to_ocr.is_empty());
assert!(result.markdown.contains("羽田空港新飛行経路"));
assert!(result.markdown.contains("|4月30日|有|81.0|"));
assert!(result.markdown.contains("※1 最大騒音レベル"));
assert!(result.pages_with_tables.contains(&1));
assert_eq!(result.pages[0].provenance.source, PageContentSource::Native);
assert!(result.pages[0].provenance.ocr_model.is_none());
}
#[cfg(all(feature = "ocr", feature = "render-pdfium"))]
#[test]
fn auto_rejects_garbled_native_recovery_and_continues_to_ocr() {
let Some(_renderer) = load_renderer() else {
return;
};
let bytes = std::fs::read("tests/fixtures/shifted_cipher_tounicode.pdf").unwrap();
let ocr = OcrOptions::new()
.mode(OcrMode::Auto)
.model_directory("/models/must-not-be-read")
.model_downloads(ModelDownloadPolicy::Offline);
let error = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().ocr(ocr)).unwrap_err();
assert!(matches!(
error,
OcrPipelineError::ModelAcquire(_) | OcrPipelineError::ModelStore(_)
));
}
fn recognize(
@@ -206,7 +151,7 @@ fn recognize(
fn assert_usable_result(results: &[pdf_inspector::vision::OcrPage]) {
assert_eq!(results.len(), 1);
assert_eq!(results[0].page_number, 1);
assert_eq!(results[0].page, 1);
assert_eq!(results[0].model.name, PP_OCR_V6_SMALL.id);
assert_eq!(results[0].model.revision, PP_OCR_V6_SMALL.revision);
assert!(!results[0].spans.is_empty());
-44
View File
@@ -77,50 +77,6 @@ class TestProcessPdfBytes:
assert result.markdown is not None
# ---------------------------------------------------------------------------
# process_pdf_with_ocr / process_pdf_with_ocr_bytes
# ---------------------------------------------------------------------------
class TestProcessPdfWithOcr:
def test_off_mode_has_full_provenance_without_external_runtimes(self):
result = pdf_inspector.process_pdf_with_ocr(
fixture_path("thermo-freon12.pdf"), mode="off"
)
assert result.page_count == 3
assert len(result.pages) == 3
assert result.pages_routed_to_ocr == []
assert all(page.provenance.source == "native" for page in result.pages)
assert all(page.provenance.ocr_model is None for page in result.pages)
assert result.markdown
assert "OcrPdfResult" in repr(result)
def test_auto_mode_skips_external_runtimes_for_clean_text(self):
result = pdf_inspector.process_pdf_with_ocr_bytes(
fixture_bytes("thermo-freon12.pdf")
)
assert result.pages_routed_to_ocr == []
assert result.render_time_ms == 0
assert result.ocr_time_ms == 0
def test_selected_pages_are_one_indexed(self):
result = pdf_inspector.process_pdf_with_ocr(
fixture_path("thermo-freon12.pdf"), mode="off", page_numbers=[2]
)
assert [page.page_number for page in result.pages] == [2]
def test_rejects_invalid_options(self):
with pytest.raises(ValueError, match="mode must be"):
pdf_inspector.process_pdf_with_ocr(
fixture_path("thermo-freon12.pdf"), mode="sometimes"
)
with pytest.raises(ValueError, match="page 0"):
pdf_inspector.process_pdf_with_ocr(
fixture_path("thermo-freon12.pdf"),
mode="off",
page_numbers=[0],
)
# ---------------------------------------------------------------------------
# detect_pdf / detect_pdf_bytes
# ---------------------------------------------------------------------------
+2 -2
View File
@@ -724,7 +724,7 @@ checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
[[package]]
name = "pdf-inspector"
version = "1.15.0"
version = "1.14.2"
dependencies = [
"env_logger",
"include_dir",
@@ -740,7 +740,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-wasm"
version = "1.15.0"
version = "1.14.2"
dependencies = [
"console_error_panic_hook",
"js-sys",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-wasm"
version = "1.15.0"
version = "1.14.2"
edition = "2021"
authors = ["Firecrawl Team"]
description = "Browser WebAssembly bindings for pdf-inspector"