Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e3f5429638 | ||
|
|
f6cbe979f6 | ||
|
|
3f43745313 | ||
|
|
a012cb65a6 | ||
|
|
6567e1ab2d | ||
|
|
3af409d27f |
@@ -1,2 +0,0 @@
|
||||
*.pdf binary
|
||||
tests/snapshots/*.md text eol=lf
|
||||
+1
-165
@@ -9,9 +9,6 @@ on:
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
test:
|
||||
name: Test
|
||||
@@ -71,12 +68,9 @@ jobs:
|
||||
with:
|
||||
key: clippy
|
||||
|
||||
- name: Run default clippy
|
||||
- name: Run clippy
|
||||
run: cargo clippy -- -D warnings
|
||||
|
||||
- name: Run OCR clippy
|
||||
run: cargo clippy --features ocr -- -D warnings
|
||||
|
||||
build:
|
||||
name: Build
|
||||
runs-on: ${{ matrix.os }}
|
||||
@@ -99,164 +93,6 @@ jobs:
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
ocr:
|
||||
name: OCR (${{ matrix.os }})
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: ocr-${{ matrix.os }}
|
||||
|
||||
- name: Test optional OCR feature
|
||||
run: cargo test --features ocr
|
||||
|
||||
ocr-runtime:
|
||||
name: OCR runtime smoke
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: ocr-runtime
|
||||
workspaces: |
|
||||
. -> target
|
||||
napi -> target
|
||||
|
||||
- name: Install PDFium
|
||||
shell: bash
|
||||
run: |
|
||||
archive="$RUNNER_TEMP/firecrawl-pdfium-linux-x64.tgz"
|
||||
directory="$RUNNER_TEMP/firecrawl-pdfium"
|
||||
curl --fail --location --silent --show-error \
|
||||
https://github.com/firecrawl/pdfium-rs/releases/download/native-v7988/firecrawl-pdfium-linux-x64.tgz \
|
||||
--output "$archive"
|
||||
echo "6248189e07bbc33cdeb31976c539a88614307c8a19f3276dbd018efbe5b4a2a2 $archive" | sha256sum --check
|
||||
mkdir -p "$directory"
|
||||
tar -xzf "$archive" -C "$directory"
|
||||
pdfium_path="$(find "$directory" -type f -name 'libpdfium.so' -print -quit)"
|
||||
test -n "$pdfium_path"
|
||||
echo "PDFIUM_LIB_PATH=$pdfium_path" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Install ONNX Runtime
|
||||
shell: bash
|
||||
run: |
|
||||
archive="$RUNNER_TEMP/onnxruntime-linux-x64-1.27.0.tgz"
|
||||
directory="$RUNNER_TEMP/onnxruntime"
|
||||
curl --fail --location --silent --show-error \
|
||||
https://github.com/microsoft/onnxruntime/releases/download/v1.27.0/onnxruntime-linux-x64-1.27.0.tgz \
|
||||
--output "$archive"
|
||||
echo "547e40a48f1fe73e3f812d7c88a948612c23f896b91e4e2ee1e232d7b468246f $archive" | sha256sum --check
|
||||
mkdir -p "$directory"
|
||||
tar -xzf "$archive" -C "$directory"
|
||||
ort_path="$(find "$directory" -type f -name 'libonnxruntime.so*' -print -quit)"
|
||||
test -n "$ort_path"
|
||||
echo "ORT_DYLIB_PATH=$ort_path" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build OCR CLI
|
||||
run: cargo build --features ocr --bin pdf2md
|
||||
|
||||
- name: Test PDFium runtime
|
||||
run: cargo test --features ocr --test local_render_tests
|
||||
|
||||
- name: Run OCR CLI
|
||||
shell: bash
|
||||
run: |
|
||||
target/debug/pdf2md \
|
||||
tests/fixtures/scan_with_native_header_text.pdf \
|
||||
--ocr auto \
|
||||
--json > "$RUNNER_TEMP/ocr-result.json"
|
||||
|
||||
- name: Validate OCR JSON contract
|
||||
shell: bash
|
||||
run: |
|
||||
python3 - <<'PY'
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
result = json.loads(
|
||||
(Path(os.environ["RUNNER_TEMP"]) / "ocr-result.json").read_text()
|
||||
)
|
||||
assert result["schema_version"] == 1
|
||||
assert result["pages_routed_to_ocr"] == [1]
|
||||
assert result["pages_recommending_hosted"] == []
|
||||
assert result["pages"][0]["source"] in {"ocr", "fused"}
|
||||
assert result["pages"][0]["markdown"].strip()
|
||||
assert "layout_ms" not in result["pages"][0]["timings"]
|
||||
PY
|
||||
|
||||
- name: Build Node binding
|
||||
working-directory: napi
|
||||
run: |
|
||||
bun install --frozen-lockfile
|
||||
bunx napi build --platform --release
|
||||
|
||||
- name: Run Node OCR binding
|
||||
shell: bash
|
||||
run: |
|
||||
node --input-type=module - <<'JS'
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { processPdfWithOcr } from './napi/index.js'
|
||||
|
||||
const pdf = readFileSync('tests/fixtures/scan_with_native_header_text.pdf')
|
||||
const result = await processPdfWithOcr(pdf, { offline: true })
|
||||
if (JSON.stringify(result.pagesRoutedToOcr) !== '[1]') throw new Error('unexpected OCR route')
|
||||
if (result.pagesRecommendingHosted.length !== 0) throw new Error('unexpected hosted recommendation')
|
||||
if (!['Ocr', 'Fused'].includes(result.pages[0].provenance.source)) throw new Error('unexpected source')
|
||||
if (!result.pages[0].markdown.trim()) throw new Error('empty OCR markdown')
|
||||
JS
|
||||
|
||||
- name: Build and install Python binding
|
||||
shell: bash
|
||||
run: |
|
||||
python -m pip install 'maturin>=1,<2'
|
||||
maturin build --release --out "$RUNNER_TEMP/python-wheels"
|
||||
python -m pip install "$RUNNER_TEMP"/python-wheels/*.whl
|
||||
|
||||
- name: Run Python OCR binding
|
||||
shell: bash
|
||||
run: |
|
||||
python - <<'PY'
|
||||
import pdf_inspector
|
||||
|
||||
result = pdf_inspector.process_pdf_with_ocr(
|
||||
"tests/fixtures/scan_with_native_header_text.pdf",
|
||||
offline=True,
|
||||
)
|
||||
assert result.pages_routed_to_ocr == [1]
|
||||
assert result.pages_recommending_hosted == []
|
||||
assert result.pages[0].provenance.source in {"ocr", "fused"}
|
||||
assert result.pages[0].markdown.strip()
|
||||
PY
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
+1
-14
@@ -57,14 +57,6 @@ firecrawl-pdfium = { version = "0.1.0", optional = true }
|
||||
dirs = { version = "6.0", optional = true }
|
||||
fs2 = { version = "0.4", optional = true }
|
||||
sha2 = { version = "0.11", optional = true }
|
||||
# Optional CPU OCR backend. Models and ONNX Runtime stay external: the latter
|
||||
# is loaded dynamically from ORT_DYLIB_PATH or the platform library search path.
|
||||
image = { version = "0.25.6", default-features = false, optional = true }
|
||||
oar-ocr = { version = "0.9.1", default-features = false, features = ["simd"], optional = true }
|
||||
ort = { version = "=2.0.0-rc.13", default-features = false, features = ["load-dynamic"], optional = true }
|
||||
# HTTPS-only streaming downloader for pinned model artifacts. Kept separate
|
||||
# from model-cache so offline and package-managed deployments avoid HTTP/TLS.
|
||||
ureq = { version = "3.4", default-features = false, features = ["rustls", "platform-verifier"], optional = true }
|
||||
|
||||
[target.'cfg(all(windows, not(target_arch = "wasm32")))'.dependencies]
|
||||
windows-sys = { version = "0.61", features = ["Win32_Storage_FileSystem"], optional = true }
|
||||
@@ -80,15 +72,10 @@ tempfile = "3.3"
|
||||
|
||||
[features]
|
||||
default = []
|
||||
python = ["pyo3", "ocr"]
|
||||
python = ["pyo3"]
|
||||
vision = []
|
||||
model-cache = ["vision", "dep:dirs", "dep:fs2", "dep:sha2", "dep:windows-sys"]
|
||||
model-download = ["model-cache", "dep:ureq"]
|
||||
ocr-oar = ["model-cache", "dep:image", "dep:oar-ocr", "dep:ort"]
|
||||
render-pdfium = ["vision", "dep:firecrawl-pdfium"]
|
||||
# Complete native OCR path. This remains opt-in so default library,
|
||||
# renderer-only, and browser consumers do not inherit inference or HTTP/TLS.
|
||||
ocr = ["render-pdfium", "ocr-oar", "model-download"]
|
||||
|
||||
[[bin]]
|
||||
name = "pdf2md"
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. By default it detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown without OCR. Native Rust and CLI consumers can opt into selective OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -18,10 +18,9 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **CID font support** — ToUnicode CMap decoding for Type0/Identity-H fonts, UTF-16BE, UTF-8, and Latin-1 encodings.
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Selective OCR** — Rust, CLI, Python, and Node can render only pages that need OCR, run PP-OCRv6 Small locally, and preserve per-page provenance and hosted-fallback recommendations.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight by default** — The default Rust and browser builds remain pure extraction. Native Python and Node packages include the OCR integration, but PDFium, ONNX Runtime, and model files remain external and are touched only when a page is routed to OCR.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
@@ -58,10 +57,6 @@ import pdf_inspector
|
||||
result = pdf_inspector.process_pdf("document.pdf")
|
||||
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
||||
print(result.markdown) # Markdown string or None
|
||||
|
||||
# Selective OCR; clean text PDFs do not load the external OCR runtime.
|
||||
ocr = pdf_inspector.process_pdf_with_ocr("document.pdf")
|
||||
print(ocr.pages_routed_to_ocr)
|
||||
```
|
||||
|
||||
> Full API reference: [docs/python.md](docs/python.md)
|
||||
@@ -74,15 +69,11 @@ npm install @firecrawl/pdf-inspector
|
||||
|
||||
```javascript
|
||||
import { readFileSync } from 'fs';
|
||||
import { processPdf, processPdfWithOcr } from '@firecrawl/pdf-inspector';
|
||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector';
|
||||
|
||||
const pdf = readFileSync('document.pdf');
|
||||
const result = processPdf(pdf);
|
||||
const result = processPdf(readFileSync('document.pdf'));
|
||||
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
||||
console.log(result.markdown); // Markdown string or null
|
||||
|
||||
const ocr = await processPdfWithOcr(pdf); // selective OCR, off the event loop
|
||||
console.log(ocr.pagesRoutedToOcr);
|
||||
```
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
@@ -169,22 +160,6 @@ detect-pdf document.pdf --json
|
||||
detect-pdf document.pdf --analyze --json
|
||||
```
|
||||
|
||||
Rust and CLI consumers opt into OCR at build time:
|
||||
|
||||
```bash
|
||||
cargo install pdf-inspector --features ocr --bin pdf2md
|
||||
PDFIUM_LIB_PATH=/path/to/libpdfium ORT_DYLIB_PATH=/path/to/libonnxruntime \
|
||||
pdf2md scan.pdf --ocr auto --json
|
||||
```
|
||||
|
||||
The OCR JSON envelope is versioned and reports routed pages, per-page source
|
||||
and confidence, warnings, and pages recommended for the hosted document
|
||||
pipeline. Native Python and Node packages expose the same pipeline without a
|
||||
source-build feature. All native entry points still require separately
|
||||
installed PDFium and ONNX Runtime libraries only when OCR is routed. See the
|
||||
[Rust API guide](docs/rust-api.md#complete-ocr-api) for model cache and offline
|
||||
configuration.
|
||||
|
||||
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
||||
|
||||
## Architecture
|
||||
|
||||
+2
-63
@@ -1,6 +1,6 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast PDF classification, text extraction, and selective OCR. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts clean native and OCR results to Markdown. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -10,8 +10,7 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Selective OCR** — `auto` routes only pages rejected by native extraction; `force` OCRs every selected page; `off` keeps the result/provenance contract without external runtime work.
|
||||
- **External artifacts** — the wheel embeds no OCR models, PDFium, or ONNX Runtime; clean `auto` requests never load or download them.
|
||||
- **Lightweight** — native Rust core, no ML models, no external services; ships type stubs.
|
||||
|
||||
## Benchmark
|
||||
|
||||
@@ -40,12 +39,6 @@ pip install maturin
|
||||
maturin develop --release
|
||||
```
|
||||
|
||||
OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||
platform library search path. The pinned OCR model set is downloaded and
|
||||
checksum-verified on the first routed page; use `offline=True` with a warm
|
||||
cache or `model_directory` to prohibit network access.
|
||||
|
||||
## Usage
|
||||
|
||||
```python
|
||||
@@ -88,19 +81,6 @@ for page in result.pages:
|
||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
|
||||
# One-call selective OCR. This releases the GIL while processing.
|
||||
ocr = pdf_inspector.process_pdf_with_ocr("document.pdf")
|
||||
for page in ocr.pages:
|
||||
print(page.page_number, page.provenance.source)
|
||||
|
||||
# Restrict OCR processing to 1-indexed PDF pages and prohibit downloads.
|
||||
ocr = pdf_inspector.process_pdf_with_ocr(
|
||||
"document.pdf",
|
||||
page_numbers=[1, 3],
|
||||
model_directory="/opt/models/pp-ocrv6-small",
|
||||
offline=True,
|
||||
)
|
||||
|
||||
# Structure-tree elements from tagged PDFs (empty list when untagged).
|
||||
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
|
||||
# against extract_text_with_positions — e.g. to recover real heading levels:
|
||||
@@ -119,8 +99,6 @@ headings = [
|
||||
|---|---|
|
||||
| `process_pdf(path, pages=None)` | Full processing (detect + extract + markdown) |
|
||||
| `process_pdf_bytes(data, pages=None)` | Full processing from bytes |
|
||||
| `process_pdf_with_ocr(path, **options)` | Native extraction + selective OCR with provenance |
|
||||
| `process_pdf_with_ocr_bytes(data, **options)` | Native extraction + selective OCR from bytes |
|
||||
| `detect_pdf(path)` | Fast detection only (returns PdfResult) |
|
||||
| `detect_pdf_bytes(data)` | Fast detection from bytes |
|
||||
| `classify_pdf(path)` | Lightweight classification (returns PdfClassification) |
|
||||
@@ -159,45 +137,6 @@ class PageOcrReasons: # per-page OCR diagnostics
|
||||
page: int # 1-indexed
|
||||
reasons: list[str] # machine-readable reason identifiers
|
||||
|
||||
class OcrModelIdentity:
|
||||
name: str # model family/name
|
||||
revision: str # immutable artifact-set revision
|
||||
|
||||
class OcrTimings: # per-page processing stages
|
||||
render_ms: int
|
||||
ocr_ms: int
|
||||
assembly_ms: int
|
||||
|
||||
class OcrPageProvenance:
|
||||
page_number: int # 1-indexed
|
||||
source: Literal["native", "ocr", "fused"]
|
||||
ocr_model: OcrModelIdentity | None
|
||||
render_dpi: float | None
|
||||
ocr_confidence: float | None
|
||||
timings: OcrTimings
|
||||
warnings: list[str]
|
||||
hosted_recommended: bool
|
||||
|
||||
class OcrPageResult:
|
||||
page_number: int # 1-indexed
|
||||
markdown: str
|
||||
provenance: OcrPageProvenance
|
||||
|
||||
class OcrPdfResult: # process_pdf_with_ocr / bytes
|
||||
markdown: str
|
||||
pages: list[OcrPageResult]
|
||||
page_count: int
|
||||
pages_recommended_for_ocr: list[int]
|
||||
pages_routed_to_ocr: list[int]
|
||||
pages_recommending_hosted: list[int]
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
is_complex: bool
|
||||
processing_time_ms: int
|
||||
render_time_ms: int
|
||||
ocr_time_ms: int
|
||||
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
|
||||
+13
-250
@@ -1,6 +1,6 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast PDF classification and text extraction. The default build detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown without OCR. It is pure Rust, has no ML models or external services, and uses [lopdf](https://crates.io/crates/lopdf) for PDF parsing. Native Rust and CLI consumers can opt into selective OCR. Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector/).
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. The default build is pure Rust, has no ML models or external services, and uses [lopdf](https://crates.io/crates/lopdf) for PDF parsing. Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector/).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -123,10 +123,10 @@ The native-only `vision` feature exposes the stable seam used by OCR
|
||||
integrations without selecting or embedding an inference runtime. The
|
||||
separate `model-cache` feature adds pinned artifact management:
|
||||
|
||||
- `PageRenderer` and `OcrEngine` traits;
|
||||
- `PageRenderer`, `OcrEngine`, and `LayoutEngine` traits;
|
||||
- renderer-neutral owned page buffers and affine pixel↔PDF transforms;
|
||||
- `OcrOptions` and opt-in `Off`/`Auto`/`Force` routing modes;
|
||||
- positioned OCR results and per-page provenance types; and
|
||||
- positioned OCR/layout results and per-page provenance types; and
|
||||
- a versioned PP-OCRv6 Small manifest with checksum-verified, locked, atomic
|
||||
model-cache installation and explicit offline-directory overrides.
|
||||
|
||||
@@ -135,14 +135,13 @@ separate `model-cache` feature adds pinned artifact management:
|
||||
pdf-inspector = { version = "1", features = ["vision", "model-cache"] }
|
||||
```
|
||||
|
||||
The OCR contracts preserve existing behavior by default: OCR is `Off` and
|
||||
model resolution is never reached. `ModelStore` itself does not access the
|
||||
network. The optional `model-download` feature provides an
|
||||
HTTPS downloader that streams pinned artifacts into the checksum-verified
|
||||
cache only after routing has selected OCR work. Offline consumers set an
|
||||
explicit model directory and `ModelDownloadPolicy::Offline`. Renderer-only
|
||||
consumers do not enable `model-cache` or `model-download` and therefore do not
|
||||
compile their filesystem, hashing, or HTTP dependencies.
|
||||
The OCR contracts preserve existing behavior by default: OCR is `Off`, learned
|
||||
layout is disabled, and model resolution is never reached. `ModelStore` itself
|
||||
does not access the network; a runtime integration can fetch a manifest's
|
||||
canonical URL only when allowed and pass the stream to `ModelStore::install`.
|
||||
Offline consumers set an explicit model directory and `ModelDownloadPolicy::Offline`.
|
||||
Renderer-only consumers do not enable `model-cache` and therefore do not compile
|
||||
its filesystem, locking, or hashing dependencies.
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
@@ -171,10 +170,9 @@ and `PdfiumRenderer` implements the renderer-neutral `PageRenderer` trait.
|
||||
pdf-inspector = { version = "1", features = ["render-pdfium"] }
|
||||
```
|
||||
|
||||
PDFium is loaded at runtime and is not bundled into the crate. Set
|
||||
`PDFIUM_LIB_PATH` to the platform shared library, place that library next to
|
||||
the executable, or use another discovery route supported by
|
||||
`firecrawl-pdfium`. A load failure reports this prerequisite directly.
|
||||
PDFium is loaded at runtime. Set `PDFIUM_LIB_PATH`, place its shared library
|
||||
next to the executable, or use another discovery route supported by
|
||||
`firecrawl-pdfium`.
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{PdfiumRenderer, RenderOptions};
|
||||
@@ -199,241 +197,6 @@ for page in pages {
|
||||
Browser WASM remains on the default text-only path and does not expose native
|
||||
PDFium rendering.
|
||||
|
||||
### Optional OCR engine
|
||||
|
||||
The native-only `ocr-oar` feature adds a CPU PP-OCRv6 Small implementation of
|
||||
`OcrEngine` backed by OAR and ONNX Runtime. It implies `model-cache`, but does
|
||||
not enable model auto-download, ONNX Runtime download, or PDF rendering. Model
|
||||
files remain external, must match the pinned manifest, and are opened only
|
||||
after `ModelStore` verifies their exact size and SHA-256 digest. Install an
|
||||
ONNX Runtime shared library separately and set `ORT_DYLIB_PATH` to its full
|
||||
path when it is not available through the platform library search path. The
|
||||
runtime is resolved only when an OCR engine is first constructed; clean
|
||||
`Auto` requests do not require it. The feature currently requires Rust 1.95
|
||||
or newer, matching OAR 0.9.1's MSRV.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = ["ocr-oar", "render-pdfium"] }
|
||||
```
|
||||
|
||||
Direct engine invocation is intentionally separate from extraction routing and
|
||||
native/OCR fusion:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
ModelDownloadPolicy, ModelStore, OarOcrEngine, OcrEngine, OcrMode,
|
||||
OcrOptions, PdfiumRenderer, RenderOptions, PP_OCR_V6_SMALL,
|
||||
};
|
||||
|
||||
let options = OcrOptions::new()
|
||||
.mode(OcrMode::Force)
|
||||
.minimum_confidence(0.45)
|
||||
.model_directory("/opt/firecrawl/models/pp-ocrv6-small")
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
let models = ModelStore::from_options(&options)?.resolve(&PP_OCR_V6_SMALL)?;
|
||||
let engine = OarOcrEngine::from_models(&models)?;
|
||||
|
||||
let renderer = PdfiumRenderer::load()?;
|
||||
let bytes = std::fs::read("scan.pdf")?;
|
||||
let pages = renderer.render_pages(&bytes, &[1], None, &RenderOptions::new())?;
|
||||
let ocr_pages = engine.recognize(&pages, &options)?;
|
||||
|
||||
for span in &ocr_pages[0].spans {
|
||||
println!("{:.3}: {}", span.confidence, span.text);
|
||||
}
|
||||
```
|
||||
|
||||
The engine accepts renderer-neutral RGB, RGBA, and grayscale pages, preserves
|
||||
OAR's positioned quadrilaterals in bitmap coordinates, filters spans using
|
||||
`minimum_confidence`, and records the pinned model revision in every `OcrPage`.
|
||||
`OcrMode::Off` is rejected at the engine boundary so default options cannot run
|
||||
inference accidentally.
|
||||
|
||||
### Selective routing and lazy model acquisition
|
||||
|
||||
`route_ocr_pages` applies the existing detector/text-quality recommendations to
|
||||
the configured mode. `Auto` processes only recommended pages, `Force` processes
|
||||
all pages (or an explicit page selection), and `Off` always returns an empty
|
||||
route. `run_ocr_pages` renders only that route, checks that both dependencies
|
||||
preserve its order, and retains each bitmap's PDF transform for fusion.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = [
|
||||
"render-pdfium",
|
||||
"ocr-oar",
|
||||
"model-download",
|
||||
] }
|
||||
```
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
route_ocr_pages, run_ocr_pages, HttpModelDownloader, ModelStore,
|
||||
OarOcrEngine, OcrMode, OcrOptions, PdfiumRenderer, RenderOptions,
|
||||
PP_OCR_V6_SMALL,
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("scan.pdf")?;
|
||||
let extraction = pdf_inspector::extract_pages_markdown_mem(&bytes, None)?;
|
||||
let options = OcrOptions::new().mode(OcrMode::Auto);
|
||||
let routed = route_ocr_pages(
|
||||
options.mode,
|
||||
extraction.pages.len() as u32,
|
||||
&extraction.pages_needing_ocr,
|
||||
None,
|
||||
)?;
|
||||
|
||||
if !routed.is_empty() {
|
||||
// No HTTP request or model initialization occurs before this point.
|
||||
let store = ModelStore::from_options(&options)?;
|
||||
let models = store.resolve_or_download(
|
||||
&PP_OCR_V6_SMALL,
|
||||
options.model_downloads,
|
||||
&HttpModelDownloader::default(),
|
||||
)?;
|
||||
let run = run_ocr_pages(
|
||||
&PdfiumRenderer::load()?,
|
||||
&OarOcrEngine::from_models(&models)?,
|
||||
&bytes,
|
||||
&routed,
|
||||
None,
|
||||
&RenderOptions::new(),
|
||||
&options,
|
||||
)?;
|
||||
println!("OCR processed {} pages", run.pages.len());
|
||||
}
|
||||
```
|
||||
|
||||
The downloader accepts HTTPS only, checks a declared content length, caps the
|
||||
response stream to the pinned size plus one byte, and delegates final size and
|
||||
SHA-256 verification to `ModelStore`. The store serializes installation across
|
||||
processes and publishes completed artifacts atomically. Warm caches make no
|
||||
network calls; offline mode and explicit model directories never download.
|
||||
|
||||
### OCR Markdown assembly and native fusion
|
||||
|
||||
`fuse_ocr_pages` maps OCR polygons back into PDF coordinates and sends the
|
||||
result through pdf-inspector's existing deterministic reading-order, table,
|
||||
and Markdown pipeline. Pages whose native extraction was rejected use OCR
|
||||
output. When `Force` runs on a clean native page, normalized duplicate OCR
|
||||
blocks are removed and only additional image-backed text is retained.
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{fuse_ocr_pages, OcrFusionOptions};
|
||||
|
||||
let fused = fuse_ocr_pages(
|
||||
&extraction.pages,
|
||||
&run,
|
||||
extraction.pages.len() as u32,
|
||||
&OcrFusionOptions::new().render_dpi(150.0),
|
||||
)?;
|
||||
|
||||
for page in &fused.pages {
|
||||
println!("{}", page.markdown);
|
||||
if page.provenance.hosted_recommended {
|
||||
eprintln!(
|
||||
"page {} needs the hosted document pipeline",
|
||||
page.page_number,
|
||||
);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Each page carries `Native`, `Ocr`, or `Fused` provenance, the exact OCR model
|
||||
revision, accepted-page confidence, local stage timings, and non-fatal
|
||||
warnings. A page that required OCR recommends the hosted pipeline when local
|
||||
OCR is missing, empty, or below the configurable page-confidence threshold.
|
||||
This keeps the lightweight path explicit about cases it cannot finish well.
|
||||
|
||||
### Complete OCR API
|
||||
|
||||
The `ocr` convenience feature enables the renderer, OCR engine, verified
|
||||
model acquisition, routing, and fusion layers together. It is the intended
|
||||
downstream application integration boundary; lower-level features remain
|
||||
available for consumers that bring their own renderer, model package manager,
|
||||
or engine.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = ["ocr"] }
|
||||
```
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{process_pdf_with_ocr, OcrPdfOptions};
|
||||
|
||||
let result = process_pdf_with_ocr(
|
||||
"document.pdf",
|
||||
OcrPdfOptions::auto().page_numbers([1, 2, 3]),
|
||||
)?;
|
||||
|
||||
println!("{}", result.markdown);
|
||||
println!("OCR pages: {:?}", result.pages_routed_to_ocr);
|
||||
println!(
|
||||
"Hosted fallback pages: {:?}",
|
||||
result.pages_recommending_hosted,
|
||||
);
|
||||
```
|
||||
|
||||
Native extraction always runs first. In `Auto`, a clean PDF returns before
|
||||
PDFium loading, model-cache access, HTTP, or OAR initialization. Model files
|
||||
remain external and the default crate feature set remains unchanged. `Off`
|
||||
provides the same native-only behavior through the OCR result/provenance
|
||||
shape; `Force` renders every selected page. OCR uses the existing deterministic
|
||||
table, column, reading-order, and Markdown assembly path; no learned layout
|
||||
model is included.
|
||||
|
||||
For ambiguous mixed pages, `Auto` privately retains clean native fragments
|
||||
instead of discarding them when OCR is selected. After recognition it compares
|
||||
script-agnostic text quality, OCR confidence, character overlap, and material
|
||||
new coverage. Exact native text wins over a duplicate or weak OCR hypothesis;
|
||||
complementary image-backed text is fused; and pages where both candidates are
|
||||
weak recommend the hosted document pipeline. A page routed because native
|
||||
coverage appeared incomplete also recommends hosted processing when confident
|
||||
OCR only duplicates the retained fragment: the agreement preserves trustworthy
|
||||
text, but neither hypothesis proves full-page coverage. Public native-only
|
||||
extraction continues to suppress pages marked unreliable, and clean text
|
||||
documents pay no renderer or model-initialization cost.
|
||||
|
||||
In `Auto`, pages routed only for suspicious font encoding or vectorized text
|
||||
first get a bounded positioned-text probe through PDFium. A credible recovered
|
||||
text layer with sufficient geometric page coverage skips rasterization and
|
||||
model loading for that page; garbled, partial, or insubstantial recovery
|
||||
continues through OCR. Recovered tables are reflected in the same document
|
||||
metadata as tables found by the primary extractor.
|
||||
|
||||
The one-call API keeps the most recently used verified OCR engine in process.
|
||||
Long-lived workers therefore verify the pinned artifacts and build the ONNX
|
||||
sessions once, then reuse those loaded sessions across documents. The cache is
|
||||
bounded to one model configuration and keyed by normalized model/runtime paths
|
||||
plus the pinned manifest revision and artifact digests; switching the model
|
||||
directory, runtime library, or compiled manifest replaces it. An active engine
|
||||
owns the model data it already verified, so mutating artifacts in place does
|
||||
not hot-reload a running process; restart the process when intentionally
|
||||
replacing files at the same paths. CPU inference uses at most four intra-op
|
||||
threads per ONNX session so a single small page does not oversubscribe larger
|
||||
hosts, and recognizes variable-width line crops individually to avoid
|
||||
padding-heavy CPU batches. The high-level pipeline renders and fuses at most
|
||||
four routed pages at a time, bounding bitmap memory on long documents.
|
||||
|
||||
Build the CLI with the same opt-in feature:
|
||||
|
||||
```bash
|
||||
cargo install pdf-inspector --features ocr --bin pdf2md
|
||||
cargo build --release --features ocr --bin pdf2md
|
||||
pdf2md document.pdf --ocr auto --raw
|
||||
pdf2md document.pdf --ocr auto --json
|
||||
pdf2md document.pdf --ocr auto --ocr-offline --ocr-model-dir /opt/models/pp-ocrv6-small
|
||||
```
|
||||
|
||||
CLI controls include `--ocr-dpi`, `--ocr-min-confidence`,
|
||||
`--ocr-hosted-threshold`, `--select-pages`, and the existing encrypted-PDF
|
||||
`--password` option. JSON output has `schema_version: 1` and includes per-page Markdown, source/model
|
||||
provenance, confidence, timings, warnings, routed pages, and hosted-fallback
|
||||
recommendations. Page numbers in `OcrPdfResult` and its per-page provenance
|
||||
are 1-indexed, matching the PDF page numbers accepted by
|
||||
`OcrPdfOptions::page_numbers`.
|
||||
|
||||
Extract per-page Markdown (one string per page, plus document-wide layout
|
||||
metadata):
|
||||
|
||||
|
||||
Generated
+30
-2253
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -7,7 +7,7 @@ edition = "2021"
|
||||
crate-type = ["cdylib"]
|
||||
|
||||
[dependencies]
|
||||
pdf-inspector = { path = "..", features = ["ocr"] }
|
||||
pdf-inspector = { path = ".." }
|
||||
napi = { version = "3.0.0", features = ["serde-json"] }
|
||||
napi-derive = "3.0.0"
|
||||
|
||||
|
||||
+1
-50
@@ -10,8 +10,7 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
||||
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Selective OCR** — `Auto` routes only pages rejected by native extraction and returns source/model provenance plus hosted-fallback recommendations.
|
||||
- **External artifacts** — the native package embeds no OCR models, PDFium, or ONNX Runtime; clean `Auto` requests never load or download them.
|
||||
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~5–6 MB platform binary, TypeScript definitions included.
|
||||
|
||||
## Benchmark
|
||||
|
||||
@@ -37,40 +36,8 @@ bun add @firecrawl/pdf-inspector
|
||||
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||
platform library search path. The pinned OCR model set is downloaded and
|
||||
checksum-verified on the first routed page; use `offline: true` with a warm
|
||||
cache or `modelDirectory` to prohibit network access.
|
||||
|
||||
## API
|
||||
|
||||
### `processPdfWithOcr(buffer: Buffer, options?: OcrOptions): Promise<OcrPdfResult>`
|
||||
|
||||
Run native extraction first and OCR only the pages selected by its quality
|
||||
signals. The default mode is `Auto`; `Off` returns the same detailed result
|
||||
shape without external runtime work, and `Force` OCRs every selected page.
|
||||
The work runs on the libuv thread pool and never blocks Node's event loop.
|
||||
|
||||
```typescript
|
||||
import { OcrMode, processPdfWithOcr } from '@firecrawl/pdf-inspector'
|
||||
|
||||
const result = await processPdfWithOcr(pdf, {
|
||||
mode: OcrMode.Auto,
|
||||
pageNumbers: [1, 3], // 1-indexed
|
||||
})
|
||||
|
||||
for (const page of result.pages) {
|
||||
console.log(page.pageNumber, page.provenance.source)
|
||||
}
|
||||
console.log(result.pagesRoutedToOcr)
|
||||
console.log(result.pagesRecommendingHosted)
|
||||
```
|
||||
|
||||
For offline deployments, pass `modelDirectory` and `offline: true`. Other
|
||||
controls include `dpi`, `minimumConfidence`,
|
||||
`hostedRecommendationConfidence`, and `password`.
|
||||
|
||||
### `classifyPdf(buffer: Buffer): PdfClassification`
|
||||
|
||||
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
|
||||
@@ -157,22 +124,6 @@ interface RegionText {
|
||||
needsOcr: boolean // true when text is unreliable
|
||||
ocrReason?: string // "suspected_garbled_text" when known
|
||||
}
|
||||
|
||||
interface OcrPdfResult {
|
||||
markdown: string
|
||||
pages: OcrPageResult[] // 1-indexed pages + provenance
|
||||
pageCount: number
|
||||
pagesRecommendedForOcr: number[]
|
||||
pagesRoutedToOcr: number[]
|
||||
pagesRecommendingHosted: number[]
|
||||
ocrReasonsByPage: PageOcrReasons[]
|
||||
pagesWithTables: number[]
|
||||
pagesWithColumns: number[]
|
||||
isComplex: boolean
|
||||
processingTimeMs: number
|
||||
renderTimeMs: number
|
||||
ocrTimeMs: number
|
||||
}
|
||||
```
|
||||
|
||||
## Platforms
|
||||
|
||||
-241
@@ -27,26 +27,6 @@ pub enum ItemType {
|
||||
FormField,
|
||||
}
|
||||
|
||||
/// Selects when OCR runs.
|
||||
#[napi(string_enum)]
|
||||
#[derive(Clone, Copy)]
|
||||
pub enum OcrMode {
|
||||
/// Never run OCR; return the native extraction in the OCR result shape.
|
||||
Off,
|
||||
/// Run OCR only on pages selected by the native quality signals.
|
||||
Auto,
|
||||
/// Run OCR on every selected page.
|
||||
Force,
|
||||
}
|
||||
|
||||
/// How final page content was sourced.
|
||||
#[napi(string_enum)]
|
||||
pub enum PageContentSource {
|
||||
Native,
|
||||
Ocr,
|
||||
Fused,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result types
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -148,84 +128,6 @@ pub struct VectorGridDetectionJs {
|
||||
pub cell_bboxes: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
/// Options for one-call native extraction with selective OCR.
|
||||
#[napi(object)]
|
||||
#[derive(Clone)]
|
||||
pub struct OcrOptions {
|
||||
/// OCR routing behavior. Defaults to Auto.
|
||||
pub mode: Option<OcrMode>,
|
||||
/// Optional 1-indexed page selection.
|
||||
pub page_numbers: Option<Vec<u32>>,
|
||||
/// Password for an encrypted PDF.
|
||||
pub password: Option<String>,
|
||||
/// Page rasterization resolution. Defaults to 150 DPI.
|
||||
pub dpi: Option<f64>,
|
||||
/// Drop OCR spans below this inclusive 0-1 threshold.
|
||||
pub minimum_confidence: Option<f64>,
|
||||
/// Recommend hosted parsing below this inclusive 0-1 page confidence.
|
||||
pub hosted_recommendation_confidence: Option<f64>,
|
||||
/// Directory containing an offline OCR model set.
|
||||
pub model_directory: Option<String>,
|
||||
/// Disable model downloads and require a model directory or warm cache.
|
||||
pub offline: Option<bool>,
|
||||
}
|
||||
|
||||
/// Exact OCR model identity retained in page provenance.
|
||||
#[napi(object)]
|
||||
pub struct OcrModelIdentity {
|
||||
pub name: String,
|
||||
pub revision: String,
|
||||
}
|
||||
|
||||
/// Per-page OCR processing timings.
|
||||
#[napi(object)]
|
||||
pub struct OcrTimings {
|
||||
pub render_ms: u32,
|
||||
pub ocr_ms: u32,
|
||||
pub assembly_ms: u32,
|
||||
}
|
||||
|
||||
/// Source, model, confidence, and fallback metadata for one page.
|
||||
#[napi(object)]
|
||||
pub struct OcrPageProvenance {
|
||||
/// 1-indexed page number.
|
||||
pub page_number: u32,
|
||||
pub source: PageContentSource,
|
||||
pub ocr_model: Option<OcrModelIdentity>,
|
||||
pub render_dpi: Option<f64>,
|
||||
pub ocr_confidence: Option<f64>,
|
||||
pub timings: OcrTimings,
|
||||
pub warnings: Vec<String>,
|
||||
pub hosted_recommended: bool,
|
||||
}
|
||||
|
||||
/// Final Markdown and provenance for one page.
|
||||
#[napi(object)]
|
||||
pub struct OcrPageResult {
|
||||
/// 1-indexed page number.
|
||||
pub page_number: u32,
|
||||
pub markdown: String,
|
||||
pub provenance: OcrPageProvenance,
|
||||
}
|
||||
|
||||
/// Complete native/OCR Markdown output.
|
||||
#[napi(object)]
|
||||
pub struct OcrPdfResult {
|
||||
pub markdown: String,
|
||||
pub pages: Vec<OcrPageResult>,
|
||||
pub page_count: u32,
|
||||
pub pages_recommended_for_ocr: Vec<u32>,
|
||||
pub pages_routed_to_ocr: Vec<u32>,
|
||||
pub pages_recommending_hosted: Vec<u32>,
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
pub is_complex: bool,
|
||||
pub processing_time_ms: u32,
|
||||
pub render_time_ms: u32,
|
||||
pub ocr_time_ms: u32,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -266,103 +168,6 @@ fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_core_ocr_options(options: Option<OcrOptions>) -> pdf_inspector::vision::OcrPdfOptions {
|
||||
let mut result = pdf_inspector::vision::OcrPdfOptions::auto();
|
||||
let Some(options) = options else {
|
||||
return result;
|
||||
};
|
||||
|
||||
if let Some(mode) = options.mode {
|
||||
result.ocr.mode = match mode {
|
||||
OcrMode::Off => pdf_inspector::vision::OcrMode::Off,
|
||||
OcrMode::Auto => pdf_inspector::vision::OcrMode::Auto,
|
||||
OcrMode::Force => pdf_inspector::vision::OcrMode::Force,
|
||||
};
|
||||
}
|
||||
if let Some(pages) = options.page_numbers {
|
||||
result = result.page_numbers(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(dpi) = options.dpi {
|
||||
result.render.dpi = dpi as f32;
|
||||
}
|
||||
if let Some(minimum_confidence) = options.minimum_confidence {
|
||||
result.ocr.minimum_confidence = minimum_confidence as f32;
|
||||
}
|
||||
if let Some(confidence) = options.hosted_recommendation_confidence {
|
||||
result.hosted_recommendation_confidence = confidence as f32;
|
||||
}
|
||||
if let Some(directory) = options.model_directory {
|
||||
result.ocr.model_directory = Some(directory.into());
|
||||
}
|
||||
if options.offline.unwrap_or(false) {
|
||||
result.ocr.model_downloads = pdf_inspector::vision::ModelDownloadPolicy::Offline;
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
fn convert_page_content_source(
|
||||
source: pdf_inspector::vision::PageContentSource,
|
||||
) -> PageContentSource {
|
||||
match source {
|
||||
pdf_inspector::vision::PageContentSource::Native => PageContentSource::Native,
|
||||
pdf_inspector::vision::PageContentSource::Ocr => PageContentSource::Ocr,
|
||||
pdf_inspector::vision::PageContentSource::Fused => PageContentSource::Fused,
|
||||
_ => PageContentSource::Native,
|
||||
}
|
||||
}
|
||||
|
||||
fn timing_ms(value: u64) -> u32 {
|
||||
u32::try_from(value).unwrap_or(u32::MAX)
|
||||
}
|
||||
|
||||
fn to_napi_ocr_result(result: pdf_inspector::vision::OcrPdfResult) -> OcrPdfResult {
|
||||
OcrPdfResult {
|
||||
markdown: result.markdown,
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|page| {
|
||||
let provenance = page.provenance;
|
||||
OcrPageResult {
|
||||
page_number: page.page_number,
|
||||
markdown: page.markdown,
|
||||
provenance: OcrPageProvenance {
|
||||
page_number: provenance.page_number,
|
||||
source: convert_page_content_source(provenance.source),
|
||||
ocr_model: provenance.ocr_model.map(|model| OcrModelIdentity {
|
||||
name: model.name,
|
||||
revision: model.revision,
|
||||
}),
|
||||
render_dpi: provenance.render_dpi.map(f64::from),
|
||||
ocr_confidence: provenance.ocr_confidence.map(f64::from),
|
||||
timings: OcrTimings {
|
||||
render_ms: timing_ms(provenance.timings.render_ms),
|
||||
ocr_ms: timing_ms(provenance.timings.ocr_ms),
|
||||
assembly_ms: timing_ms(provenance.timings.assembly_ms),
|
||||
},
|
||||
warnings: provenance.warnings,
|
||||
hosted_recommended: provenance.hosted_recommended,
|
||||
},
|
||||
}
|
||||
})
|
||||
.collect(),
|
||||
page_count: result.page_count,
|
||||
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
|
||||
pages_routed_to_ocr: result.pages_routed_to_ocr,
|
||||
pages_recommending_hosted: result.pages_recommending_hosted,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
is_complex: result.is_complex,
|
||||
processing_time_ms: timing_ms(result.processing_time_ms),
|
||||
render_time_ms: timing_ms(result.render_time_ms),
|
||||
ocr_time_ms: timing_ms(result.ocr_time_ms),
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||
@@ -414,13 +219,6 @@ fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult>
|
||||
Ok(to_napi_result(result))
|
||||
}
|
||||
|
||||
fn process_pdf_with_ocr_impl(bytes: &[u8], options: Option<OcrOptions>) -> Result<OcrPdfResult> {
|
||||
let options = to_core_ocr_options(options);
|
||||
let result = pdf_inspector::vision::process_pdf_with_ocr_mem(bytes, options)
|
||||
.map_err(|error| to_napi_err(error, "process_pdf_with_ocr"))?;
|
||||
Ok(to_napi_ocr_result(result))
|
||||
}
|
||||
|
||||
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
@@ -1020,45 +818,6 @@ pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<P
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ProcessPdfWithOcrTask {
|
||||
bytes: Vec<u8>,
|
||||
options: Option<OcrOptions>,
|
||||
}
|
||||
|
||||
impl Task for ProcessPdfWithOcrTask {
|
||||
type Output = OcrPdfResult;
|
||||
type JsValue = OcrPdfResult;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
let options = self.options.take();
|
||||
catch_panic(
|
||||
"process_pdf_with_ocr",
|
||||
panic::AssertUnwindSafe(move || process_pdf_with_ocr_impl(&bytes, options)),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Process a PDF with selective OCR on the libuv thread pool.
|
||||
///
|
||||
/// OCR defaults to Auto, which only loads PDFium, ONNX Runtime, and the OCR
|
||||
/// model if native extraction routes at least one page. The input buffer is
|
||||
/// copied before the promise is returned and is safe to reuse immediately.
|
||||
#[napi(ts_return_type = "Promise<OcrPdfResult>")]
|
||||
pub fn process_pdf_with_ocr(
|
||||
buffer: Buffer,
|
||||
options: Option<OcrOptions>,
|
||||
) -> AsyncTask<ProcessPdfWithOcrTask> {
|
||||
AsyncTask::new(ProcessPdfWithOcrTask {
|
||||
bytes: buffer.to_vec(),
|
||||
options,
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ClassifyPdfTask {
|
||||
bytes: Vec<u8>,
|
||||
}
|
||||
|
||||
@@ -3,7 +3,6 @@ import { strict as assert } from 'assert';
|
||||
import {
|
||||
processPdf,
|
||||
processPdfAsync,
|
||||
processPdfWithOcr,
|
||||
detectPdf,
|
||||
classifyPdf,
|
||||
classifyPdfAsync,
|
||||
@@ -221,37 +220,6 @@ const fromMutated = await inFlight;
|
||||
assert.equal(fromMutated.markdown, result.markdown);
|
||||
console.log(' processPdfAsync input copied at call time: OK');
|
||||
|
||||
// --- Selective OCR ---
|
||||
console.log('Testing processPdfWithOcr...');
|
||||
|
||||
// Off exercises the complete result/provenance contract without loading
|
||||
// external PDFium, ONNX Runtime, or model artifacts.
|
||||
const ocrOff = await processPdfWithOcr(fixture, { mode: 'Off' });
|
||||
assert.equal(ocrOff.pageCount, 3);
|
||||
assert.equal(ocrOff.pages.length, 3);
|
||||
assert.deepEqual(ocrOff.pagesRoutedToOcr, []);
|
||||
assert.ok(ocrOff.pages.every(page => page.provenance.source === 'Native'));
|
||||
assert.ok(ocrOff.pages.every(page => page.provenance.ocrModel === undefined));
|
||||
assert.ok(ocrOff.markdown.length > 0);
|
||||
|
||||
// Auto must preserve the lightweight path for clean text PDFs.
|
||||
const ocrAuto = await processPdfWithOcr(fixture);
|
||||
assert.deepEqual(ocrAuto.pagesRoutedToOcr, []);
|
||||
assert.equal(ocrAuto.renderTimeMs, 0);
|
||||
assert.equal(ocrAuto.ocrTimeMs, 0);
|
||||
|
||||
const ocrSelected = await processPdfWithOcr(fixture, {
|
||||
mode: 'Off',
|
||||
pageNumbers: [2],
|
||||
});
|
||||
assert.deepEqual(ocrSelected.pages.map(page => page.pageNumber), [2]);
|
||||
|
||||
await assert.rejects(
|
||||
processPdfWithOcr(fixture, { mode: 'Off', pageNumbers: [0] }),
|
||||
/page 0/,
|
||||
);
|
||||
console.log(' processPdfWithOcr: OK');
|
||||
|
||||
// concurrent async calls all settle
|
||||
const [c1, c2, c3] = await Promise.all([
|
||||
processPdfAsync(fixture),
|
||||
|
||||
+1
-81
@@ -1,6 +1,6 @@
|
||||
"""Type stubs for pdf_inspector."""
|
||||
|
||||
from typing import Literal, Optional
|
||||
from typing import Optional
|
||||
|
||||
class PdfResult:
|
||||
"""Result of processing a PDF file."""
|
||||
@@ -27,53 +27,6 @@ class PageOcrReasons:
|
||||
reasons: list[str]
|
||||
"""Machine-readable OCR reason identifiers."""
|
||||
|
||||
class OcrModelIdentity:
|
||||
"""Exact OCR model identity retained in page provenance."""
|
||||
name: str
|
||||
revision: str
|
||||
|
||||
class OcrTimings:
|
||||
"""Per-page OCR processing timings."""
|
||||
render_ms: int
|
||||
ocr_ms: int
|
||||
assembly_ms: int
|
||||
|
||||
class OcrPageProvenance:
|
||||
"""Source, model, confidence, and fallback metadata for one page."""
|
||||
page_number: int
|
||||
"""1-indexed page number."""
|
||||
source: Literal["native", "ocr", "fused"]
|
||||
"""'native', 'ocr', or 'fused'."""
|
||||
ocr_model: Optional[OcrModelIdentity]
|
||||
render_dpi: Optional[float]
|
||||
ocr_confidence: Optional[float]
|
||||
timings: OcrTimings
|
||||
warnings: list[str]
|
||||
hosted_recommended: bool
|
||||
|
||||
class OcrPageResult:
|
||||
"""Final Markdown and provenance for one page."""
|
||||
page_number: int
|
||||
"""1-indexed page number."""
|
||||
markdown: str
|
||||
provenance: OcrPageProvenance
|
||||
|
||||
class OcrPdfResult:
|
||||
"""Complete native/OCR Markdown output."""
|
||||
markdown: str
|
||||
pages: list[OcrPageResult]
|
||||
page_count: int
|
||||
pages_recommended_for_ocr: list[int]
|
||||
pages_routed_to_ocr: list[int]
|
||||
pages_recommending_hosted: list[int]
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
is_complex: bool
|
||||
processing_time_ms: int
|
||||
render_time_ms: int
|
||||
ocr_time_ms: int
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
@@ -161,39 +114,6 @@ def process_pdf_bytes(data: bytes, pages: Optional[list[int]] = None) -> PdfResu
|
||||
"""Process a PDF from bytes in memory."""
|
||||
...
|
||||
|
||||
def process_pdf_with_ocr(
|
||||
path: str,
|
||||
*,
|
||||
mode: Literal["off", "auto", "force"] = "auto",
|
||||
page_numbers: Optional[list[int]] = None,
|
||||
password: Optional[str] = None,
|
||||
dpi: float = 150.0,
|
||||
minimum_confidence: float = 0.0,
|
||||
hosted_recommendation_confidence: float = 0.5,
|
||||
model_directory: Optional[str] = None,
|
||||
offline: bool = False,
|
||||
) -> OcrPdfResult:
|
||||
"""Process a PDF through native extraction and selective OCR.
|
||||
|
||||
Page numbers are 1-indexed. OCR runs without holding the Python GIL.
|
||||
"""
|
||||
...
|
||||
|
||||
def process_pdf_with_ocr_bytes(
|
||||
data: bytes,
|
||||
*,
|
||||
mode: Literal["off", "auto", "force"] = "auto",
|
||||
page_numbers: Optional[list[int]] = None,
|
||||
password: Optional[str] = None,
|
||||
dpi: float = 150.0,
|
||||
minimum_confidence: float = 0.0,
|
||||
hosted_recommendation_confidence: float = 0.5,
|
||||
model_directory: Optional[str] = None,
|
||||
offline: bool = False,
|
||||
) -> OcrPdfResult:
|
||||
"""Process PDF bytes through native extraction and selective OCR."""
|
||||
...
|
||||
|
||||
def detect_pdf(path: str) -> PdfResult:
|
||||
"""Fast detection only — no text extraction."""
|
||||
...
|
||||
|
||||
+6
-311
@@ -1,11 +1,6 @@
|
||||
//! CLI tool for PDF to Markdown conversion
|
||||
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
use pdf_inspector::vision::{
|
||||
process_pdf_with_ocr, ModelDownloadPolicy, OcrMode, OcrOptions, OcrPdfOptions, OcrPdfResult,
|
||||
PageContentSource, RenderOptions,
|
||||
};
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages_with_password, process_pdf_with_options, LayoutComplexity,
|
||||
PdfOptions, PdfType, ProcessMode, TextItem,
|
||||
@@ -108,146 +103,6 @@ fn format_items_json(items: &[TextItem]) -> String {
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
fn optional_json_number(value: Option<f32>) -> String {
|
||||
value
|
||||
.filter(|value| value.is_finite())
|
||||
.map(|value| format!("{value:.4}"))
|
||||
.unwrap_or_else(|| "null".to_string())
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
fn format_ocr_json(result: &OcrPdfResult) -> String {
|
||||
let routed = result
|
||||
.pages_routed_to_ocr
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let recommended = result
|
||||
.pages_recommended_for_ocr
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let hosted = result
|
||||
.pages_recommending_hosted
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let pages = result
|
||||
.pages
|
||||
.iter()
|
||||
.map(|page| {
|
||||
let provenance = &page.provenance;
|
||||
let source = match provenance.source {
|
||||
PageContentSource::Native => "native",
|
||||
PageContentSource::Ocr => "ocr",
|
||||
PageContentSource::Fused => "fused",
|
||||
_ => "unknown",
|
||||
};
|
||||
let model = provenance
|
||||
.ocr_model
|
||||
.as_ref()
|
||||
.map(|model| {
|
||||
format!(
|
||||
r#"{{"name":"{}","revision":"{}"}}"#,
|
||||
json_escape(&model.name),
|
||||
json_escape(&model.revision)
|
||||
)
|
||||
})
|
||||
.unwrap_or_else(|| "null".to_string());
|
||||
let warnings = provenance
|
||||
.warnings
|
||||
.iter()
|
||||
.map(|warning| format!(r#""{}""#, json_escape(warning)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(
|
||||
r#"{{"page":{},"source":"{}","markdown":"{}","ocr_model":{},"render_dpi":{},"ocr_confidence":{},"hosted_recommended":{},"timings":{{"render_ms":{},"ocr_ms":{},"assembly_ms":{}}},"warnings":[{}]}}"#,
|
||||
provenance.page_number,
|
||||
source,
|
||||
json_escape(&page.markdown),
|
||||
model,
|
||||
optional_json_number(provenance.render_dpi),
|
||||
optional_json_number(provenance.ocr_confidence),
|
||||
provenance.hosted_recommended,
|
||||
provenance.timings.render_ms,
|
||||
provenance.timings.ocr_ms,
|
||||
provenance.timings.assembly_ms,
|
||||
warnings,
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let table_pages = result
|
||||
.pages_with_tables
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let column_pages = result
|
||||
.pages_with_columns
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
format!(
|
||||
r#"{{"schema_version":1,"page_count":{},"processing_time_ms":{},"render_time_ms":{},"ocr_time_ms":{},"pages_recommended_for_ocr":[{}],"pages_routed_to_ocr":[{}],"pages_recommending_hosted":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"pages":[{}],"markdown":"{}"}}"#,
|
||||
result.page_count,
|
||||
result.processing_time_ms,
|
||||
result.render_time_ms,
|
||||
result.ocr_time_ms,
|
||||
recommended,
|
||||
routed,
|
||||
hosted,
|
||||
ocr_reasons,
|
||||
result.is_complex,
|
||||
table_pages,
|
||||
column_pages,
|
||||
pages,
|
||||
json_escape(&result.markdown),
|
||||
)
|
||||
}
|
||||
|
||||
fn argument_value<'a>(args: &'a [String], name: &str) -> Result<Option<&'a str>, String> {
|
||||
args.iter()
|
||||
.position(|argument| argument == name)
|
||||
.map(|index| {
|
||||
args.get(index + 1)
|
||||
.map(String::as_str)
|
||||
.ok_or_else(|| format!("{name} requires a value"))
|
||||
})
|
||||
.transpose()
|
||||
}
|
||||
|
||||
fn format_ocr_error_json(error: &str) -> String {
|
||||
format!(r#"{{"schema_version":1,"error":"{}"}}"#, json_escape(error))
|
||||
}
|
||||
|
||||
fn exit_ocr_error(error: &str, json_output: bool) -> ! {
|
||||
if json_output {
|
||||
println!("{}", format_ocr_error_json(error));
|
||||
} else {
|
||||
eprintln!("Error: {error}");
|
||||
}
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
fn float_argument(args: &[String], name: &str, default: f32) -> Result<f32, String> {
|
||||
argument_value(args, name)?
|
||||
.map(|value| {
|
||||
value
|
||||
.parse::<f32>()
|
||||
.map_err(|_| format!("{name} requires a number, got {value:?}"))
|
||||
})
|
||||
.transpose()
|
||||
.map(|value| value.unwrap_or(default))
|
||||
}
|
||||
|
||||
fn extract_items_json(
|
||||
pdf_path: &str,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
@@ -259,9 +114,7 @@ fn extract_items_json(
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{extract_items_json, format_items_json, format_ocr_error_json};
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
use super::{format_ocr_json, process_pdf_with_ocr, OcrPdfOptions};
|
||||
use super::{extract_items_json, format_items_json};
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
@@ -311,27 +164,6 @@ mod tests {
|
||||
"decrypted item JSON should contain fixture text, got {json}"
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
#[test]
|
||||
fn ocr_json_has_a_versioned_stable_envelope() {
|
||||
let result =
|
||||
process_pdf_with_ocr("tests/fixtures/thermo-freon12.pdf", OcrPdfOptions::new())
|
||||
.unwrap();
|
||||
let json = format_ocr_json(&result);
|
||||
|
||||
assert!(json.starts_with(r#"{"schema_version":1,"page_count":3,"#));
|
||||
assert!(json.contains(r#""page":1,"source":"native""#));
|
||||
assert!(!json.contains("layout_ms"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ocr_json_errors_use_the_same_versioned_envelope() {
|
||||
assert_eq!(
|
||||
format_ocr_error_json("bad \"value\""),
|
||||
r#"{"schema_version":1,"error":"bad \"value\""}"#
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
@@ -410,12 +242,6 @@ fn main() {
|
||||
eprintln!(" --password PW Password for an encrypted PDF");
|
||||
eprintln!(" --detect-only Only detect PDF type (no extraction)");
|
||||
eprintln!(" --analyze Detect + extract + layout analysis (no markdown)");
|
||||
eprintln!(" --ocr MODE OCR mode: off, auto, or force (requires feature `ocr`)");
|
||||
eprintln!(" --ocr-dpi N OCR render resolution (default: 150)");
|
||||
eprintln!(" --ocr-min-confidence N Drop OCR spans below N (default: 0)");
|
||||
eprintln!(" --ocr-hosted-threshold N Recommend hosted parsing below N (default: 0.5)");
|
||||
eprintln!(" --ocr-model-dir DIR Use a package-managed local model directory");
|
||||
eprintln!(" --ocr-offline Never download missing OCR models");
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
@@ -427,10 +253,6 @@ fn main() {
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
let analyze = args.iter().any(|a| a == "--analyze");
|
||||
let ocr_mode_argument = argument_value(&args, "--ocr").unwrap_or_else(|error| {
|
||||
eprintln!("Error: {error}");
|
||||
process::exit(1);
|
||||
});
|
||||
|
||||
// Parse --password value
|
||||
let password = args.iter().position(|a| a == "--password").map(|i| {
|
||||
@@ -461,138 +283,6 @@ fn main() {
|
||||
})
|
||||
});
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
.map(|s| s.as_str());
|
||||
|
||||
let has_ocr_only_option = [
|
||||
"--ocr-dpi",
|
||||
"--ocr-min-confidence",
|
||||
"--ocr-hosted-threshold",
|
||||
"--ocr-model-dir",
|
||||
"--ocr-offline",
|
||||
]
|
||||
.iter()
|
||||
.any(|option| args.iter().any(|argument| argument == option));
|
||||
if ocr_mode_argument.is_none() && has_ocr_only_option {
|
||||
exit_ocr_error(
|
||||
"OCR options require --ocr off, --ocr auto, or --ocr force",
|
||||
json_output,
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(mode) = ocr_mode_argument {
|
||||
if items_json_output || detect_only || analyze {
|
||||
exit_ocr_error(
|
||||
"--ocr cannot be combined with --items-json, --detect-only, or --analyze",
|
||||
json_output,
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(not(all(feature = "ocr", not(target_arch = "wasm32"))))]
|
||||
{
|
||||
let _ = mode;
|
||||
exit_ocr_error(
|
||||
"this pdf2md build does not include OCR; rebuild with --features ocr",
|
||||
json_output,
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
{
|
||||
let mode = match mode {
|
||||
"off" => OcrMode::Off,
|
||||
"auto" => OcrMode::Auto,
|
||||
"force" => OcrMode::Force,
|
||||
value => {
|
||||
exit_ocr_error(
|
||||
&format!("invalid --ocr mode {value:?}; expected off, auto, or force"),
|
||||
json_output,
|
||||
);
|
||||
}
|
||||
};
|
||||
let dpi = float_argument(&args, "--ocr-dpi", 150.0).unwrap_or_else(|error| {
|
||||
exit_ocr_error(&error, json_output);
|
||||
});
|
||||
let minimum_confidence = float_argument(&args, "--ocr-min-confidence", 0.0)
|
||||
.unwrap_or_else(|error| {
|
||||
exit_ocr_error(&error, json_output);
|
||||
});
|
||||
let hosted_threshold = float_argument(&args, "--ocr-hosted-threshold", 0.5)
|
||||
.unwrap_or_else(|error| {
|
||||
exit_ocr_error(&error, json_output);
|
||||
});
|
||||
let model_directory =
|
||||
argument_value(&args, "--ocr-model-dir").unwrap_or_else(|error| {
|
||||
exit_ocr_error(&error, json_output);
|
||||
});
|
||||
|
||||
let mut ocr = OcrOptions::new()
|
||||
.mode(mode)
|
||||
.minimum_confidence(minimum_confidence);
|
||||
if let Some(directory) = model_directory {
|
||||
ocr = ocr.model_directory(directory);
|
||||
}
|
||||
if args.iter().any(|argument| argument == "--ocr-offline") {
|
||||
ocr = ocr.model_downloads(ModelDownloadPolicy::Offline);
|
||||
}
|
||||
let mut markdown = pdf_inspector::MarkdownOptions::default();
|
||||
if compact_output {
|
||||
markdown.profile = pdf_inspector::MarkdownProfile::Compact;
|
||||
}
|
||||
markdown.include_page_numbers = page_numbers;
|
||||
let mut pdf_options = OcrPdfOptions::new()
|
||||
.render(RenderOptions::new().dpi(dpi))
|
||||
.ocr(ocr)
|
||||
.markdown(markdown)
|
||||
.hosted_recommendation_confidence(hosted_threshold);
|
||||
if let Some(pages) = page_filter.clone() {
|
||||
pdf_options = pdf_options.page_numbers(pages);
|
||||
}
|
||||
if let Some(password) = password.clone() {
|
||||
pdf_options = pdf_options.password(password);
|
||||
}
|
||||
|
||||
match process_pdf_with_ocr(pdf_path, pdf_options) {
|
||||
Ok(result) => {
|
||||
if json_output {
|
||||
println!("{}", format_ocr_json(&result));
|
||||
} else if raw_output {
|
||||
print!("{}", result.markdown);
|
||||
} else {
|
||||
eprintln!("PDF to Markdown Conversion (OCR)");
|
||||
eprintln!("======================================");
|
||||
eprintln!("File: {pdf_path}");
|
||||
eprintln!("Pages: {}", result.page_count);
|
||||
eprintln!("Pages routed to OCR: {:?}", result.pages_routed_to_ocr);
|
||||
if !result.pages_recommending_hosted.is_empty() {
|
||||
eprintln!(
|
||||
"Hosted parsing recommended for pages: {:?}",
|
||||
result.pages_recommending_hosted
|
||||
);
|
||||
}
|
||||
eprintln!("Processing time: {}ms", result.processing_time_ms);
|
||||
if let Some(output) = output_file {
|
||||
fs::write(output, &result.markdown)
|
||||
.expect("Failed to write output file");
|
||||
eprintln!("Markdown written to: {output}");
|
||||
} else {
|
||||
eprintln!();
|
||||
eprintln!("--- Markdown Output ---");
|
||||
eprintln!();
|
||||
print!("{}", result.markdown);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(error) => {
|
||||
exit_ocr_error(&error.to_string(), json_output);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if items_json_output {
|
||||
match extract_items_json(pdf_path, page_filter.as_ref(), password.as_deref()) {
|
||||
Ok(json) => println!("{}", json),
|
||||
@@ -604,6 +294,11 @@ fn main() {
|
||||
return;
|
||||
}
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
.map(|s| s.as_str());
|
||||
|
||||
let process_mode = if detect_only {
|
||||
ProcessMode::DetectOnly
|
||||
} else if analyze {
|
||||
|
||||
+12
-179
@@ -459,44 +459,8 @@ pub fn extract_pages_markdown_mem(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<PagesExtractionResult, PdfError> {
|
||||
extract_pages_markdown_mem_impl(
|
||||
buffer,
|
||||
pages,
|
||||
None,
|
||||
&MarkdownOptions::default(),
|
||||
false,
|
||||
false,
|
||||
)
|
||||
.map(|(result, _)| result)
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
pub(crate) fn extract_pages_markdown_mem_for_ocr(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
password: Option<&str>,
|
||||
markdown_options: &MarkdownOptions,
|
||||
) -> Result<(PagesExtractionResult, u32), PdfError> {
|
||||
extract_pages_markdown_mem_impl(
|
||||
buffer,
|
||||
pages,
|
||||
password,
|
||||
markdown_options,
|
||||
markdown_options.strip_headers_footers,
|
||||
true,
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_pages_markdown_mem_impl(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
password: Option<&str>,
|
||||
markdown_options: &MarkdownOptions,
|
||||
strip_repeated_headers_footers: bool,
|
||||
preserve_ocr_candidates: bool,
|
||||
) -> Result<(PagesExtractionResult, u32), PdfError> {
|
||||
validate_pdf_bytes(buffer)?;
|
||||
let (doc, page_count) = load_document_from_mem_with_password(buffer, password)?;
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
@@ -539,11 +503,6 @@ fn extract_pages_markdown_mem_impl(
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
let repeated_header_footer_items = if strip_repeated_headers_footers {
|
||||
repeated_header_footer_item_keys(&all_items, &page_thresholds, &chart_regions, page_count)
|
||||
} else {
|
||||
HashSet::new()
|
||||
};
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -579,10 +538,7 @@ fn extract_pages_markdown_mem_impl(
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| {
|
||||
item.page == page_1idx
|
||||
&& !repeated_header_footer_items.contains(&HeaderFooterItemKey::from(*item))
|
||||
})
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
@@ -619,7 +575,7 @@ fn extract_pages_markdown_mem_impl(
|
||||
base_font_size: Some(font_stats.most_common_size),
|
||||
include_page_numbers: false,
|
||||
strip_headers_footers: false,
|
||||
..markdown_options.clone()
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = if has_text_quality_issue {
|
||||
@@ -672,143 +628,20 @@ fn extract_pages_markdown_mem_impl(
|
||||
|
||||
results.push(PageMarkdown {
|
||||
page: page_0idx,
|
||||
// The public native extractor continues to suppress unreliable
|
||||
// text. The OCR orchestrator retains clean partial text
|
||||
// internally so it can compare/fuse it with OCR before deciding
|
||||
// what is safe to return.
|
||||
markdown: if needs_ocr && !preserve_ocr_candidates {
|
||||
String::new()
|
||||
} else {
|
||||
md
|
||||
},
|
||||
markdown: if needs_ocr { String::new() } else { md },
|
||||
needs_ocr,
|
||||
ocr_reason,
|
||||
});
|
||||
}
|
||||
|
||||
Ok((
|
||||
PagesExtractionResult {
|
||||
pages: results,
|
||||
pages_with_tables: complexity.pages_with_tables,
|
||||
pages_with_columns: complexity.pages_with_columns,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(ocr_reasons_by_page),
|
||||
is_complex: complexity.is_complex,
|
||||
},
|
||||
page_count,
|
||||
))
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
|
||||
struct HeaderFooterItemKey {
|
||||
page: u32,
|
||||
x: u32,
|
||||
y: u32,
|
||||
text: String,
|
||||
}
|
||||
|
||||
impl From<&TextItem> for HeaderFooterItemKey {
|
||||
fn from(item: &TextItem) -> Self {
|
||||
Self {
|
||||
page: item.page,
|
||||
x: item.x.to_bits(),
|
||||
y: item.y.to_bits(),
|
||||
text: item.text.clone(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn repeated_header_footer_item_keys(
|
||||
items: &[TextItem],
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
page_count: u32,
|
||||
) -> HashSet<HeaderFooterItemKey> {
|
||||
let candidates = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
matches!(
|
||||
item.item_type,
|
||||
types::ItemType::Text | types::ItemType::FormField
|
||||
)
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
let lines = extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
candidates,
|
||||
page_thresholds,
|
||||
&HashSet::new(),
|
||||
chart_regions,
|
||||
);
|
||||
let all_items: HashSet<_> = lines
|
||||
.iter()
|
||||
.flat_map(|line| line.items.iter().map(HeaderFooterItemKey::from))
|
||||
.collect();
|
||||
let kept = markdown::strip_repeated_header_footer_lines(lines, page_count);
|
||||
let kept_items: HashSet<_> = kept
|
||||
.iter()
|
||||
.flat_map(|line| line.items.iter().map(HeaderFooterItemKey::from))
|
||||
.collect();
|
||||
all_items.difference(&kept_items).cloned().collect()
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "ocr", not(target_arch = "wasm32")))]
|
||||
mod ocr_header_footer_tests {
|
||||
use super::*;
|
||||
|
||||
fn item(page: u32, text: &str, y: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x: 10.0,
|
||||
y,
|
||||
width: 120.0,
|
||||
height: 10.0,
|
||||
font: "Test".to_string(),
|
||||
font_size: 10.0,
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: types::ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn local_pipeline_prefilters_document_wide_repeated_headers() {
|
||||
let mut items = Vec::new();
|
||||
let mut thresholds = HashMap::new();
|
||||
for page in 1..=3 {
|
||||
items.push(item(page, "Repeated report header", 800.0));
|
||||
for line in 0..12 {
|
||||
items.push(item(
|
||||
page,
|
||||
&format!("Page {page} paragraph {line} unique content"),
|
||||
700.0 - line as f32 * 40.0,
|
||||
));
|
||||
}
|
||||
thresholds.insert(page, 0.1);
|
||||
}
|
||||
|
||||
let removed = repeated_header_footer_item_keys(&items, &thresholds, &HashMap::new(), 3);
|
||||
assert_eq!(removed.len(), 2);
|
||||
for page in 1..=3 {
|
||||
assert_eq!(
|
||||
removed.contains(&HeaderFooterItemKey::from(&item(
|
||||
page,
|
||||
"Repeated report header",
|
||||
800.0,
|
||||
))),
|
||||
page > 1,
|
||||
);
|
||||
assert!(!removed.contains(&HeaderFooterItemKey::from(&item(
|
||||
page,
|
||||
&format!("Page {page} paragraph 5 unique content"),
|
||||
500.0,
|
||||
))));
|
||||
}
|
||||
}
|
||||
Ok(PagesExtractionResult {
|
||||
pages: results,
|
||||
pages_with_tables: complexity.pages_with_tables,
|
||||
pages_with_columns: complexity.pages_with_columns,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(ocr_reasons_by_page),
|
||||
is_complex: complexity.is_complex,
|
||||
})
|
||||
}
|
||||
|
||||
/// Path-based wrapper for [`extract_pages_markdown_mem`].
|
||||
|
||||
@@ -1144,14 +1144,6 @@ pub fn to_markdown(text: &str, options: MarkdownOptions) -> String {
|
||||
output
|
||||
}
|
||||
|
||||
/// Applies the document-wide repeated header/footer classifier to grouped lines.
|
||||
pub(crate) fn strip_repeated_header_footer_lines(
|
||||
lines: Vec<crate::types::TextLine>,
|
||||
page_count: u32,
|
||||
) -> Vec<crate::types::TextLine> {
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown with structure detection
|
||||
pub fn to_markdown_from_items(items: Vec<TextItem>, options: MarkdownOptions) -> String {
|
||||
to_markdown_from_items_with_rects(items, options, &[])
|
||||
|
||||
-296
@@ -85,107 +85,6 @@ impl PyPageOcrReasons {
|
||||
}
|
||||
}
|
||||
|
||||
/// Exact OCR model identity retained in page provenance.
|
||||
#[pyclass(name = "OcrModelIdentity")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyOcrModelIdentity {
|
||||
#[pyo3(get)]
|
||||
pub name: String,
|
||||
#[pyo3(get)]
|
||||
pub revision: String,
|
||||
}
|
||||
|
||||
/// Per-page OCR processing timings.
|
||||
#[pyclass(name = "OcrTimings")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyOcrTimings {
|
||||
#[pyo3(get)]
|
||||
pub render_ms: u64,
|
||||
#[pyo3(get)]
|
||||
pub ocr_ms: u64,
|
||||
#[pyo3(get)]
|
||||
pub assembly_ms: u64,
|
||||
}
|
||||
|
||||
/// Source, model, confidence, and fallback metadata for one page.
|
||||
#[pyclass(name = "OcrPageProvenance")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyOcrPageProvenance {
|
||||
/// 1-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page_number: u32,
|
||||
/// "native", "ocr", or "fused".
|
||||
#[pyo3(get)]
|
||||
pub source: String,
|
||||
#[pyo3(get)]
|
||||
pub ocr_model: Option<PyOcrModelIdentity>,
|
||||
#[pyo3(get)]
|
||||
pub render_dpi: Option<f32>,
|
||||
#[pyo3(get)]
|
||||
pub ocr_confidence: Option<f32>,
|
||||
#[pyo3(get)]
|
||||
pub timings: PyOcrTimings,
|
||||
#[pyo3(get)]
|
||||
pub warnings: Vec<String>,
|
||||
#[pyo3(get)]
|
||||
pub hosted_recommended: bool,
|
||||
}
|
||||
|
||||
/// Final Markdown and provenance for one page.
|
||||
#[pyclass(name = "OcrPageResult")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyOcrPageResult {
|
||||
/// 1-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page_number: u32,
|
||||
#[pyo3(get)]
|
||||
pub markdown: String,
|
||||
#[pyo3(get)]
|
||||
pub provenance: PyOcrPageProvenance,
|
||||
}
|
||||
|
||||
/// Complete native/OCR Markdown output.
|
||||
#[pyclass(name = "OcrPdfResult")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyOcrPdfResult {
|
||||
#[pyo3(get)]
|
||||
pub markdown: String,
|
||||
#[pyo3(get)]
|
||||
pub pages: Vec<PyOcrPageResult>,
|
||||
#[pyo3(get)]
|
||||
pub page_count: u32,
|
||||
#[pyo3(get)]
|
||||
pub pages_recommended_for_ocr: Vec<u32>,
|
||||
#[pyo3(get)]
|
||||
pub pages_routed_to_ocr: Vec<u32>,
|
||||
#[pyo3(get)]
|
||||
pub pages_recommending_hosted: Vec<u32>,
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
#[pyo3(get)]
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
#[pyo3(get)]
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
#[pyo3(get)]
|
||||
pub is_complex: bool,
|
||||
#[pyo3(get)]
|
||||
pub processing_time_ms: u64,
|
||||
#[pyo3(get)]
|
||||
pub render_time_ms: u64,
|
||||
#[pyo3(get)]
|
||||
pub ocr_time_ms: u64,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyOcrPdfResult {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"OcrPdfResult(pages={}, routed_to_ocr={:?}, recommending_hosted={:?})",
|
||||
self.page_count, self.pages_routed_to_ocr, self.pages_recommending_hosted
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classification wrapper (lightweight)
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -463,101 +362,6 @@ fn to_py_err(e: crate::PdfError) -> PyErr {
|
||||
PyValueError::new_err(e.to_string())
|
||||
}
|
||||
|
||||
struct PythonOcrOptions {
|
||||
mode: String,
|
||||
page_numbers: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
dpi: f32,
|
||||
minimum_confidence: f32,
|
||||
hosted_recommendation_confidence: f32,
|
||||
model_directory: Option<String>,
|
||||
offline: bool,
|
||||
}
|
||||
|
||||
fn build_ocr_options(binding: PythonOcrOptions) -> PyResult<crate::vision::OcrPdfOptions> {
|
||||
let mode = match binding.mode.trim().to_ascii_lowercase().as_str() {
|
||||
"off" => crate::vision::OcrMode::Off,
|
||||
"auto" => crate::vision::OcrMode::Auto,
|
||||
"force" => crate::vision::OcrMode::Force,
|
||||
_ => {
|
||||
return Err(PyValueError::new_err(
|
||||
"mode must be 'off', 'auto', or 'force'",
|
||||
));
|
||||
}
|
||||
};
|
||||
|
||||
let mut options = crate::vision::OcrPdfOptions::new().mode(mode);
|
||||
options.render.dpi = binding.dpi;
|
||||
options.ocr.minimum_confidence = binding.minimum_confidence;
|
||||
options.hosted_recommendation_confidence = binding.hosted_recommendation_confidence;
|
||||
if let Some(pages) = binding.page_numbers {
|
||||
options = options.page_numbers(pages);
|
||||
}
|
||||
if let Some(password) = binding.password {
|
||||
options = options.password(password);
|
||||
}
|
||||
if let Some(directory) = binding.model_directory {
|
||||
options.ocr.model_directory = Some(directory.into());
|
||||
}
|
||||
if binding.offline {
|
||||
options.ocr.model_downloads = crate::vision::ModelDownloadPolicy::Offline;
|
||||
}
|
||||
Ok(options)
|
||||
}
|
||||
|
||||
fn page_content_source_str(source: crate::vision::PageContentSource) -> String {
|
||||
match source {
|
||||
crate::vision::PageContentSource::Native => "native".into(),
|
||||
crate::vision::PageContentSource::Ocr => "ocr".into(),
|
||||
crate::vision::PageContentSource::Fused => "fused".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_ocr_result(result: crate::vision::OcrPdfResult) -> PyOcrPdfResult {
|
||||
PyOcrPdfResult {
|
||||
markdown: result.markdown,
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|page| {
|
||||
let provenance = page.provenance;
|
||||
PyOcrPageResult {
|
||||
page_number: page.page_number,
|
||||
markdown: page.markdown,
|
||||
provenance: PyOcrPageProvenance {
|
||||
page_number: provenance.page_number,
|
||||
source: page_content_source_str(provenance.source),
|
||||
ocr_model: provenance.ocr_model.map(|model| PyOcrModelIdentity {
|
||||
name: model.name,
|
||||
revision: model.revision,
|
||||
}),
|
||||
render_dpi: provenance.render_dpi,
|
||||
ocr_confidence: provenance.ocr_confidence,
|
||||
timings: PyOcrTimings {
|
||||
render_ms: provenance.timings.render_ms,
|
||||
ocr_ms: provenance.timings.ocr_ms,
|
||||
assembly_ms: provenance.timings.assembly_ms,
|
||||
},
|
||||
warnings: provenance.warnings,
|
||||
hosted_recommended: provenance.hosted_recommended,
|
||||
},
|
||||
}
|
||||
})
|
||||
.collect(),
|
||||
page_count: result.page_count,
|
||||
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
|
||||
pages_routed_to_ocr: result.pages_routed_to_ocr,
|
||||
pages_recommending_hosted: result.pages_recommending_hosted,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
is_complex: result.is_complex,
|
||||
processing_time_ms: result.processing_time_ms,
|
||||
render_time_ms: result.render_time_ms,
|
||||
ocr_time_ms: result.ocr_time_ms,
|
||||
}
|
||||
}
|
||||
|
||||
fn item_type_str(t: &ItemType) -> String {
|
||||
match t {
|
||||
ItemType::Text => "text".into(),
|
||||
@@ -698,99 +502,6 @@ fn process_pdf_bytes(data: &[u8], pages: Option<Vec<u32>>) -> PyResult<PyPdfResu
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Process a PDF file through native extraction and selective OCR.
|
||||
///
|
||||
/// OCR defaults to ``auto`` and only initializes its external runtime and
|
||||
/// model when native quality signals route at least one page. Page numbers
|
||||
/// are 1-indexed. The GIL is released for the complete processing call.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (
|
||||
path,
|
||||
*,
|
||||
mode="auto",
|
||||
page_numbers=None,
|
||||
password=None,
|
||||
dpi=150.0,
|
||||
minimum_confidence=0.0,
|
||||
hosted_recommendation_confidence=0.5,
|
||||
model_directory=None,
|
||||
offline=false
|
||||
))]
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn process_pdf_with_ocr(
|
||||
py: Python<'_>,
|
||||
path: String,
|
||||
mode: &str,
|
||||
page_numbers: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
dpi: f32,
|
||||
minimum_confidence: f32,
|
||||
hosted_recommendation_confidence: f32,
|
||||
model_directory: Option<String>,
|
||||
offline: bool,
|
||||
) -> PyResult<PyOcrPdfResult> {
|
||||
let options = build_ocr_options(PythonOcrOptions {
|
||||
mode: mode.to_string(),
|
||||
page_numbers,
|
||||
password,
|
||||
dpi,
|
||||
minimum_confidence,
|
||||
hosted_recommendation_confidence,
|
||||
model_directory,
|
||||
offline,
|
||||
})?;
|
||||
let result = py
|
||||
.allow_threads(move || crate::vision::process_pdf_with_ocr(path, options))
|
||||
.map_err(|error| PyValueError::new_err(error.to_string()))?;
|
||||
Ok(to_py_ocr_result(result))
|
||||
}
|
||||
|
||||
/// Process PDF bytes through native extraction and selective OCR.
|
||||
///
|
||||
/// See [`process_pdf_with_ocr`] for options and result semantics.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (
|
||||
data,
|
||||
*,
|
||||
mode="auto",
|
||||
page_numbers=None,
|
||||
password=None,
|
||||
dpi=150.0,
|
||||
minimum_confidence=0.0,
|
||||
hosted_recommendation_confidence=0.5,
|
||||
model_directory=None,
|
||||
offline=false
|
||||
))]
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn process_pdf_with_ocr_bytes(
|
||||
py: Python<'_>,
|
||||
data: &[u8],
|
||||
mode: &str,
|
||||
page_numbers: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
dpi: f32,
|
||||
minimum_confidence: f32,
|
||||
hosted_recommendation_confidence: f32,
|
||||
model_directory: Option<String>,
|
||||
offline: bool,
|
||||
) -> PyResult<PyOcrPdfResult> {
|
||||
let options = build_ocr_options(PythonOcrOptions {
|
||||
mode: mode.to_string(),
|
||||
page_numbers,
|
||||
password,
|
||||
dpi,
|
||||
minimum_confidence,
|
||||
hosted_recommendation_confidence,
|
||||
model_directory,
|
||||
offline,
|
||||
})?;
|
||||
let data = data.to_vec();
|
||||
let result = py
|
||||
.allow_threads(move || crate::vision::process_pdf_with_ocr_mem(&data, options))
|
||||
.map_err(|error| PyValueError::new_err(error.to_string()))?;
|
||||
Ok(to_py_ocr_result(result))
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
#[pyfunction]
|
||||
fn detect_pdf(path: &str) -> PyResult<PyPdfResult> {
|
||||
@@ -993,11 +704,6 @@ fn extract_structure_elements_bytes(
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPdfResult>()?;
|
||||
m.add_class::<PyPageOcrReasons>()?;
|
||||
m.add_class::<PyOcrModelIdentity>()?;
|
||||
m.add_class::<PyOcrTimings>()?;
|
||||
m.add_class::<PyOcrPageProvenance>()?;
|
||||
m.add_class::<PyOcrPageResult>()?;
|
||||
m.add_class::<PyOcrPdfResult>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyStructureElement>()?;
|
||||
@@ -1007,8 +713,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPagesExtractionResult>()?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_with_ocr, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_with_ocr_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(classify_pdf, m)?)?;
|
||||
|
||||
@@ -442,16 +442,6 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
fn is_footnote_row(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
|
||||
// Japanese documents commonly use the reference mark followed by an
|
||||
// ASCII or full-width number (for example `※1` / `※1`). These rows often
|
||||
// sit immediately below a wide table and must not be merged into its last
|
||||
// data row as wrapped first-column content.
|
||||
if let Some(rest) = trimmed.strip_prefix('※') {
|
||||
return rest.chars().next().is_some_and(|character| {
|
||||
character.is_ascii_digit() || ('0'..='9').contains(&character)
|
||||
});
|
||||
}
|
||||
|
||||
// Check for common footnote patterns
|
||||
// (1), (2), etc.
|
||||
if trimmed.starts_with('(') && trimmed.len() >= 2 {
|
||||
@@ -513,13 +503,6 @@ mod tests {
|
||||
assert!(is_footnote_row("NOTES: uppercase"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_footnote_row_reference_mark_number() {
|
||||
assert!(is_footnote_row("※1 explanation"));
|
||||
assert!(is_footnote_row("※1 説明"));
|
||||
assert!(!is_footnote_row("※ general marker"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_footnote_row_plain_text_false() {
|
||||
assert!(!is_footnote_row("Regular cell text"));
|
||||
|
||||
+2
-2
@@ -604,7 +604,7 @@ mod tests {
|
||||
let items: Vec<(usize, &TextItem)> = vec![];
|
||||
assert_eq!(
|
||||
find_column_boundaries(&items, TableDetectionMode::SmallFont),
|
||||
Vec::<f32>::new()
|
||||
vec![]
|
||||
);
|
||||
}
|
||||
|
||||
@@ -661,7 +661,7 @@ mod tests {
|
||||
#[test]
|
||||
fn test_find_row_boundaries_empty() {
|
||||
let items: Vec<(usize, &TextItem)> = vec![];
|
||||
assert_eq!(find_row_boundaries(&items), Vec::<f32>::new());
|
||||
assert_eq!(find_row_boundaries(&items), vec![]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
+152
-3
@@ -1,4 +1,4 @@
|
||||
//! Public contracts between rendering, OCR, and orchestration.
|
||||
//! Public contracts between rendering, OCR, layout, and orchestration.
|
||||
|
||||
use std::error::Error;
|
||||
use std::path::PathBuf;
|
||||
@@ -18,6 +18,19 @@ pub enum OcrMode {
|
||||
Force,
|
||||
}
|
||||
|
||||
/// Resource/quality profile for the OCR engine.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrProfile {
|
||||
/// Lowest latency and memory footprint.
|
||||
Edge,
|
||||
/// OCR-oriented balance of quality and CPU cost.
|
||||
#[default]
|
||||
Balanced,
|
||||
/// Highest quality within the lightweight model family.
|
||||
Quality,
|
||||
}
|
||||
|
||||
/// Controls whether missing model artifacts may be fetched.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
@@ -34,8 +47,12 @@ pub enum ModelDownloadPolicy {
|
||||
pub struct OcrOptions {
|
||||
/// Page-level routing behavior.
|
||||
pub mode: OcrMode,
|
||||
/// Local quality/resource profile.
|
||||
pub profile: OcrProfile,
|
||||
/// Drop recognition spans below this confidence threshold.
|
||||
pub minimum_confidence: f32,
|
||||
/// Optional language hints understood by the selected engine.
|
||||
pub languages: Vec<String>,
|
||||
/// Optional directory containing an offline model set.
|
||||
pub model_directory: Option<PathBuf>,
|
||||
/// Whether a missing pinned artifact may be downloaded.
|
||||
@@ -46,7 +63,9 @@ impl Default for OcrOptions {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
mode: OcrMode::Off,
|
||||
profile: OcrProfile::Balanced,
|
||||
minimum_confidence: 0.0,
|
||||
languages: Vec::new(),
|
||||
model_directory: None,
|
||||
model_downloads: ModelDownloadPolicy::IfMissing,
|
||||
}
|
||||
@@ -65,12 +84,24 @@ impl OcrOptions {
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the local resource/quality profile.
|
||||
pub fn profile(mut self, profile: OcrProfile) -> Self {
|
||||
self.profile = profile;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the minimum accepted recognition confidence.
|
||||
pub fn minimum_confidence(mut self, minimum_confidence: f32) -> Self {
|
||||
self.minimum_confidence = minimum_confidence;
|
||||
self
|
||||
}
|
||||
|
||||
/// Replaces the language hints passed to the OCR engine.
|
||||
pub fn languages(mut self, languages: impl IntoIterator<Item = impl Into<String>>) -> Self {
|
||||
self.languages = languages.into_iter().map(Into::into).collect();
|
||||
self
|
||||
}
|
||||
|
||||
/// Uses an explicit model directory, suitable for offline packaging.
|
||||
pub fn model_directory(mut self, directory: impl Into<PathBuf>) -> Self {
|
||||
self.model_directory = Some(directory.into());
|
||||
@@ -84,6 +115,55 @@ impl OcrOptions {
|
||||
}
|
||||
}
|
||||
|
||||
/// Configuration for an optional learned layout engine.
|
||||
///
|
||||
/// Layout inference is disabled by default. Existing deterministic layout,
|
||||
/// table, and Markdown logic remains the assembly path when this is disabled.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct LayoutOptions {
|
||||
/// Whether the learned layout extension may run.
|
||||
pub enabled: bool,
|
||||
/// Drop layout regions below this confidence threshold.
|
||||
pub minimum_confidence: f32,
|
||||
/// Optional directory containing an offline layout model set.
|
||||
pub model_directory: Option<PathBuf>,
|
||||
}
|
||||
|
||||
impl Default for LayoutOptions {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
enabled: false,
|
||||
minimum_confidence: 0.0,
|
||||
model_directory: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl LayoutOptions {
|
||||
/// Creates layout options with learned layout disabled.
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Enables or disables learned layout inference.
|
||||
pub fn enabled(mut self, enabled: bool) -> Self {
|
||||
self.enabled = enabled;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the minimum accepted region confidence.
|
||||
pub fn minimum_confidence(mut self, minimum_confidence: f32) -> Self {
|
||||
self.minimum_confidence = minimum_confidence;
|
||||
self
|
||||
}
|
||||
|
||||
/// Uses an explicit layout model directory.
|
||||
pub fn model_directory(mut self, directory: impl Into<PathBuf>) -> Self {
|
||||
self.model_directory = Some(directory.into());
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// A point in bitmap space, measured from the top-left in pixels.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq)]
|
||||
pub struct ImagePoint {
|
||||
@@ -150,7 +230,7 @@ pub struct OcrSpan {
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct OcrPage {
|
||||
/// 1-indexed PDF page number.
|
||||
pub page_number: u32,
|
||||
pub page: u32,
|
||||
/// Positioned recognition spans.
|
||||
pub spans: Vec<OcrSpan>,
|
||||
/// Mean confidence across accepted spans, when available.
|
||||
@@ -163,6 +243,54 @@ pub struct OcrPage {
|
||||
pub warnings: Vec<String>,
|
||||
}
|
||||
|
||||
/// Normalized semantic class emitted by a learned layout engine.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum LayoutRegionKind {
|
||||
/// Body or other prose text.
|
||||
Text,
|
||||
/// Document heading or title.
|
||||
Heading,
|
||||
/// Table region.
|
||||
Table,
|
||||
/// Figure/image region.
|
||||
Figure,
|
||||
/// Figure or table caption.
|
||||
Caption,
|
||||
/// Header/footer/page furniture.
|
||||
Furniture,
|
||||
/// Model-specific class retained without changing the common taxonomy.
|
||||
Other(String),
|
||||
}
|
||||
|
||||
/// One learned layout region in bitmap coordinates.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct LayoutRegion {
|
||||
/// Normalized semantic class.
|
||||
pub kind: LayoutRegionKind,
|
||||
/// Region polygon in the original rendered page's pixel space.
|
||||
pub polygon: ImageQuad,
|
||||
/// Model confidence in the inclusive range 0–1.
|
||||
pub confidence: f32,
|
||||
/// Optional model-provided reading-order position.
|
||||
pub reading_order: Option<u32>,
|
||||
}
|
||||
|
||||
/// Learned layout output for one 1-indexed page.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct LayoutPage {
|
||||
/// 1-indexed PDF page number.
|
||||
pub page: u32,
|
||||
/// Semantic regions.
|
||||
pub regions: Vec<LayoutRegion>,
|
||||
/// Exact model identity used for this result.
|
||||
pub model: ModelIdentity,
|
||||
/// Layout inference wall time for this page.
|
||||
pub processing_time_ms: u64,
|
||||
/// Non-fatal engine warnings.
|
||||
pub warnings: Vec<String>,
|
||||
}
|
||||
|
||||
/// How final page content was sourced.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
@@ -182,6 +310,8 @@ pub struct VisionTimings {
|
||||
pub render_ms: u64,
|
||||
/// OCR wall time.
|
||||
pub ocr_ms: u64,
|
||||
/// Optional learned layout wall time.
|
||||
pub layout_ms: u64,
|
||||
/// Native/OCR fusion and assembly wall time.
|
||||
pub assembly_ms: u64,
|
||||
}
|
||||
@@ -190,11 +320,13 @@ pub struct VisionTimings {
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct PageProvenance {
|
||||
/// 1-indexed PDF page number.
|
||||
pub page_number: u32,
|
||||
pub page: u32,
|
||||
/// Final page-content source.
|
||||
pub source: PageContentSource,
|
||||
/// OCR model, when OCR ran.
|
||||
pub ocr_model: Option<ModelIdentity>,
|
||||
/// Learned layout model, when layout inference ran.
|
||||
pub layout_model: Option<ModelIdentity>,
|
||||
/// Render resolution used for local vision.
|
||||
pub render_dpi: Option<f32>,
|
||||
/// Mean accepted OCR confidence, when available.
|
||||
@@ -239,6 +371,23 @@ pub trait OcrEngine: Send + Sync {
|
||||
) -> Result<Vec<OcrPage>, Self::Error>;
|
||||
}
|
||||
|
||||
/// Optional learned semantic layout extension.
|
||||
pub trait LayoutEngine: Send + Sync {
|
||||
/// Engine-specific failure type.
|
||||
type Error: Error + Send + Sync + 'static;
|
||||
|
||||
/// Exact model identity used by this engine instance.
|
||||
fn model(&self) -> &ModelIdentity;
|
||||
|
||||
/// Analyzes rendered pages, optionally using their OCR spans.
|
||||
fn analyze(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
ocr: &[OcrPage],
|
||||
options: &LayoutOptions,
|
||||
) -> Result<Vec<LayoutPage>, Self::Error>;
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
@@ -1,84 +0,0 @@
|
||||
//! HTTPS acquisition for pinned local model artifacts.
|
||||
|
||||
use std::fmt;
|
||||
use std::io::Read;
|
||||
use std::time::Duration;
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{ModelArtifact, ModelDownloader};
|
||||
|
||||
/// Default end-to-end timeout for one model artifact request.
|
||||
pub const DEFAULT_MODEL_DOWNLOAD_TIMEOUT: Duration = Duration::from_secs(5 * 60);
|
||||
|
||||
/// Streaming HTTPS downloader used by lazy model resolution.
|
||||
#[derive(Clone)]
|
||||
pub struct HttpModelDownloader {
|
||||
agent: ureq::Agent,
|
||||
}
|
||||
|
||||
impl fmt::Debug for HttpModelDownloader {
|
||||
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
formatter
|
||||
.debug_struct("HttpModelDownloader")
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for HttpModelDownloader {
|
||||
fn default() -> Self {
|
||||
Self::new(DEFAULT_MODEL_DOWNLOAD_TIMEOUT)
|
||||
}
|
||||
}
|
||||
|
||||
impl HttpModelDownloader {
|
||||
/// Creates an HTTPS-only downloader with an end-to-end request timeout.
|
||||
pub fn new(timeout: Duration) -> Self {
|
||||
let config = ureq::Agent::config_builder()
|
||||
.https_only(true)
|
||||
.timeout_global(Some(timeout))
|
||||
.user_agent(concat!("pdf-inspector/", env!("CARGO_PKG_VERSION")))
|
||||
.build();
|
||||
Self {
|
||||
agent: ureq::Agent::new_with_config(config),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ModelDownloader for HttpModelDownloader {
|
||||
type Error = HttpModelDownloadError;
|
||||
|
||||
fn open(&self, artifact: &ModelArtifact) -> Result<Box<dyn Read + Send>, Self::Error> {
|
||||
let response = self.agent.get(artifact.url).call()?;
|
||||
if let Some(actual) = response.body().content_length() {
|
||||
if actual != artifact.size {
|
||||
return Err(HttpModelDownloadError::ContentLength {
|
||||
expected: artifact.size,
|
||||
actual,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// One extra byte lets ModelStore report an exact size mismatch while
|
||||
// preventing a malicious or broken server from filling the disk.
|
||||
let limit = artifact.size.saturating_add(1);
|
||||
Ok(Box::new(response.into_body().into_reader().take(limit)))
|
||||
}
|
||||
}
|
||||
|
||||
/// Failures before a response stream reaches [`super::ModelStore`].
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum HttpModelDownloadError {
|
||||
/// DNS, TLS, redirect, HTTP status, or response-stream setup failed.
|
||||
#[error(transparent)]
|
||||
Request(#[from] ureq::Error),
|
||||
/// The server declared a size that disagrees with the pinned manifest.
|
||||
#[error("server declared {actual} bytes; manifest requires {expected}")]
|
||||
ContentLength {
|
||||
/// Pinned artifact size.
|
||||
expected: u64,
|
||||
/// Server-declared size.
|
||||
actual: u64,
|
||||
},
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
+8
-36
@@ -3,62 +3,34 @@
|
||||
//! The existing lopdf extractor remains the default path. Native page
|
||||
//! rendering is available only with the `render-pdfium` feature. Engine
|
||||
//! contracts are available with `vision`, while checksum-verified model
|
||||
//! resolution is a separate `model-cache` feature. The `ocr-oar` feature adds
|
||||
//! a CPU PP-OCRv6 Small implementation of [`OcrEngine`]. These remain separate
|
||||
//! so browser WASM, text-only consumers, and renderer-only users take on no
|
||||
//! model-management or inference dependencies.
|
||||
//! resolution is a separate `model-cache` feature. These remain separate so
|
||||
//! browser WASM, text-only consumers, and renderer-only users take on no model
|
||||
//! management dependencies.
|
||||
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod contracts;
|
||||
#[cfg(all(feature = "model-download", not(target_arch = "wasm32")))]
|
||||
mod download;
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod fusion;
|
||||
#[cfg(all(feature = "model-cache", not(target_arch = "wasm32")))]
|
||||
mod models;
|
||||
#[cfg(all(feature = "ocr-oar", not(target_arch = "wasm32")))]
|
||||
mod oar;
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
mod pipeline;
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod render;
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod routing;
|
||||
|
||||
#[cfg(all(feature = "render-pdfium", not(target_arch = "wasm32")))]
|
||||
mod pdfium;
|
||||
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use contracts::{
|
||||
ImagePoint, ImageQuad, ModelDownloadPolicy, ModelIdentity, OcrEngine, OcrMode, OcrOptions,
|
||||
OcrPage, OcrSpan, PageContentSource, PageProvenance, PageRenderer, VisionTimings,
|
||||
};
|
||||
#[cfg(all(feature = "model-download", not(target_arch = "wasm32")))]
|
||||
pub use download::{HttpModelDownloadError, HttpModelDownloader, DEFAULT_MODEL_DOWNLOAD_TIMEOUT};
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use fusion::{
|
||||
fuse_ocr_pages, ocr_page_to_markdown, FusedPageMarkdown, FusedPages, OcrFusionError,
|
||||
OcrFusionOptions,
|
||||
ImagePoint, ImageQuad, LayoutEngine, LayoutOptions, LayoutPage, LayoutRegion, LayoutRegionKind,
|
||||
ModelDownloadPolicy, ModelIdentity, OcrEngine, OcrMode, OcrOptions, OcrPage, OcrProfile,
|
||||
OcrSpan, PageContentSource, PageProvenance, PageRenderer, VisionTimings,
|
||||
};
|
||||
#[cfg(all(feature = "model-cache", not(target_arch = "wasm32")))]
|
||||
pub use models::{
|
||||
ModelAcquireError, ModelArtifact, ModelArtifactKind, ModelDownloader, ModelManifest,
|
||||
ModelPaths, ModelStore, ModelStoreError, PP_OCR_V6_SMALL,
|
||||
};
|
||||
#[cfg(all(feature = "ocr-oar", not(target_arch = "wasm32")))]
|
||||
pub use oar::{OarOcrEngine, OarOcrError, ONNX_RUNTIME_LIBRARY_ENV};
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
pub use pipeline::{
|
||||
process_pdf_with_ocr, process_pdf_with_ocr_mem, OcrPdfOptions, OcrPdfResult, OcrPipelineError,
|
||||
ModelArtifact, ModelArtifactKind, ModelManifest, ModelPaths, ModelStore, ModelStoreError,
|
||||
PP_OCR_V6_SMALL,
|
||||
};
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use render::{
|
||||
PagePoint, PageTransform, RenderBufferError, RenderOptions, RenderPixelFormat, RenderedPage,
|
||||
DEFAULT_RENDER_DPI,
|
||||
};
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use routing::{
|
||||
route_ocr_pages, run_ocr_pages, OcrRoutingError, OcrRun, OcrRunError, RoutedOcrPage,
|
||||
};
|
||||
|
||||
#[cfg(all(feature = "render-pdfium", not(target_arch = "wasm32")))]
|
||||
|
||||
+60
-298
@@ -12,7 +12,7 @@ use fs2::FileExt;
|
||||
use sha2::{Digest, Sha256};
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{ModelDownloadPolicy, OcrOptions};
|
||||
use super::OcrOptions;
|
||||
|
||||
/// Environment variable overriding the default local model cache.
|
||||
pub const MODEL_CACHE_ENV: &str = "PDF_INSPECTOR_MODEL_CACHE";
|
||||
@@ -138,19 +138,6 @@ pub struct ModelStore {
|
||||
override_root: Option<PathBuf>,
|
||||
}
|
||||
|
||||
/// Streaming source for a pinned model artifact.
|
||||
///
|
||||
/// Implementations may use HTTP, an object store, or an application-owned
|
||||
/// package manager. [`ModelStore`] remains responsible for locking, atomic
|
||||
/// installation, exact size validation, and SHA-256 verification.
|
||||
pub trait ModelDownloader: Send + Sync {
|
||||
/// Downloader-specific failure type.
|
||||
type Error: std::error::Error + Send + Sync + 'static;
|
||||
|
||||
/// Opens a streaming reader for one manifest artifact.
|
||||
fn open(&self, artifact: &ModelArtifact) -> Result<Box<dyn Read + Send>, Self::Error>;
|
||||
}
|
||||
|
||||
impl ModelStore {
|
||||
/// Creates a model store rooted at an explicit cache directory.
|
||||
pub fn new(cache_root: impl Into<PathBuf>) -> Self {
|
||||
@@ -190,17 +177,16 @@ impl ModelStore {
|
||||
&self.cache_root
|
||||
}
|
||||
|
||||
/// Effective directory containing one manifest's artifacts.
|
||||
pub(crate) fn model_root(&self, manifest: &ModelManifest) -> PathBuf {
|
||||
self.override_root
|
||||
.clone()
|
||||
.unwrap_or_else(|| self.manifest_cache_root(manifest))
|
||||
}
|
||||
|
||||
/// Validates and resolves every required artifact.
|
||||
pub fn resolve(&self, manifest: &ModelManifest) -> Result<ModelPaths, ModelStoreError> {
|
||||
validate_manifest(manifest)?;
|
||||
let root = self.model_root(manifest);
|
||||
let managed_root;
|
||||
let root = if let Some(root) = self.override_root.as_deref() {
|
||||
root
|
||||
} else {
|
||||
managed_root = self.manifest_cache_root(manifest);
|
||||
managed_root.as_path()
|
||||
};
|
||||
|
||||
let mut artifacts = BTreeMap::new();
|
||||
for artifact in manifest.artifacts {
|
||||
@@ -215,54 +201,6 @@ impl ModelStore {
|
||||
})
|
||||
}
|
||||
|
||||
/// Resolves a complete model set, fetching only missing or invalid managed
|
||||
/// cache artifacts when policy permits.
|
||||
///
|
||||
/// Explicit [`OcrOptions::model_directory`] overrides are never mutated or
|
||||
/// supplemented from the network. This method also avoids all downloader
|
||||
/// calls when the cache is already valid or downloads are offline.
|
||||
pub fn resolve_or_download<D: ModelDownloader>(
|
||||
&self,
|
||||
manifest: &ModelManifest,
|
||||
policy: ModelDownloadPolicy,
|
||||
downloader: &D,
|
||||
) -> Result<ModelPaths, ModelAcquireError<D::Error>> {
|
||||
validate_manifest(manifest).map_err(ModelAcquireError::Store)?;
|
||||
match self.resolve(manifest) {
|
||||
Ok(paths) => return Ok(paths),
|
||||
Err(source) if self.override_root.is_some() => {
|
||||
return Err(ModelAcquireError::ExplicitDirectoryIncomplete { source });
|
||||
}
|
||||
Err(source) if !source.permits_download_recovery() => {
|
||||
return Err(ModelAcquireError::Store(source));
|
||||
}
|
||||
Err(source) if policy == ModelDownloadPolicy::Offline => {
|
||||
return Err(ModelAcquireError::DownloadsDisabled { source });
|
||||
}
|
||||
Err(_) => {}
|
||||
}
|
||||
|
||||
let root = self.manifest_cache_root(manifest);
|
||||
for artifact in manifest.artifacts {
|
||||
let _lock = lock_artifact(&root, artifact).map_err(ModelAcquireError::Store)?;
|
||||
match verify_artifact(&root.join(artifact.filename), artifact) {
|
||||
Ok(()) => continue,
|
||||
Err(source) if source.permits_download_recovery() => {}
|
||||
Err(source) => return Err(ModelAcquireError::Store(source)),
|
||||
}
|
||||
let reader =
|
||||
downloader
|
||||
.open(artifact)
|
||||
.map_err(|source| ModelAcquireError::Download {
|
||||
kind: artifact.kind,
|
||||
url: artifact.url,
|
||||
source,
|
||||
})?;
|
||||
install_locked(&root, artifact, reader).map_err(ModelAcquireError::Store)?;
|
||||
}
|
||||
self.resolve(manifest).map_err(ModelAcquireError::Store)
|
||||
}
|
||||
|
||||
/// Atomically installs one artifact from a reader after validating its
|
||||
/// exact size and SHA-256 digest.
|
||||
///
|
||||
@@ -281,8 +219,58 @@ impl ModelStore {
|
||||
.find(|artifact| artifact.kind == kind)
|
||||
.ok_or(ModelStoreError::ArtifactNotInManifest { kind })?;
|
||||
let root = self.manifest_cache_root(manifest);
|
||||
let _lock = lock_artifact(&root, artifact)?;
|
||||
install_locked(&root, artifact, &mut reader)
|
||||
fs::create_dir_all(&root).map_err(|source| ModelStoreError::Io {
|
||||
path: root.clone(),
|
||||
source,
|
||||
})?;
|
||||
|
||||
let target = root.join(artifact.filename);
|
||||
let lock_path = root.join(format!(".{}.lock", artifact.filename));
|
||||
let lock = OpenOptions::new()
|
||||
.create(true)
|
||||
.read(true)
|
||||
.write(true)
|
||||
.truncate(false)
|
||||
.open(&lock_path)
|
||||
.map_err(|source| ModelStoreError::Io {
|
||||
path: lock_path.clone(),
|
||||
source,
|
||||
})?;
|
||||
FileExt::lock_exclusive(&lock).map_err(|source| ModelStoreError::Io {
|
||||
path: lock_path,
|
||||
source,
|
||||
})?;
|
||||
|
||||
if verify_artifact(&target, artifact).is_ok() {
|
||||
return Ok(target);
|
||||
}
|
||||
|
||||
sweep_stale_install_files(&root, artifact.filename)?;
|
||||
let (temporary, mut output) = create_temporary_file(&root, artifact.filename)?;
|
||||
let result = (|| {
|
||||
let mut limited = reader.by_ref().take(artifact.size.saturating_add(1));
|
||||
let (size, digest) =
|
||||
copy_and_hash(&mut limited, &mut output).map_err(|source| ModelStoreError::Io {
|
||||
path: temporary.clone(),
|
||||
source,
|
||||
})?;
|
||||
output.sync_all().map_err(|source| ModelStoreError::Io {
|
||||
path: temporary.clone(),
|
||||
source,
|
||||
})?;
|
||||
validate_size_and_hash(artifact, size, &digest, &temporary)?;
|
||||
|
||||
replace_file_atomic(&temporary, &target).map_err(|source| ModelStoreError::Io {
|
||||
path: target.clone(),
|
||||
source,
|
||||
})?;
|
||||
Ok(target.clone())
|
||||
})();
|
||||
|
||||
if result.is_err() {
|
||||
let _ = fs::remove_file(&temporary);
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
fn manifest_cache_root(&self, manifest: &ModelManifest) -> PathBuf {
|
||||
@@ -290,43 +278,6 @@ impl ModelStore {
|
||||
}
|
||||
}
|
||||
|
||||
/// Failures while resolving or lazily acquiring a model set.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum ModelAcquireError<E>
|
||||
where
|
||||
E: std::error::Error + Send + Sync + 'static,
|
||||
{
|
||||
/// Manifest validation, cache verification, or installation failed.
|
||||
#[error(transparent)]
|
||||
Store(#[from] ModelStoreError),
|
||||
/// A managed-cache artifact could not be downloaded.
|
||||
#[error("failed to download {kind:?} from {url}: {source}")]
|
||||
Download {
|
||||
/// Artifact role.
|
||||
kind: ModelArtifactKind,
|
||||
/// Pinned source URL.
|
||||
url: &'static str,
|
||||
/// Downloader failure.
|
||||
#[source]
|
||||
source: E,
|
||||
},
|
||||
/// Offline policy prevented acquisition of an unavailable artifact.
|
||||
#[error("model artifacts are unavailable and downloads are disabled: {source}")]
|
||||
DownloadsDisabled {
|
||||
/// Original verification failure.
|
||||
#[source]
|
||||
source: ModelStoreError,
|
||||
},
|
||||
/// An explicit package-managed directory was incomplete or invalid.
|
||||
#[error("explicit model directory is incomplete or invalid: {source}")]
|
||||
ExplicitDirectoryIncomplete {
|
||||
/// Original verification failure.
|
||||
#[source]
|
||||
source: ModelStoreError,
|
||||
},
|
||||
}
|
||||
|
||||
/// Failures while validating or installing model artifacts.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
@@ -382,17 +333,6 @@ pub enum ModelStoreError {
|
||||
},
|
||||
}
|
||||
|
||||
impl ModelStoreError {
|
||||
fn permits_download_recovery(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Self::MissingArtifact { .. }
|
||||
| Self::SizeMismatch { .. }
|
||||
| Self::ChecksumMismatch { .. }
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_manifest(manifest: &ModelManifest) -> Result<(), ModelStoreError> {
|
||||
if manifest.schema_version != 1 {
|
||||
return Err(ModelStoreError::InvalidManifest(format!(
|
||||
@@ -463,67 +403,6 @@ fn is_single_normal_path_component(value: &str) -> bool {
|
||||
matches!(components.next(), Some(Component::Normal(_))) && components.next().is_none()
|
||||
}
|
||||
|
||||
fn lock_artifact(root: &Path, artifact: &ModelArtifact) -> Result<File, ModelStoreError> {
|
||||
fs::create_dir_all(root).map_err(|source| ModelStoreError::Io {
|
||||
path: root.to_path_buf(),
|
||||
source,
|
||||
})?;
|
||||
let lock_path = root.join(format!(".{}.lock", artifact.filename));
|
||||
let lock = OpenOptions::new()
|
||||
.create(true)
|
||||
.read(true)
|
||||
.write(true)
|
||||
.truncate(false)
|
||||
.open(&lock_path)
|
||||
.map_err(|source| ModelStoreError::Io {
|
||||
path: lock_path.clone(),
|
||||
source,
|
||||
})?;
|
||||
FileExt::lock_exclusive(&lock).map_err(|source| ModelStoreError::Io {
|
||||
path: lock_path,
|
||||
source,
|
||||
})?;
|
||||
Ok(lock)
|
||||
}
|
||||
|
||||
fn install_locked(
|
||||
root: &Path,
|
||||
artifact: &ModelArtifact,
|
||||
mut reader: impl Read,
|
||||
) -> Result<PathBuf, ModelStoreError> {
|
||||
let target = root.join(artifact.filename);
|
||||
if verify_artifact(&target, artifact).is_ok() {
|
||||
return Ok(target);
|
||||
}
|
||||
|
||||
sweep_stale_install_files(root, artifact.filename)?;
|
||||
let (temporary, mut output) = create_temporary_file(root, artifact.filename)?;
|
||||
let result = (|| {
|
||||
let mut limited = reader.by_ref().take(artifact.size.saturating_add(1));
|
||||
let (size, digest) =
|
||||
copy_and_hash(&mut limited, &mut output).map_err(|source| ModelStoreError::Io {
|
||||
path: temporary.clone(),
|
||||
source,
|
||||
})?;
|
||||
output.sync_all().map_err(|source| ModelStoreError::Io {
|
||||
path: temporary.clone(),
|
||||
source,
|
||||
})?;
|
||||
validate_size_and_hash(artifact, size, &digest, &temporary)?;
|
||||
|
||||
replace_file_atomic(&temporary, &target).map_err(|source| ModelStoreError::Io {
|
||||
path: target.clone(),
|
||||
source,
|
||||
})?;
|
||||
Ok(target.clone())
|
||||
})();
|
||||
|
||||
if result.is_err() {
|
||||
let _ = fs::remove_file(&temporary);
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
fn sweep_stale_install_files(root: &Path, filename: &str) -> Result<(), ModelStoreError> {
|
||||
let prefix = format!(".{filename}.");
|
||||
for entry in fs::read_dir(root).map_err(|source| ModelStoreError::Io {
|
||||
@@ -722,34 +601,6 @@ mod tests {
|
||||
artifacts: TEST_ARTIFACTS,
|
||||
};
|
||||
|
||||
#[derive(Debug)]
|
||||
struct StaticDownloader {
|
||||
bytes: &'static [u8],
|
||||
requests: std::sync::Mutex<Vec<ModelArtifactKind>>,
|
||||
}
|
||||
|
||||
impl StaticDownloader {
|
||||
fn new(bytes: &'static [u8]) -> Self {
|
||||
Self {
|
||||
bytes,
|
||||
requests: std::sync::Mutex::new(Vec::new()),
|
||||
}
|
||||
}
|
||||
|
||||
fn request_count(&self) -> usize {
|
||||
self.requests.lock().unwrap().len()
|
||||
}
|
||||
}
|
||||
|
||||
impl ModelDownloader for StaticDownloader {
|
||||
type Error = io::Error;
|
||||
|
||||
fn open(&self, artifact: &ModelArtifact) -> Result<Box<dyn Read + Send>, Self::Error> {
|
||||
self.requests.lock().unwrap().push(artifact.kind);
|
||||
Ok(Box::new(io::Cursor::new(self.bytes)))
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pinned_pp_ocr_manifest_is_well_formed() {
|
||||
validate_manifest(&PP_OCR_V6_SMALL).unwrap();
|
||||
@@ -913,93 +764,4 @@ mod tests {
|
||||
assert_eq!(first, second);
|
||||
assert_eq!(fs::read(first).unwrap(), b"hello");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_or_download_fetches_once_then_reuses_verified_cache() {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let store = ModelStore::new(temp.path());
|
||||
let downloader = StaticDownloader::new(b"hello");
|
||||
|
||||
let first = store
|
||||
.resolve_or_download(&TEST_MANIFEST, ModelDownloadPolicy::IfMissing, &downloader)
|
||||
.unwrap();
|
||||
let second = store
|
||||
.resolve_or_download(&TEST_MANIFEST, ModelDownloadPolicy::IfMissing, &downloader)
|
||||
.unwrap();
|
||||
assert_eq!(first, second);
|
||||
assert_eq!(downloader.request_count(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn concurrent_resolve_downloads_once_under_the_artifact_lock() {
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let store = ModelStore::new(temp.path());
|
||||
let downloader = std::sync::Arc::new(StaticDownloader::new(b"hello"));
|
||||
let barrier = std::sync::Arc::new(std::sync::Barrier::new(3));
|
||||
|
||||
let handles: Vec<_> = (0..2)
|
||||
.map(|_| {
|
||||
let store = store.clone();
|
||||
let downloader = std::sync::Arc::clone(&downloader);
|
||||
let barrier = std::sync::Arc::clone(&barrier);
|
||||
std::thread::spawn(move || {
|
||||
barrier.wait();
|
||||
store.resolve_or_download(
|
||||
&TEST_MANIFEST,
|
||||
ModelDownloadPolicy::IfMissing,
|
||||
downloader.as_ref(),
|
||||
)
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
barrier.wait();
|
||||
for handle in handles {
|
||||
handle.join().unwrap().unwrap();
|
||||
}
|
||||
assert_eq!(downloader.request_count(), 1);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn cache_io_failures_do_not_trigger_downloads() {
|
||||
use std::os::unix::fs::symlink;
|
||||
|
||||
let temp = tempfile::tempdir().unwrap();
|
||||
let store = ModelStore::new(temp.path());
|
||||
let root = store.manifest_cache_root(&TEST_MANIFEST);
|
||||
fs::create_dir_all(&root).unwrap();
|
||||
symlink("hello.txt", root.join("hello.txt")).unwrap();
|
||||
let downloader = StaticDownloader::new(b"hello");
|
||||
|
||||
assert!(matches!(
|
||||
store.resolve_or_download(&TEST_MANIFEST, ModelDownloadPolicy::IfMissing, &downloader,),
|
||||
Err(ModelAcquireError::Store(ModelStoreError::Io { .. }))
|
||||
));
|
||||
assert_eq!(downloader.request_count(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_or_download_honors_offline_and_explicit_directory_boundaries() {
|
||||
let cache = tempfile::tempdir().unwrap();
|
||||
let override_dir = tempfile::tempdir().unwrap();
|
||||
let downloader = StaticDownloader::new(b"hello");
|
||||
|
||||
let offline = ModelStore::new(cache.path())
|
||||
.resolve_or_download(&TEST_MANIFEST, ModelDownloadPolicy::Offline, &downloader)
|
||||
.unwrap_err();
|
||||
assert!(matches!(
|
||||
offline,
|
||||
ModelAcquireError::DownloadsDisabled { .. }
|
||||
));
|
||||
|
||||
let explicit = ModelStore::new(cache.path())
|
||||
.override_root(override_dir.path())
|
||||
.resolve_or_download(&TEST_MANIFEST, ModelDownloadPolicy::IfMissing, &downloader)
|
||||
.unwrap_err();
|
||||
assert!(matches!(
|
||||
explicit,
|
||||
ModelAcquireError::ExplicitDirectoryIncomplete { .. }
|
||||
));
|
||||
assert_eq!(downloader.request_count(), 0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,496 +0,0 @@
|
||||
//! PP-OCRv6 Small implementation backed by OAR and ONNX Runtime.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::Instant;
|
||||
|
||||
use image::RgbImage;
|
||||
use oar_ocr::core::config::onnx::OrtSessionConfig;
|
||||
use oar_ocr::oarocr::{OAROCRBuilder, OAROCR};
|
||||
use oar_ocr::processors::BoundingBox;
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{
|
||||
ImagePoint, ImageQuad, ModelArtifactKind, ModelIdentity, ModelPaths, OcrEngine, OcrMode,
|
||||
OcrOptions, OcrPage, OcrSpan, RenderPixelFormat, RenderedPage,
|
||||
};
|
||||
|
||||
/// Environment variable selecting the ONNX Runtime shared library.
|
||||
pub const ONNX_RUNTIME_LIBRARY_ENV: &str = "ORT_DYLIB_PATH";
|
||||
|
||||
/// Failures while constructing or running the OAR OCR backend.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OarOcrError {
|
||||
/// A required file is missing from the resolved model set.
|
||||
#[error("resolved OCR model set is missing {kind:?}")]
|
||||
MissingModelArtifact {
|
||||
/// Missing artifact role.
|
||||
kind: ModelArtifactKind,
|
||||
},
|
||||
/// OCR was invoked while the caller explicitly disabled it.
|
||||
#[error("OCR is disabled; select Auto or Force before invoking the engine")]
|
||||
OcrDisabled,
|
||||
/// Confidence thresholds must match the normalized engine output range.
|
||||
#[error("minimum OCR confidence must be finite and between 0 and 1, got {value}")]
|
||||
InvalidMinimumConfidence {
|
||||
/// Invalid threshold.
|
||||
value: f32,
|
||||
},
|
||||
/// Bitmap dimension arithmetic exceeded the host address space.
|
||||
#[error("rendered page {page} bitmap dimensions overflow the host address space")]
|
||||
ImageSizeOverflow {
|
||||
/// 1-indexed page number.
|
||||
page: u32,
|
||||
},
|
||||
/// A validated renderer buffer could not be represented as an RGB image.
|
||||
#[error("rendered page {page} could not be converted to an RGB image")]
|
||||
InvalidImageBuffer {
|
||||
/// 1-indexed page number.
|
||||
page: u32,
|
||||
},
|
||||
/// The external ONNX Runtime shared library could not be loaded.
|
||||
#[error(
|
||||
"failed to load ONNX Runtime from {path}; install a compatible ONNX Runtime shared library or set ORT_DYLIB_PATH to its path: {source}"
|
||||
)]
|
||||
OnnxRuntimeLoad {
|
||||
/// Requested shared-library path or platform library name.
|
||||
path: PathBuf,
|
||||
/// Dynamic-loader failure.
|
||||
#[source]
|
||||
source: ort::LoadDynamicError,
|
||||
},
|
||||
/// OAR returned no result for a submitted page.
|
||||
#[error("OAR returned no result for rendered page {page}")]
|
||||
MissingPageResult {
|
||||
/// 1-indexed page number.
|
||||
page: u32,
|
||||
},
|
||||
/// OAR or ONNX Runtime rejected the models or failed during inference.
|
||||
#[error(transparent)]
|
||||
Backend(#[from] oar_ocr::core::OCRError),
|
||||
}
|
||||
|
||||
/// CPU PP-OCRv6 Small engine using OAR's detection and recognition pipeline.
|
||||
///
|
||||
/// Construction accepts only [`ModelPaths`] that have already passed
|
||||
/// pdf-inspector's manifest size and SHA-256 verification. OAR's independent
|
||||
/// model auto-download feature is deliberately not enabled.
|
||||
#[derive(Debug)]
|
||||
pub struct OarOcrEngine {
|
||||
pipeline: OAROCR,
|
||||
model: ModelIdentity,
|
||||
}
|
||||
|
||||
impl OarOcrEngine {
|
||||
/// Loads PP-OCRv6 Small from a resolved, verified model set.
|
||||
pub fn from_models(models: &ModelPaths) -> Result<Self, OarOcrError> {
|
||||
load_onnx_runtime()?;
|
||||
let detection = required_model(models, ModelArtifactKind::TextDetection)?;
|
||||
let recognition = required_model(models, ModelArtifactKind::TextRecognition)?;
|
||||
let dictionary = required_model(models, ModelArtifactKind::CharacterDictionary)?;
|
||||
|
||||
let pipeline = OAROCRBuilder::new(detection, recognition, dictionary)
|
||||
.ort_session(ocr_session_config())
|
||||
// Document line crops often have very different widths. Keeping
|
||||
// CPU recognition batches at one avoids padding every crop to the
|
||||
// widest line, reducing both inference work and peak memory.
|
||||
.region_batch_size(1)
|
||||
.build()?;
|
||||
let model = ModelIdentity::new(models.manifest_id(), models.revision());
|
||||
Ok(Self { pipeline, model })
|
||||
}
|
||||
|
||||
fn recognize_page(
|
||||
&self,
|
||||
page: &RenderedPage,
|
||||
options: &OcrOptions,
|
||||
) -> Result<OcrPage, OarOcrError> {
|
||||
let started = Instant::now();
|
||||
let image = rendered_page_to_rgb(page)?;
|
||||
let result = self
|
||||
.pipeline
|
||||
.predict(vec![image])?
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or(OarOcrError::MissingPageResult { page: page.page() })?;
|
||||
|
||||
let mut spans = Vec::with_capacity(result.text_regions.len());
|
||||
let mut invalid_geometry = 0usize;
|
||||
let mut missing_recognition = 0usize;
|
||||
for region in result.text_regions {
|
||||
let (Some(text), Some(confidence)) = (region.text, region.confidence) else {
|
||||
missing_recognition += 1;
|
||||
continue;
|
||||
};
|
||||
if text.trim().is_empty() || !confidence.is_finite() {
|
||||
missing_recognition += 1;
|
||||
continue;
|
||||
}
|
||||
let confidence = confidence.clamp(0.0, 1.0);
|
||||
if confidence < options.minimum_confidence {
|
||||
continue;
|
||||
}
|
||||
|
||||
let polygon = region.dt_poly.as_ref().unwrap_or(®ion.bounding_box);
|
||||
let Some(polygon) = bounding_box_to_quad(polygon, page.width(), page.height()) else {
|
||||
invalid_geometry += 1;
|
||||
continue;
|
||||
};
|
||||
spans.push(OcrSpan {
|
||||
text: text.to_string(),
|
||||
polygon,
|
||||
confidence,
|
||||
orientation_degrees: region.orientation_angle,
|
||||
});
|
||||
}
|
||||
|
||||
let mut warnings = Vec::new();
|
||||
if missing_recognition > 0 {
|
||||
warnings.push(format!(
|
||||
"discarded {missing_recognition} regions without usable recognition output"
|
||||
));
|
||||
}
|
||||
if invalid_geometry > 0 {
|
||||
warnings.push(format!(
|
||||
"discarded {invalid_geometry} recognized regions with invalid geometry"
|
||||
));
|
||||
}
|
||||
|
||||
let mean_confidence = if spans.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(spans.iter().map(|span| span.confidence).sum::<f32>() / spans.len() as f32)
|
||||
};
|
||||
let processing_time_ms = u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX);
|
||||
|
||||
Ok(OcrPage {
|
||||
page_number: page.page(),
|
||||
spans,
|
||||
mean_confidence,
|
||||
model: self.model.clone(),
|
||||
processing_time_ms,
|
||||
warnings,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn ocr_session_config() -> OrtSessionConfig {
|
||||
let available = std::thread::available_parallelism()
|
||||
.map(std::num::NonZeroUsize::get)
|
||||
.unwrap_or(1);
|
||||
OrtSessionConfig::new()
|
||||
.with_intra_threads(available.min(4))
|
||||
.with_inter_threads(1)
|
||||
.with_parallel_execution(false)
|
||||
}
|
||||
|
||||
fn load_onnx_runtime() -> Result<(), OarOcrError> {
|
||||
let path = onnx_runtime_library_path();
|
||||
drop(
|
||||
ort::init_from(&path).map_err(|source| OarOcrError::OnnxRuntimeLoad {
|
||||
path: path.clone(),
|
||||
source,
|
||||
})?,
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub(crate) fn onnx_runtime_library_path() -> PathBuf {
|
||||
std::env::var_os(ONNX_RUNTIME_LIBRARY_ENV)
|
||||
.filter(|path| !path.is_empty())
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(default_onnx_runtime_library)
|
||||
}
|
||||
|
||||
fn default_onnx_runtime_library() -> PathBuf {
|
||||
#[cfg(target_os = "windows")]
|
||||
const NAME: &str = "onnxruntime.dll";
|
||||
#[cfg(any(target_os = "linux", target_os = "android", target_os = "freebsd"))]
|
||||
const NAME: &str = "libonnxruntime.so";
|
||||
#[cfg(any(target_os = "macos", target_os = "ios"))]
|
||||
const NAME: &str = "libonnxruntime.dylib";
|
||||
PathBuf::from(NAME)
|
||||
}
|
||||
|
||||
impl OcrEngine for OarOcrEngine {
|
||||
type Error = OarOcrError;
|
||||
|
||||
fn model(&self) -> &ModelIdentity {
|
||||
&self.model
|
||||
}
|
||||
|
||||
fn recognize(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
options: &OcrOptions,
|
||||
) -> Result<Vec<OcrPage>, Self::Error> {
|
||||
validate_options(options)?;
|
||||
|
||||
pages
|
||||
.iter()
|
||||
.map(|page| self.recognize_page(page, options))
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_options(options: &OcrOptions) -> Result<(), OarOcrError> {
|
||||
if options.mode == OcrMode::Off {
|
||||
return Err(OarOcrError::OcrDisabled);
|
||||
}
|
||||
if !options.minimum_confidence.is_finite() || !(0.0..=1.0).contains(&options.minimum_confidence)
|
||||
{
|
||||
return Err(OarOcrError::InvalidMinimumConfidence {
|
||||
value: options.minimum_confidence,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn required_model(
|
||||
models: &ModelPaths,
|
||||
kind: ModelArtifactKind,
|
||||
) -> Result<&std::path::Path, OarOcrError> {
|
||||
models
|
||||
.get(kind)
|
||||
.ok_or(OarOcrError::MissingModelArtifact { kind })
|
||||
}
|
||||
|
||||
fn rendered_page_to_rgb(page: &RenderedPage) -> Result<RgbImage, OarOcrError> {
|
||||
let width = usize::try_from(page.width())
|
||||
.map_err(|_| OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let height = usize::try_from(page.height())
|
||||
.map_err(|_| OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let output_len = width
|
||||
.checked_mul(height)
|
||||
.and_then(|pixels| pixels.checked_mul(3))
|
||||
.ok_or(OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let input_bpp = page.format().bytes_per_pixel();
|
||||
let active_input_row = width
|
||||
.checked_mul(input_bpp)
|
||||
.ok_or(OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let output_row = width
|
||||
.checked_mul(3)
|
||||
.ok_or(OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
|
||||
let mut rgb = vec![0u8; output_len];
|
||||
for row in 0..height {
|
||||
let input_start = row * page.stride();
|
||||
let input = &page.pixels()[input_start..input_start + active_input_row];
|
||||
let output_start = row * output_row;
|
||||
let output = &mut rgb[output_start..output_start + output_row];
|
||||
match page.format() {
|
||||
RenderPixelFormat::Rgb8 => output.copy_from_slice(input),
|
||||
RenderPixelFormat::Rgba8 => {
|
||||
for (rgba, rgb) in input.chunks_exact(4).zip(output.chunks_exact_mut(3)) {
|
||||
rgb.copy_from_slice(&rgba[..3]);
|
||||
}
|
||||
}
|
||||
RenderPixelFormat::Gray8 => {
|
||||
for (&gray, rgb) in input.iter().zip(output.chunks_exact_mut(3)) {
|
||||
rgb.fill(gray);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RgbImage::from_raw(page.width(), page.height(), rgb)
|
||||
.ok_or(OarOcrError::InvalidImageBuffer { page: page.page() })
|
||||
}
|
||||
|
||||
fn bounding_box_to_quad(bounding_box: &BoundingBox, width: u32, height: u32) -> Option<ImageQuad> {
|
||||
let points: Vec<ImagePoint> = bounding_box
|
||||
.points
|
||||
.iter()
|
||||
.filter(|point| point.x.is_finite() && point.y.is_finite())
|
||||
.map(|point| {
|
||||
ImagePoint::new(
|
||||
point.x.clamp(0.0, width as f32),
|
||||
point.y.clamp(0.0, height as f32),
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
if bounding_box.points.len() == 4 && points.len() == 4 && is_ordered_convex_quad(&points) {
|
||||
return Some(ImageQuad::new([points[0], points[1], points[2], points[3]]));
|
||||
}
|
||||
if points.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let min_x = points
|
||||
.iter()
|
||||
.map(|point| point.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_x = points
|
||||
.iter()
|
||||
.map(|point| point.x)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let min_y = points
|
||||
.iter()
|
||||
.map(|point| point.y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_y = points
|
||||
.iter()
|
||||
.map(|point| point.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if max_x <= min_x || max_y <= min_y {
|
||||
return None;
|
||||
}
|
||||
Some(ImageQuad::new([
|
||||
ImagePoint::new(min_x, min_y),
|
||||
ImagePoint::new(max_x, min_y),
|
||||
ImagePoint::new(max_x, max_y),
|
||||
ImagePoint::new(min_x, max_y),
|
||||
]))
|
||||
}
|
||||
|
||||
fn is_ordered_convex_quad(points: &[ImagePoint]) -> bool {
|
||||
if points.len() != 4 {
|
||||
return false;
|
||||
}
|
||||
let mut orientation = 0.0_f32;
|
||||
for index in 0..4 {
|
||||
let first = points[index];
|
||||
let second = points[(index + 1) % 4];
|
||||
let third = points[(index + 2) % 4];
|
||||
let cross = (second.x - first.x) * (third.y - second.y)
|
||||
- (second.y - first.y) * (third.x - second.x);
|
||||
if cross.abs() <= f32::EPSILON {
|
||||
return false;
|
||||
}
|
||||
if orientation == 0.0 {
|
||||
orientation = cross.signum();
|
||||
} else if cross.signum() != orientation {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use oar_ocr::processors::Point;
|
||||
|
||||
use super::*;
|
||||
use crate::vision::PageTransform;
|
||||
|
||||
#[test]
|
||||
fn cpu_session_budget_is_bounded_for_small_ocr_models() {
|
||||
let config = ocr_session_config();
|
||||
assert!((1..=4).contains(&config.intra_threads.unwrap()));
|
||||
assert_eq!(config.inter_threads, Some(1));
|
||||
assert_eq!(config.parallel_execution, Some(false));
|
||||
}
|
||||
|
||||
fn page(format: RenderPixelFormat, stride: usize, pixels: Vec<u8>) -> RenderedPage {
|
||||
let transform =
|
||||
PageTransform::from_corners(2, 2, (0.0, 2.0), (2.0, 2.0), (0.0, 0.0)).unwrap();
|
||||
RenderedPage::new(1, 2.0, 2.0, 2, 2, stride, format, pixels, transform).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_padded_rgb_without_exposing_padding() {
|
||||
let page = page(
|
||||
RenderPixelFormat::Rgb8,
|
||||
8,
|
||||
vec![1, 2, 3, 4, 5, 6, 99, 99, 7, 8, 9, 10, 11, 12, 99, 99],
|
||||
);
|
||||
let image = rendered_page_to_rgb(&page).unwrap();
|
||||
assert_eq!(image.as_raw(), &[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_rgba_and_gray_to_rgb() {
|
||||
let rgba = page(
|
||||
RenderPixelFormat::Rgba8,
|
||||
8,
|
||||
vec![1, 2, 3, 44, 4, 5, 6, 55, 7, 8, 9, 66, 10, 11, 12, 77],
|
||||
);
|
||||
assert_eq!(
|
||||
rendered_page_to_rgb(&rgba).unwrap().as_raw(),
|
||||
&[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]
|
||||
);
|
||||
|
||||
let gray = page(RenderPixelFormat::Gray8, 2, vec![1, 2, 3, 4]);
|
||||
assert_eq!(
|
||||
rendered_page_to_rgb(&gray).unwrap().as_raw(),
|
||||
&[1, 1, 1, 2, 2, 2, 3, 3, 3, 4, 4, 4]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserves_quads_and_clamps_them_to_the_bitmap() {
|
||||
let bbox = BoundingBox::new(vec![
|
||||
Point::new(-1.0, 2.0),
|
||||
Point::new(11.0, 2.0),
|
||||
Point::new(11.0, 9.0),
|
||||
Point::new(-1.0, 9.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&bbox, 10, 8).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(0.0, 2.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(10.0, 8.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reduces_polygons_to_a_stable_axis_aligned_quad() {
|
||||
let bbox = BoundingBox::new(vec![
|
||||
Point::new(2.0, 1.0),
|
||||
Point::new(7.0, 2.0),
|
||||
Point::new(8.0, 6.0),
|
||||
Point::new(5.0, 9.0),
|
||||
Point::new(1.0, 5.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&bbox, 10, 10).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(1.0, 1.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(8.0, 9.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalizes_unordered_or_partially_invalid_quads() {
|
||||
let unordered = BoundingBox::new(vec![
|
||||
Point::new(1.0, 1.0),
|
||||
Point::new(8.0, 8.0),
|
||||
Point::new(8.0, 1.0),
|
||||
Point::new(1.0, 8.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&unordered, 10, 10).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(1.0, 1.0));
|
||||
assert_eq!(quad.points[1], ImagePoint::new(8.0, 1.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(8.0, 8.0));
|
||||
|
||||
let partially_invalid = BoundingBox::new(vec![
|
||||
Point::new(8.0, 8.0),
|
||||
Point::new(f32::NAN, 4.0),
|
||||
Point::new(1.0, 8.0),
|
||||
Point::new(8.0, 1.0),
|
||||
Point::new(1.0, 1.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&partially_invalid, 10, 10).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(1.0, 1.0));
|
||||
assert_eq!(quad.points[1], ImagePoint::new(8.0, 1.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(8.0, 8.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn refuses_disabled_or_invalid_options_before_inference() {
|
||||
assert!(matches!(
|
||||
validate_options(&OcrOptions::new()),
|
||||
Err(OarOcrError::OcrDisabled)
|
||||
));
|
||||
for value in [-0.1, 1.1, f32::NAN, f32::INFINITY] {
|
||||
let options = OcrOptions::new()
|
||||
.mode(OcrMode::Force)
|
||||
.minimum_confidence(value);
|
||||
assert!(matches!(
|
||||
validate_options(&options),
|
||||
Err(OarOcrError::InvalidMinimumConfidence { .. })
|
||||
));
|
||||
}
|
||||
assert!(validate_options(
|
||||
&OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.minimum_confidence(1.0)
|
||||
)
|
||||
.is_ok());
|
||||
}
|
||||
}
|
||||
+3
-215
@@ -2,11 +2,9 @@
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use firecrawl_pdfium::{PageChar, Pdfium, PixelFormat, PixelPoint, RenderConfig};
|
||||
use firecrawl_pdfium::{Pdfium, PixelFormat, PixelPoint, RenderConfig};
|
||||
use thiserror::Error;
|
||||
|
||||
use crate::types::{ItemType, TextItem};
|
||||
|
||||
use super::{
|
||||
PageRenderer, PageTransform, RenderBufferError, RenderOptions, RenderPixelFormat, RenderedPage,
|
||||
};
|
||||
@@ -49,15 +47,6 @@ pub enum RenderError {
|
||||
/// Number of pages in the document.
|
||||
page_count: usize,
|
||||
},
|
||||
/// The PDFium shared library could not be discovered or loaded.
|
||||
#[error(
|
||||
"failed to load PDFium; install a compatible PDFium shared library or set PDFIUM_LIB_PATH to its path"
|
||||
)]
|
||||
PdfiumLoad {
|
||||
/// Dynamic loading failure.
|
||||
#[source]
|
||||
source: firecrawl_pdfium::Error,
|
||||
},
|
||||
/// PDFium loading, document parsing, form setup, or rendering failed.
|
||||
#[error(transparent)]
|
||||
Pdfium(#[from] firecrawl_pdfium::Error),
|
||||
@@ -76,28 +65,18 @@ pub struct PdfiumRenderer {
|
||||
pdfium: Pdfium,
|
||||
}
|
||||
|
||||
/// Positioned native text recovered from one selected PDF page.
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct PdfiumTextPage {
|
||||
pub(crate) page: u32,
|
||||
pub(crate) page_width: f32,
|
||||
pub(crate) page_height: f32,
|
||||
pub(crate) items: Vec<TextItem>,
|
||||
}
|
||||
|
||||
impl PdfiumRenderer {
|
||||
/// Loads PDFium using `firecrawl-pdfium`'s documented discovery chain.
|
||||
pub fn load() -> Result<Self, RenderError> {
|
||||
Ok(Self {
|
||||
pdfium: Pdfium::load().map_err(|source| RenderError::PdfiumLoad { source })?,
|
||||
pdfium: Pdfium::load()?,
|
||||
})
|
||||
}
|
||||
|
||||
/// Loads PDFium from an explicit native library path.
|
||||
pub fn load_from_path(path: impl AsRef<Path>) -> Result<Self, RenderError> {
|
||||
Ok(Self {
|
||||
pdfium: Pdfium::load_from_path(path)
|
||||
.map_err(|source| RenderError::PdfiumLoad { source })?,
|
||||
pdfium: Pdfium::load_from_path(path)?,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -121,56 +100,6 @@ impl PdfiumRenderer {
|
||||
self.render_pages_impl(pdf_bytes, pages, password, options)
|
||||
}
|
||||
|
||||
/// Extracts positioned native text from selected 1-indexed pages.
|
||||
///
|
||||
/// This is deliberately separate from rendering: callers can probe a
|
||||
/// suspicious embedded text layer before paying for rasterization and
|
||||
/// OCR. A page-level text failure is treated as an unavailable recovery
|
||||
/// candidate so the caller can continue to its normal OCR fallback.
|
||||
pub(crate) fn extract_text_pages(
|
||||
&self,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
) -> Result<Vec<PdfiumTextPage>, RenderError> {
|
||||
const MAX_TEXT_CHARS_PER_PAGE: usize = 250_000;
|
||||
|
||||
if pages.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
if pages.contains(&0) {
|
||||
return Err(RenderError::InvalidPageNumber);
|
||||
}
|
||||
|
||||
let document = self.pdfium.load_document(pdf_bytes.to_vec(), password)?;
|
||||
let page_count = document.page_count();
|
||||
if let Some(&page) = pages.iter().find(|&&page| page as usize > page_count) {
|
||||
return Err(RenderError::PageOutOfBounds { page, page_count });
|
||||
}
|
||||
|
||||
let mut recovered = Vec::with_capacity(pages.len());
|
||||
for &page_number in pages {
|
||||
let page = document.page(page_number as usize - 1)?;
|
||||
let page_size = page.size();
|
||||
let text = match page.text_with_limit(MAX_TEXT_CHARS_PER_PAGE) {
|
||||
Ok(text) => text,
|
||||
Err(error) => {
|
||||
log::debug!(
|
||||
"page {page_number}: positioned native text recovery unavailable: {error}"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
recovered.push(PdfiumTextPage {
|
||||
page: page_number,
|
||||
page_width: page_size.width,
|
||||
page_height: page_size.height,
|
||||
items: text_chars_to_items(text.chars(), page_number),
|
||||
});
|
||||
}
|
||||
Ok(recovered)
|
||||
}
|
||||
|
||||
fn render_pages_impl(
|
||||
&self,
|
||||
pdf_bytes: &[u8],
|
||||
@@ -216,106 +145,6 @@ impl PdfiumRenderer {
|
||||
}
|
||||
}
|
||||
|
||||
fn text_chars_to_items(chars: &[PageChar], page: u32) -> Vec<TextItem> {
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
struct Bounds {
|
||||
left: f64,
|
||||
bottom: f64,
|
||||
right: f64,
|
||||
top: f64,
|
||||
}
|
||||
|
||||
fn flush(items: &mut Vec<TextItem>, text: &mut String, bounds: &mut Option<Bounds>, page: u32) {
|
||||
let Some(bounds) = bounds.take() else {
|
||||
text.clear();
|
||||
return;
|
||||
};
|
||||
if text.is_empty() {
|
||||
return;
|
||||
}
|
||||
let width = (bounds.right - bounds.left) as f32;
|
||||
let height = (bounds.top - bounds.bottom) as f32;
|
||||
let x = bounds.left as f32;
|
||||
let y = bounds.bottom as f32;
|
||||
if !x.is_finite()
|
||||
|| !y.is_finite()
|
||||
|| !width.is_finite()
|
||||
|| !height.is_finite()
|
||||
|| width <= 0.0
|
||||
|| height <= 0.0
|
||||
{
|
||||
text.clear();
|
||||
return;
|
||||
}
|
||||
items.push(TextItem {
|
||||
text: std::mem::take(text),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
font: "PDFium native text".to_string(),
|
||||
font_size: height.max(1.0),
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
let mut items = Vec::new();
|
||||
let mut text = String::new();
|
||||
let mut bounds: Option<Bounds> = None;
|
||||
for character in chars {
|
||||
let Some(value) = character.unicode else {
|
||||
flush(&mut items, &mut text, &mut bounds, page);
|
||||
continue;
|
||||
};
|
||||
if value.is_whitespace() {
|
||||
flush(&mut items, &mut text, &mut bounds, page);
|
||||
continue;
|
||||
}
|
||||
|
||||
let rect = character.loose_bounds.normalized();
|
||||
if !rect.left.is_finite()
|
||||
|| !rect.bottom.is_finite()
|
||||
|| !rect.right.is_finite()
|
||||
|| !rect.top.is_finite()
|
||||
|| rect.width() <= 0.0
|
||||
|| rect.height() <= 0.0
|
||||
{
|
||||
flush(&mut items, &mut text, &mut bounds, page);
|
||||
continue;
|
||||
}
|
||||
text.push(value);
|
||||
bounds = Some(match bounds {
|
||||
Some(bounds) => Bounds {
|
||||
left: bounds.left.min(rect.left),
|
||||
bottom: bounds.bottom.min(rect.bottom),
|
||||
right: bounds.right.max(rect.right),
|
||||
top: bounds.top.max(rect.top),
|
||||
},
|
||||
None => Bounds {
|
||||
left: rect.left,
|
||||
bottom: rect.bottom,
|
||||
right: rect.right,
|
||||
top: rect.top,
|
||||
},
|
||||
});
|
||||
}
|
||||
flush(&mut items, &mut text, &mut bounds, page);
|
||||
items.sort_by(|first, second| {
|
||||
first
|
||||
.page
|
||||
.cmp(&second.page)
|
||||
.then(second.y.total_cmp(&first.y))
|
||||
.then(first.x.total_cmp(&second.x))
|
||||
});
|
||||
items
|
||||
}
|
||||
|
||||
impl PageRenderer for PdfiumRenderer {
|
||||
type Error = RenderError;
|
||||
|
||||
@@ -408,17 +237,6 @@ fn bgr_to_rgb_in_place(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use firecrawl_pdfium::{PagePoint, PageRect};
|
||||
|
||||
fn page_char(value: char, bounds: PageRect) -> PageChar {
|
||||
PageChar {
|
||||
unicode: Some(value),
|
||||
code: value as u32,
|
||||
bounds,
|
||||
loose_bounds: bounds,
|
||||
origin: PagePoint::new(bounds.left, bounds.bottom),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bgr_pixels_are_converted_to_rgb_in_place() {
|
||||
@@ -445,34 +263,4 @@ mod tests {
|
||||
Err(RenderBufferError::InvalidBufferLength { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_character_geometry_splits_text_runs() {
|
||||
let chars = [
|
||||
page_char('A', PageRect::new(0.0, 0.0, 8.0, 10.0)),
|
||||
page_char('X', PageRect::new(10.0, 0.0, 10.0, 10.0)),
|
||||
page_char('B', PageRect::new(20.0, 0.0, 28.0, 10.0)),
|
||||
];
|
||||
|
||||
let items = text_chars_to_items(&chars, 1);
|
||||
|
||||
assert_eq!(
|
||||
items
|
||||
.iter()
|
||||
.map(|item| item.text.as_str())
|
||||
.collect::<Vec<_>>(),
|
||||
["A", "B"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn coordinates_that_overflow_f32_are_discarded() {
|
||||
let left = f64::from(f32::MAX) * 2.0;
|
||||
let chars = [page_char(
|
||||
'A',
|
||||
PageRect::new(left, 0.0, left + 1.0e30, 10.0),
|
||||
)];
|
||||
|
||||
assert!(text_chars_to_items(&chars, 1).is_empty());
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,384 +0,0 @@
|
||||
//! Page routing and renderer/OCR orchestration without Markdown fusion.
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
use std::error::Error;
|
||||
use std::time::Instant;
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{OcrEngine, OcrMode, OcrOptions, OcrPage, PageRenderer, RenderOptions, RenderedPage};
|
||||
|
||||
/// A rendered page paired with OCR output in the same bitmap coordinate space.
|
||||
#[derive(Debug)]
|
||||
pub struct RoutedOcrPage {
|
||||
/// Renderer-owned bitmap and pixel↔PDF transform.
|
||||
pub rendered: RenderedPage,
|
||||
/// Positioned OCR spans for the bitmap.
|
||||
pub ocr: OcrPage,
|
||||
}
|
||||
|
||||
/// Output of one selective OCR invocation.
|
||||
#[derive(Debug)]
|
||||
pub struct OcrRun {
|
||||
/// Pages processed in ascending document order.
|
||||
pub pages: Vec<RoutedOcrPage>,
|
||||
/// Total page-rendering wall time.
|
||||
pub render_time_ms: u64,
|
||||
/// Total engine wall time.
|
||||
pub ocr_time_ms: u64,
|
||||
}
|
||||
|
||||
/// Selects 1-indexed pages for OCR.
|
||||
///
|
||||
/// `recommended_pages` comes from pdf-inspector's existing detector/text
|
||||
/// quality signals. `selected_pages` is an optional user page filter. Results
|
||||
/// are validated, deduplicated, and returned in document order.
|
||||
pub fn route_ocr_pages(
|
||||
mode: OcrMode,
|
||||
page_count: u32,
|
||||
recommended_pages: &[u32],
|
||||
selected_pages: Option<&[u32]>,
|
||||
) -> Result<Vec<u32>, OcrRoutingError> {
|
||||
match mode {
|
||||
OcrMode::Off => Ok(Vec::new()),
|
||||
OcrMode::Auto => {
|
||||
let mut routed = validated_page_set("recommended", recommended_pages, page_count)?;
|
||||
if let Some(selected) = selected_pages {
|
||||
let selected = validated_page_set("selected", selected, page_count)?;
|
||||
routed.retain(|page| selected.contains(page));
|
||||
}
|
||||
Ok(routed.into_iter().collect())
|
||||
}
|
||||
OcrMode::Force => {
|
||||
let routed = selected_pages
|
||||
.map(|pages| validated_page_set("selected", pages, page_count))
|
||||
.transpose()?
|
||||
.unwrap_or_else(|| (1..=page_count).collect());
|
||||
Ok(routed.into_iter().collect())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Renders and recognizes already-routed pages while retaining transforms for
|
||||
/// the following fusion layer.
|
||||
///
|
||||
/// An empty page list returns without calling either dependency, which keeps
|
||||
/// model resolution and inference lazy when Auto routing finds no OCR work.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn run_ocr_pages<R, O>(
|
||||
renderer: &R,
|
||||
engine: &O,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
render_options: &RenderOptions,
|
||||
ocr_options: &OcrOptions,
|
||||
) -> Result<OcrRun, OcrRunError>
|
||||
where
|
||||
R: PageRenderer,
|
||||
O: OcrEngine,
|
||||
{
|
||||
if pages.is_empty() {
|
||||
return Ok(OcrRun {
|
||||
pages: Vec::new(),
|
||||
render_time_ms: 0,
|
||||
ocr_time_ms: 0,
|
||||
});
|
||||
}
|
||||
if ocr_options.mode == OcrMode::Off {
|
||||
return Err(OcrRunError::OcrDisabled);
|
||||
}
|
||||
|
||||
let render_started = Instant::now();
|
||||
let rendered = renderer
|
||||
.render_pages(pdf_bytes, pages, password, render_options)
|
||||
.map_err(|source| OcrRunError::Render {
|
||||
source: Box::new(source),
|
||||
})?;
|
||||
let render_time_ms = elapsed_ms(render_started);
|
||||
validate_page_order("renderer", pages, rendered.iter().map(RenderedPage::page))?;
|
||||
|
||||
let ocr_started = Instant::now();
|
||||
let recognized =
|
||||
engine
|
||||
.recognize(&rendered, ocr_options)
|
||||
.map_err(|source| OcrRunError::Ocr {
|
||||
source: Box::new(source),
|
||||
})?;
|
||||
let ocr_time_ms = elapsed_ms(ocr_started);
|
||||
validate_page_order(
|
||||
"OCR engine",
|
||||
pages,
|
||||
recognized.iter().map(|page| page.page_number),
|
||||
)?;
|
||||
|
||||
Ok(OcrRun {
|
||||
pages: rendered
|
||||
.into_iter()
|
||||
.zip(recognized)
|
||||
.map(|(rendered, ocr)| RoutedOcrPage { rendered, ocr })
|
||||
.collect(),
|
||||
render_time_ms,
|
||||
ocr_time_ms,
|
||||
})
|
||||
}
|
||||
|
||||
/// Invalid page routing or renderer/engine contract output.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrRoutingError {
|
||||
/// A page list contained zero or a page beyond the document.
|
||||
#[error("{source_name} OCR page {page} is outside the valid range 1..={page_count}")]
|
||||
InvalidPage {
|
||||
/// Page-list source.
|
||||
source_name: &'static str,
|
||||
/// Invalid 1-indexed page.
|
||||
page: u32,
|
||||
/// Document page count.
|
||||
page_count: u32,
|
||||
},
|
||||
}
|
||||
|
||||
/// Failures while rendering and recognizing a routed page set.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrRunError {
|
||||
/// A non-empty route cannot execute with OCR disabled.
|
||||
#[error("cannot process routed pages while OCR mode is Off")]
|
||||
OcrDisabled,
|
||||
/// Page rasterization failed.
|
||||
#[error("page rendering failed: {source}")]
|
||||
Render {
|
||||
/// Renderer-specific failure.
|
||||
#[source]
|
||||
source: Box<dyn Error + Send + Sync>,
|
||||
},
|
||||
/// OCR inference failed.
|
||||
#[error("OCR inference failed: {source}")]
|
||||
Ocr {
|
||||
/// Engine-specific failure.
|
||||
#[source]
|
||||
source: Box<dyn Error + Send + Sync>,
|
||||
},
|
||||
/// A dependency returned the wrong count or order.
|
||||
#[error("{stage} returned pages {actual:?}; expected {expected:?}")]
|
||||
PageOrderMismatch {
|
||||
/// Dependency boundary that violated the contract.
|
||||
stage: &'static str,
|
||||
/// Requested 1-indexed pages.
|
||||
expected: Vec<u32>,
|
||||
/// Returned 1-indexed pages.
|
||||
actual: Vec<u32>,
|
||||
},
|
||||
}
|
||||
|
||||
fn validated_page_set(
|
||||
source_name: &'static str,
|
||||
pages: &[u32],
|
||||
page_count: u32,
|
||||
) -> Result<BTreeSet<u32>, OcrRoutingError> {
|
||||
let mut result = BTreeSet::new();
|
||||
for &page in pages {
|
||||
if page == 0 || page > page_count {
|
||||
return Err(OcrRoutingError::InvalidPage {
|
||||
source_name,
|
||||
page,
|
||||
page_count,
|
||||
});
|
||||
}
|
||||
result.insert(page);
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn validate_page_order(
|
||||
stage: &'static str,
|
||||
expected: &[u32],
|
||||
actual: impl IntoIterator<Item = u32>,
|
||||
) -> Result<(), OcrRunError> {
|
||||
let actual: Vec<u32> = actual.into_iter().collect();
|
||||
if actual == expected {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(OcrRunError::PageOrderMismatch {
|
||||
stage,
|
||||
expected: expected.to_vec(),
|
||||
actual,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(started: Instant) -> u64 {
|
||||
u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::vision::{
|
||||
ImagePoint, ImageQuad, ModelIdentity, OcrSpan, PageTransform, RenderBufferError,
|
||||
RenderPixelFormat,
|
||||
};
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
#[error("fake failure")]
|
||||
struct FakeError;
|
||||
|
||||
struct FakeRenderer;
|
||||
|
||||
impl PageRenderer for FakeRenderer {
|
||||
type Error = FakeError;
|
||||
|
||||
fn render_pages(
|
||||
&self,
|
||||
_pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
_password: Option<&str>,
|
||||
_options: &RenderOptions,
|
||||
) -> Result<Vec<RenderedPage>, Self::Error> {
|
||||
pages
|
||||
.iter()
|
||||
.copied()
|
||||
.map(rendered_page)
|
||||
.collect::<Result<_, _>>()
|
||||
.map_err(|_| FakeError)
|
||||
}
|
||||
}
|
||||
|
||||
struct FakeEngine {
|
||||
model: ModelIdentity,
|
||||
}
|
||||
|
||||
impl FakeEngine {
|
||||
fn new() -> Self {
|
||||
Self {
|
||||
model: ModelIdentity::new("fake", "v1"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl OcrEngine for FakeEngine {
|
||||
type Error = FakeError;
|
||||
|
||||
fn model(&self) -> &ModelIdentity {
|
||||
&self.model
|
||||
}
|
||||
|
||||
fn recognize(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
_options: &OcrOptions,
|
||||
) -> Result<Vec<OcrPage>, Self::Error> {
|
||||
Ok(pages
|
||||
.iter()
|
||||
.map(|page| OcrPage {
|
||||
page_number: page.page(),
|
||||
spans: vec![OcrSpan {
|
||||
text: format!("page {}", page.page()),
|
||||
polygon: ImageQuad::new([
|
||||
ImagePoint::new(0.0, 0.0),
|
||||
ImagePoint::new(1.0, 0.0),
|
||||
ImagePoint::new(1.0, 1.0),
|
||||
ImagePoint::new(0.0, 1.0),
|
||||
]),
|
||||
confidence: 0.9,
|
||||
orientation_degrees: None,
|
||||
}],
|
||||
mean_confidence: Some(0.9),
|
||||
model: self.model.clone(),
|
||||
processing_time_ms: 1,
|
||||
warnings: Vec::new(),
|
||||
})
|
||||
.collect())
|
||||
}
|
||||
}
|
||||
|
||||
fn rendered_page(page: u32) -> Result<RenderedPage, RenderBufferError> {
|
||||
let transform =
|
||||
PageTransform::from_corners(1, 1, (0.0, 1.0), (1.0, 1.0), (0.0, 0.0)).unwrap();
|
||||
RenderedPage::new(
|
||||
page,
|
||||
1.0,
|
||||
1.0,
|
||||
1,
|
||||
1,
|
||||
3,
|
||||
RenderPixelFormat::Rgb8,
|
||||
vec![255; 3],
|
||||
transform,
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn off_auto_and_force_route_expected_pages() {
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Off, 0, &[99], Some(&[0])).unwrap(),
|
||||
Vec::<u32>::new()
|
||||
);
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Auto, 5, &[5, 3, 3, 1], Some(&[2, 3, 5])).unwrap(),
|
||||
vec![3, 5]
|
||||
);
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Force, 4, &[], None).unwrap(),
|
||||
vec![1, 2, 3, 4]
|
||||
);
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Force, 4, &[], Some(&[4, 2, 2])).unwrap(),
|
||||
vec![2, 4]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn routing_rejects_invalid_page_numbers() {
|
||||
assert!(matches!(
|
||||
route_ocr_pages(OcrMode::Auto, 2, &[0], None),
|
||||
Err(OcrRoutingError::InvalidPage { page: 0, .. })
|
||||
));
|
||||
assert!(matches!(
|
||||
route_ocr_pages(OcrMode::Force, 2, &[], Some(&[3])),
|
||||
Err(OcrRoutingError::InvalidPage { page: 3, .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_retains_render_transforms_and_input_order() {
|
||||
let options = OcrOptions::new().mode(OcrMode::Auto);
|
||||
let run = run_ocr_pages(
|
||||
&FakeRenderer,
|
||||
&FakeEngine::new(),
|
||||
b"pdf",
|
||||
&[2, 4],
|
||||
None,
|
||||
&RenderOptions::new(),
|
||||
&options,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
run.pages
|
||||
.iter()
|
||||
.map(|page| page.rendered.page())
|
||||
.collect::<Vec<_>>(),
|
||||
vec![2, 4]
|
||||
);
|
||||
assert_eq!(run.pages[1].ocr.spans[0].text, "page 4");
|
||||
assert_eq!(run.pages[1].rendered.pixel_to_page(0.0, 0.0).y, 1.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_route_is_a_noop_even_when_ocr_is_off() {
|
||||
let run = run_ocr_pages(
|
||||
&FakeRenderer,
|
||||
&FakeEngine::new(),
|
||||
b"pdf",
|
||||
&[],
|
||||
None,
|
||||
&RenderOptions::new(),
|
||||
&OcrOptions::new(),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(run.pages.is_empty());
|
||||
assert_eq!(run.render_time_ms, 0);
|
||||
assert_eq!(run.ocr_time_ms, 0);
|
||||
}
|
||||
}
|
||||
@@ -5,7 +5,9 @@ use pdf_inspector::vision::{PdfiumRenderer, RenderError, RenderOptions, RenderPi
|
||||
fn load_renderer() -> Option<PdfiumRenderer> {
|
||||
match PdfiumRenderer::load() {
|
||||
Ok(renderer) => Some(renderer),
|
||||
Err(RenderError::PdfiumLoad { .. }) => {
|
||||
Err(RenderError::Pdfium(firecrawl_pdfium::Error::Load(
|
||||
firecrawl_pdfium::LoadError::LibraryNotFound { .. },
|
||||
))) => {
|
||||
eprintln!("skipping PDFium runtime test because no native library is installed");
|
||||
None
|
||||
}
|
||||
|
||||
@@ -1,214 +0,0 @@
|
||||
#![cfg(all(feature = "ocr-oar", not(target_arch = "wasm32")))]
|
||||
|
||||
#[cfg(feature = "ocr")]
|
||||
use pdf_inspector::vision::{
|
||||
process_pdf_with_ocr_mem, ModelDownloadPolicy, OcrPdfOptions, OcrPipelineError,
|
||||
PageContentSource,
|
||||
};
|
||||
use pdf_inspector::vision::{
|
||||
ModelStore, OarOcrEngine, OcrEngine, OcrMode, OcrOptions, PageTransform, RenderPixelFormat,
|
||||
RenderedPage, PP_OCR_V6_SMALL,
|
||||
};
|
||||
#[cfg(feature = "render-pdfium")]
|
||||
use pdf_inspector::vision::{PdfiumRenderer, RenderError, RenderOptions};
|
||||
|
||||
const MODEL_DIRECTORY_ENV: &str = "PDF_INSPECTOR_OCR_TEST_MODELS";
|
||||
const IMAGE_ENV: &str = "PDF_INSPECTOR_OCR_TEST_IMAGE";
|
||||
const EXPECTED_TEXT_ENV: &str = "PDF_INSPECTOR_OCR_TEST_EXPECTED";
|
||||
|
||||
#[cfg(feature = "render-pdfium")]
|
||||
fn load_renderer() -> Option<PdfiumRenderer> {
|
||||
match PdfiumRenderer::load() {
|
||||
Ok(renderer) => Some(renderer),
|
||||
Err(RenderError::PdfiumLoad { .. }) => {
|
||||
eprintln!("skipping OCR runtime test because no native PDFium library is installed");
|
||||
None
|
||||
}
|
||||
Err(error) => panic!("failed to load PDFium: {error}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recognizes_an_rgb_image_with_verified_models() {
|
||||
let Some(model_directory) = std::env::var_os(MODEL_DIRECTORY_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {MODEL_DIRECTORY_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
let Some(image_path) = std::env::var_os(IMAGE_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {IMAGE_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
|
||||
let image = image::open(image_path).unwrap().into_rgb8();
|
||||
let (width, height) = image.dimensions();
|
||||
let transform = PageTransform::from_corners(
|
||||
width,
|
||||
height,
|
||||
(0.0, f64::from(height)),
|
||||
(f64::from(width), f64::from(height)),
|
||||
(0.0, 0.0),
|
||||
)
|
||||
.unwrap();
|
||||
let page = RenderedPage::new(
|
||||
1,
|
||||
width as f32,
|
||||
height as f32,
|
||||
width,
|
||||
height,
|
||||
width as usize * 3,
|
||||
RenderPixelFormat::Rgb8,
|
||||
image.into_raw(),
|
||||
transform,
|
||||
)
|
||||
.unwrap();
|
||||
let results = recognize(&model_directory, &[page]);
|
||||
assert_usable_result(&results);
|
||||
|
||||
let text = results[0]
|
||||
.spans
|
||||
.iter()
|
||||
.map(|span| span.text.as_str())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
eprintln!("recognized: {text}");
|
||||
if let Ok(expected) = std::env::var(EXPECTED_TEXT_ENV) {
|
||||
assert!(
|
||||
text.to_lowercase().contains(&expected.to_lowercase()),
|
||||
"expected OCR output to contain {expected:?}, got {text:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "render-pdfium")]
|
||||
#[test]
|
||||
fn recognizes_a_pdfium_rendered_fixture_with_verified_models() {
|
||||
let Some(model_directory) = std::env::var_os(MODEL_DIRECTORY_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {MODEL_DIRECTORY_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
let Some(renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let pages = renderer
|
||||
.render_pages(
|
||||
&bytes,
|
||||
&[1],
|
||||
None,
|
||||
&RenderOptions::new().dpi(150.0).form_fields(false),
|
||||
)
|
||||
.unwrap();
|
||||
let results = recognize(&model_directory, &pages);
|
||||
assert_usable_result(&results);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", feature = "render-pdfium"))]
|
||||
#[test]
|
||||
fn complete_ocr_pipeline_routes_and_assembles_a_scanned_fixture() {
|
||||
let Some(model_directory) = std::env::var_os(MODEL_DIRECTORY_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {MODEL_DIRECTORY_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
let Some(_renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/scan_with_native_header_text.pdf").unwrap();
|
||||
let ocr = OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.minimum_confidence(0.3)
|
||||
.model_directory(model_directory)
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
let options = OcrPdfOptions::new().ocr(ocr);
|
||||
let result = process_pdf_with_ocr_mem(&bytes, options.clone()).unwrap();
|
||||
let repeated = process_pdf_with_ocr_mem(&bytes, options).unwrap();
|
||||
|
||||
assert_eq!(result.pages_routed_to_ocr, vec![1]);
|
||||
assert!(!result.markdown.trim().is_empty());
|
||||
assert!(result
|
||||
.markdown
|
||||
.contains("Order Date Item Code Description Status Unit Cost\n\n03/14/2024"));
|
||||
assert!(result.markdown.contains("$482,110.40\n\n05/02/2024"));
|
||||
assert_eq!(result.pages[0].provenance.source, PageContentSource::Fused);
|
||||
assert!(result.pages[0]
|
||||
.provenance
|
||||
.warnings
|
||||
.iter()
|
||||
.any(|warning| warning.contains("complementary OCR")));
|
||||
assert_eq!(
|
||||
result.pages[0].provenance.ocr_model.as_ref().unwrap().name,
|
||||
PP_OCR_V6_SMALL.id
|
||||
);
|
||||
assert_eq!(repeated.markdown, result.markdown);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", feature = "render-pdfium"))]
|
||||
#[test]
|
||||
fn auto_recovers_credible_native_text_before_loading_ocr_models() {
|
||||
let Some(_renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
let ocr = OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.model_directory("/models/must-not-be-read")
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
let result = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().ocr(ocr)).unwrap();
|
||||
|
||||
assert_eq!(result.pages_recommended_for_ocr, vec![1]);
|
||||
assert!(result.pages_routed_to_ocr.is_empty());
|
||||
assert!(result.markdown.contains("羽田空港新飛行経路"));
|
||||
assert!(result.markdown.contains("|4月30日|有|81.0|"));
|
||||
assert!(result.markdown.contains("※1 最大騒音レベル"));
|
||||
assert!(result.pages_with_tables.contains(&1));
|
||||
assert_eq!(result.pages[0].provenance.source, PageContentSource::Native);
|
||||
assert!(result.pages[0].provenance.ocr_model.is_none());
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", feature = "render-pdfium"))]
|
||||
#[test]
|
||||
fn auto_rejects_garbled_native_recovery_and_continues_to_ocr() {
|
||||
let Some(_renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/shifted_cipher_tounicode.pdf").unwrap();
|
||||
let ocr = OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.model_directory("/models/must-not-be-read")
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
let error = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().ocr(ocr)).unwrap_err();
|
||||
|
||||
assert!(matches!(
|
||||
error,
|
||||
OcrPipelineError::ModelAcquire(_) | OcrPipelineError::ModelStore(_)
|
||||
));
|
||||
}
|
||||
|
||||
fn recognize(
|
||||
model_directory: &std::ffi::OsStr,
|
||||
pages: &[RenderedPage],
|
||||
) -> Vec<pdf_inspector::vision::OcrPage> {
|
||||
let store = ModelStore::new(model_directory).override_root(model_directory);
|
||||
let models = store.resolve(&PP_OCR_V6_SMALL).unwrap();
|
||||
let engine = OarOcrEngine::from_models(&models).unwrap();
|
||||
engine
|
||||
.recognize(
|
||||
pages,
|
||||
&OcrOptions::new()
|
||||
.mode(OcrMode::Force)
|
||||
.minimum_confidence(0.3),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn assert_usable_result(results: &[pdf_inspector::vision::OcrPage]) {
|
||||
assert_eq!(results.len(), 1);
|
||||
assert_eq!(results[0].page_number, 1);
|
||||
assert_eq!(results[0].model.name, PP_OCR_V6_SMALL.id);
|
||||
assert_eq!(results[0].model.revision, PP_OCR_V6_SMALL.revision);
|
||||
assert!(!results[0].spans.is_empty());
|
||||
assert!(results[0].spans.iter().all(|span| span.confidence >= 0.3));
|
||||
}
|
||||
@@ -77,50 +77,6 @@ class TestProcessPdfBytes:
|
||||
assert result.markdown is not None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# process_pdf_with_ocr / process_pdf_with_ocr_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestProcessPdfWithOcr:
|
||||
def test_off_mode_has_full_provenance_without_external_runtimes(self):
|
||||
result = pdf_inspector.process_pdf_with_ocr(
|
||||
fixture_path("thermo-freon12.pdf"), mode="off"
|
||||
)
|
||||
assert result.page_count == 3
|
||||
assert len(result.pages) == 3
|
||||
assert result.pages_routed_to_ocr == []
|
||||
assert all(page.provenance.source == "native" for page in result.pages)
|
||||
assert all(page.provenance.ocr_model is None for page in result.pages)
|
||||
assert result.markdown
|
||||
assert "OcrPdfResult" in repr(result)
|
||||
|
||||
def test_auto_mode_skips_external_runtimes_for_clean_text(self):
|
||||
result = pdf_inspector.process_pdf_with_ocr_bytes(
|
||||
fixture_bytes("thermo-freon12.pdf")
|
||||
)
|
||||
assert result.pages_routed_to_ocr == []
|
||||
assert result.render_time_ms == 0
|
||||
assert result.ocr_time_ms == 0
|
||||
|
||||
def test_selected_pages_are_one_indexed(self):
|
||||
result = pdf_inspector.process_pdf_with_ocr(
|
||||
fixture_path("thermo-freon12.pdf"), mode="off", page_numbers=[2]
|
||||
)
|
||||
assert [page.page_number for page in result.pages] == [2]
|
||||
|
||||
def test_rejects_invalid_options(self):
|
||||
with pytest.raises(ValueError, match="mode must be"):
|
||||
pdf_inspector.process_pdf_with_ocr(
|
||||
fixture_path("thermo-freon12.pdf"), mode="sometimes"
|
||||
)
|
||||
with pytest.raises(ValueError, match="page 0"):
|
||||
pdf_inspector.process_pdf_with_ocr(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
mode="off",
|
||||
page_numbers=[0],
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# detect_pdf / detect_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user