Compare commits
21
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
21a436ac1b | ||
|
|
d2d8e35a7b | ||
|
|
30c9dbbc72 | ||
|
|
162c5cbf10 | ||
|
|
4e4cfdc74c | ||
|
|
24e66245cb | ||
|
|
cee7b6c381 | ||
|
|
b5a63d7036 | ||
|
|
6829397bce | ||
|
|
fc16133a58 | ||
|
|
41a6a67c03 | ||
|
|
58fe5a0224 | ||
|
|
c4c969a562 | ||
|
|
bf0cd8decb | ||
|
|
232b4cdef5 | ||
|
|
e3f5429638 | ||
|
|
f6cbe979f6 | ||
|
|
3f43745313 | ||
|
|
a012cb65a6 | ||
|
|
6567e1ab2d | ||
|
|
3af409d27f |
+27
@@ -49,6 +49,25 @@ ttf-parser = "0.25"
|
||||
lopdf = { version = "0.42.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
# Optional native page rendering for OCR pipelines. PDFium is loaded at
|
||||
# runtime, so enabling this feature does not link or download a native library.
|
||||
firecrawl-pdfium = { version = "0.1.0", optional = true }
|
||||
# Small support crates used only by the opt-in model cache. Model files remain
|
||||
# external and are never embedded in pdf-inspector artifacts.
|
||||
dirs = { version = "6.0", optional = true }
|
||||
fs2 = { version = "0.4", optional = true }
|
||||
sha2 = { version = "0.11", optional = true }
|
||||
# Optional CPU OCR backend. Models and ONNX Runtime stay external: the latter
|
||||
# is loaded dynamically from ORT_DYLIB_PATH or the platform library search path.
|
||||
image = { version = "0.25.6", default-features = false, optional = true }
|
||||
oar-ocr = { version = "0.9.1", default-features = false, features = ["simd"], optional = true }
|
||||
ort = { version = "=2.0.0-rc.13", default-features = false, features = ["load-dynamic"], optional = true }
|
||||
# HTTPS-only streaming downloader for pinned model artifacts. Kept separate
|
||||
# from model-cache so offline and package-managed deployments avoid HTTP/TLS.
|
||||
ureq = { version = "3.4", default-features = false, features = ["rustls", "platform-verifier"], optional = true }
|
||||
|
||||
[target.'cfg(all(windows, not(target_arch = "wasm32")))'.dependencies]
|
||||
windows-sys = { version = "0.61", features = ["Win32_Storage_FileSystem"], optional = true }
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
@@ -62,6 +81,14 @@ tempfile = "3.3"
|
||||
[features]
|
||||
default = []
|
||||
python = ["pyo3"]
|
||||
vision = []
|
||||
model-cache = ["vision", "dep:dirs", "dep:fs2", "dep:sha2", "dep:windows-sys"]
|
||||
model-download = ["model-cache", "dep:ureq"]
|
||||
ocr-oar = ["model-cache", "dep:image", "dep:oar-ocr", "dep:ort"]
|
||||
render-pdfium = ["vision", "dep:firecrawl-pdfium"]
|
||||
# Complete native OCR path. This remains opt-in so default library,
|
||||
# renderer-only, and browser consumers do not inherit inference or HTTP/TLS.
|
||||
ocr = ["render-pdfium", "ocr-oar", "model-download"]
|
||||
|
||||
[[bin]]
|
||||
name = "pdf2md"
|
||||
|
||||
+280
-1
@@ -1,6 +1,6 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Pure Rust, no ML models, no external services; the only PDF dependency is [lopdf](https://crates.io/crates/lopdf). Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector).
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. The default build is pure Rust, has no ML models or external services, and uses [lopdf](https://crates.io/crates/lopdf) for PDF parsing. Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector/).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -117,6 +117,285 @@ let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
|
||||
### Vision extension contracts
|
||||
|
||||
The native-only `vision` feature exposes the stable seam used by OCR
|
||||
integrations without selecting or embedding an inference runtime. The
|
||||
separate `model-cache` feature adds pinned artifact management:
|
||||
|
||||
- `PageRenderer`, `OcrEngine`, and `LayoutEngine` traits;
|
||||
- renderer-neutral owned page buffers and affine pixel↔PDF transforms;
|
||||
- `OcrOptions` and opt-in `Off`/`Auto`/`Force` routing modes;
|
||||
- positioned OCR/layout results and per-page provenance types; and
|
||||
- a versioned PP-OCRv6 Small manifest with checksum-verified, locked, atomic
|
||||
model-cache installation and explicit offline-directory overrides.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = ["vision", "model-cache"] }
|
||||
```
|
||||
|
||||
The OCR contracts preserve existing behavior by default: OCR is `Off`, learned
|
||||
layout is disabled, and model resolution is never reached. `ModelStore` itself
|
||||
does not access the network. The optional `model-download` feature provides an
|
||||
HTTPS downloader that streams pinned artifacts into the checksum-verified
|
||||
cache only after routing has selected OCR work. Offline consumers set an
|
||||
explicit model directory and `ModelDownloadPolicy::Offline`. Renderer-only
|
||||
consumers do not enable `model-cache` or `model-download` and therefore do not
|
||||
compile their filesystem, hashing, or HTTP dependencies.
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
ModelDownloadPolicy, ModelStore, OcrMode, OcrOptions, PP_OCR_V6_SMALL,
|
||||
};
|
||||
|
||||
let ocr = OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.model_directory("/opt/firecrawl/models/pp-ocrv6-small")
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
// Verifies exact sizes and SHA-256 digests before an engine opens the files.
|
||||
let models = ModelStore::from_options(&ocr)?.resolve(&PP_OCR_V6_SMALL)?;
|
||||
println!("using {} at {}", models.manifest_id(), models.revision());
|
||||
```
|
||||
|
||||
### Optional native page rendering
|
||||
|
||||
The `render-pdfium` feature adds a native-only page renderer backed by
|
||||
[`firecrawl-pdfium`](https://crates.io/crates/firecrawl-pdfium). It is the
|
||||
rendering boundary for OCR pipelines; enabling it does not include an OCR
|
||||
model or change the existing extraction functions. It implies `vision`,
|
||||
and `PdfiumRenderer` implements the renderer-neutral `PageRenderer` trait.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = ["render-pdfium"] }
|
||||
```
|
||||
|
||||
PDFium is loaded at runtime. Set `PDFIUM_LIB_PATH`, place its shared library
|
||||
next to the executable, or use another discovery route supported by
|
||||
`firecrawl-pdfium`.
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{PdfiumRenderer, RenderOptions};
|
||||
|
||||
let renderer = PdfiumRenderer::load()?;
|
||||
let bytes = std::fs::read("document.pdf")?;
|
||||
let pages = renderer.render_pages(
|
||||
&bytes,
|
||||
&[1, 3], // 1-indexed, matching pages_needing_ocr
|
||||
None, // optional PDF password
|
||||
&RenderOptions::new().dpi(150.0),
|
||||
)?;
|
||||
|
||||
for page in pages {
|
||||
// Owned RGB pixels can leave the PDFium critical section and be sent to
|
||||
// an OCR worker. OCR pixel boxes can be mapped back to PDF coordinates.
|
||||
let rect = page.pixel_rect_to_pdf_rect(20.0, 30.0, 100.0, 24.0);
|
||||
println!("page {}: {}x{}, rect={rect:?}", page.page(), page.width(), page.height());
|
||||
}
|
||||
```
|
||||
|
||||
Browser WASM remains on the default text-only path and does not expose native
|
||||
PDFium rendering.
|
||||
|
||||
### Optional OCR engine
|
||||
|
||||
The native-only `ocr-oar` feature adds a CPU PP-OCRv6 Small implementation of
|
||||
`OcrEngine` backed by OAR and ONNX Runtime. It implies `model-cache`, but does
|
||||
not enable model auto-download, ONNX Runtime download, or PDF rendering. Model
|
||||
files remain external, must match the pinned manifest, and are opened only
|
||||
after `ModelStore` verifies their exact size and SHA-256 digest. Install an
|
||||
ONNX Runtime shared library separately and set `ORT_DYLIB_PATH` when it is not
|
||||
available through the platform library search path. The feature currently
|
||||
requires Rust 1.95 or newer, matching OAR 0.9.1's MSRV.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = ["ocr-oar", "render-pdfium"] }
|
||||
```
|
||||
|
||||
Direct engine invocation is intentionally separate from extraction routing and
|
||||
native/OCR fusion:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
ModelDownloadPolicy, ModelStore, OarOcrEngine, OcrEngine, OcrMode,
|
||||
OcrOptions, PdfiumRenderer, RenderOptions, PP_OCR_V6_SMALL,
|
||||
};
|
||||
|
||||
let options = OcrOptions::new()
|
||||
.mode(OcrMode::Force)
|
||||
.minimum_confidence(0.45)
|
||||
.model_directory("/opt/firecrawl/models/pp-ocrv6-small")
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
let models = ModelStore::from_options(&options)?.resolve(&PP_OCR_V6_SMALL)?;
|
||||
let engine = OarOcrEngine::from_models(&models)?;
|
||||
|
||||
let renderer = PdfiumRenderer::load()?;
|
||||
let bytes = std::fs::read("scan.pdf")?;
|
||||
let pages = renderer.render_pages(&bytes, &[1], None, &RenderOptions::new())?;
|
||||
let ocr_pages = engine.recognize(&pages, &options)?;
|
||||
|
||||
for span in &ocr_pages[0].spans {
|
||||
println!("{:.3}: {}", span.confidence, span.text);
|
||||
}
|
||||
```
|
||||
|
||||
The engine accepts renderer-neutral RGB, RGBA, and grayscale pages, preserves
|
||||
OAR's positioned quadrilaterals in bitmap coordinates, filters spans using
|
||||
`minimum_confidence`, and records the pinned model revision in every `OcrPage`.
|
||||
`OcrMode::Off` is rejected at the engine boundary so default options cannot run
|
||||
inference accidentally.
|
||||
|
||||
### Selective routing and lazy model acquisition
|
||||
|
||||
`route_ocr_pages` applies the existing detector/text-quality recommendations to
|
||||
the configured mode. `Auto` processes only recommended pages, `Force` processes
|
||||
all pages (or an explicit page selection), and `Off` always returns an empty
|
||||
route. `run_ocr_pages` renders only that route, checks that both dependencies
|
||||
preserve its order, and retains each bitmap's PDF transform for fusion.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = [
|
||||
"render-pdfium",
|
||||
"ocr-oar",
|
||||
"model-download",
|
||||
] }
|
||||
```
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
route_ocr_pages, run_ocr_pages, HttpModelDownloader, ModelStore,
|
||||
OarOcrEngine, OcrMode, OcrOptions, PdfiumRenderer, RenderOptions,
|
||||
PP_OCR_V6_SMALL,
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("scan.pdf")?;
|
||||
let extraction = pdf_inspector::extract_pages_markdown_mem(&bytes, None)?;
|
||||
let options = OcrOptions::new().mode(OcrMode::Auto);
|
||||
let routed = route_ocr_pages(
|
||||
options.mode,
|
||||
extraction.pages.len() as u32,
|
||||
&extraction.pages_needing_ocr,
|
||||
None,
|
||||
)?;
|
||||
|
||||
if !routed.is_empty() {
|
||||
// No HTTP request or model initialization occurs before this point.
|
||||
let store = ModelStore::from_options(&options)?;
|
||||
let models = store.resolve_or_download(
|
||||
&PP_OCR_V6_SMALL,
|
||||
options.model_downloads,
|
||||
&HttpModelDownloader::default(),
|
||||
)?;
|
||||
let run = run_ocr_pages(
|
||||
&PdfiumRenderer::load()?,
|
||||
&OarOcrEngine::from_models(&models)?,
|
||||
&bytes,
|
||||
&routed,
|
||||
None,
|
||||
&RenderOptions::new(),
|
||||
&options,
|
||||
)?;
|
||||
println!("OCR processed {} pages", run.pages.len());
|
||||
}
|
||||
```
|
||||
|
||||
The downloader accepts HTTPS only, checks a declared content length, caps the
|
||||
response stream to the pinned size plus one byte, and delegates final size and
|
||||
SHA-256 verification to `ModelStore`. The store serializes installation across
|
||||
processes and publishes completed artifacts atomically. Warm caches make no
|
||||
network calls; offline mode and explicit model directories never download.
|
||||
|
||||
### OCR Markdown assembly and native fusion
|
||||
|
||||
`fuse_ocr_pages` maps OCR polygons back into PDF coordinates and sends the
|
||||
result through pdf-inspector's existing deterministic reading-order, table,
|
||||
and Markdown pipeline. Pages whose native extraction was rejected use OCR
|
||||
output. When `Force` runs on a clean native page, normalized duplicate OCR
|
||||
blocks are removed and only additional image-backed text is retained.
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{fuse_ocr_pages, OcrFusionOptions};
|
||||
|
||||
let fused = fuse_ocr_pages(
|
||||
&extraction.pages,
|
||||
&run,
|
||||
extraction.pages.len() as u32,
|
||||
&OcrFusionOptions::new().render_dpi(150.0),
|
||||
)?;
|
||||
|
||||
for page in &fused.pages {
|
||||
println!("{}", page.markdown);
|
||||
if page.provenance.hosted_recommended {
|
||||
eprintln!("page {} needs the hosted document pipeline", page.page + 1);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Each page carries `Native`, `Ocr`, or `Fused` provenance, the exact OCR model
|
||||
revision, accepted-page confidence, local stage timings, and non-fatal
|
||||
warnings. A page that required OCR recommends the hosted pipeline when local
|
||||
OCR is missing, empty, or below the configurable page-confidence threshold.
|
||||
This keeps the lightweight path explicit about cases it cannot finish well.
|
||||
|
||||
### Complete OCR API
|
||||
|
||||
The `ocr` convenience feature enables the renderer, OCR engine, verified
|
||||
model acquisition, routing, and fusion layers together. It is the intended
|
||||
downstream application integration boundary; lower-level features remain
|
||||
available for consumers that bring their own renderer, model package manager,
|
||||
or engine.
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { version = "1", features = ["ocr"] }
|
||||
```
|
||||
|
||||
```rust
|
||||
use pdf_inspector::vision::{
|
||||
process_pdf_with_ocr, OcrMode, OcrPdfOptions,
|
||||
};
|
||||
|
||||
let result = process_pdf_with_ocr(
|
||||
"document.pdf",
|
||||
OcrPdfOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.pages([1, 2, 3]),
|
||||
)?;
|
||||
|
||||
println!("{}", result.markdown);
|
||||
println!("OCR pages: {:?}", result.pages_routed_to_ocr);
|
||||
println!(
|
||||
"Hosted fallback pages: {:?}",
|
||||
result.pages_recommending_hosted,
|
||||
);
|
||||
```
|
||||
|
||||
Native extraction always runs first. In `Auto`, a clean PDF returns before
|
||||
PDFium loading, model-cache access, HTTP, or OAR initialization. Model files
|
||||
remain external and the default crate feature set remains unchanged. `Off`
|
||||
provides the same native-only behavior through the OCR result/provenance
|
||||
shape; `Force` renders every selected page. Learned layout intentionally
|
||||
returns an explicit unsupported error in this lightweight pipeline.
|
||||
|
||||
Build the CLI with the same opt-in feature:
|
||||
|
||||
```bash
|
||||
cargo build --release --features ocr --bin pdf2md
|
||||
pdf2md document.pdf --ocr auto --raw
|
||||
pdf2md document.pdf --ocr auto --json
|
||||
pdf2md document.pdf --ocr auto --ocr-offline --ocr-model-dir /opt/models/pp-ocrv6-small
|
||||
```
|
||||
|
||||
CLI controls include `--ocr-dpi`, `--ocr-min-confidence`,
|
||||
`--ocr-hosted-threshold`, `--select-pages`, and the existing encrypted-PDF
|
||||
`--password` option. JSON output includes per-page Markdown, source/model
|
||||
provenance, confidence, timings, warnings, routed pages, and hosted-fallback
|
||||
recommendations. Page numbers in `OcrPdfResult` and its per-page provenance
|
||||
are 1-indexed, matching the PDF page numbers accepted by `OcrPdfOptions::pages`.
|
||||
|
||||
Extract per-page Markdown (one string per page, plus document-wide layout
|
||||
metadata):
|
||||
|
||||
|
||||
+278
-5
@@ -1,6 +1,11 @@
|
||||
//! CLI tool for PDF to Markdown conversion
|
||||
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
use pdf_inspector::vision::{
|
||||
process_pdf_with_ocr, ModelDownloadPolicy, OcrMode, OcrOptions, OcrPdfOptions, OcrPdfResult,
|
||||
PageContentSource, RenderOptions,
|
||||
};
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages_with_password, process_pdf_with_options, LayoutComplexity,
|
||||
PdfOptions, PdfType, ProcessMode, TextItem,
|
||||
@@ -103,6 +108,134 @@ fn format_items_json(items: &[TextItem]) -> String {
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
fn optional_json_number(value: Option<f32>) -> String {
|
||||
value
|
||||
.filter(|value| value.is_finite())
|
||||
.map(|value| format!("{value:.4}"))
|
||||
.unwrap_or_else(|| "null".to_string())
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
fn format_ocr_json(result: &OcrPdfResult) -> String {
|
||||
let routed = result
|
||||
.pages_routed_to_ocr
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let recommended = result
|
||||
.pages_recommended_for_ocr
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let hosted = result
|
||||
.pages_recommending_hosted
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let pages = result
|
||||
.pages
|
||||
.iter()
|
||||
.map(|page| {
|
||||
let provenance = &page.provenance;
|
||||
let source = match provenance.source {
|
||||
PageContentSource::Native => "native",
|
||||
PageContentSource::Ocr => "ocr",
|
||||
PageContentSource::Fused => "fused",
|
||||
_ => "unknown",
|
||||
};
|
||||
let model = provenance
|
||||
.ocr_model
|
||||
.as_ref()
|
||||
.map(|model| {
|
||||
format!(
|
||||
r#"{{"name":"{}","revision":"{}"}}"#,
|
||||
json_escape(&model.name),
|
||||
json_escape(&model.revision)
|
||||
)
|
||||
})
|
||||
.unwrap_or_else(|| "null".to_string());
|
||||
let warnings = provenance
|
||||
.warnings
|
||||
.iter()
|
||||
.map(|warning| format!(r#""{}""#, json_escape(warning)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(
|
||||
r#"{{"page":{},"source":"{}","markdown":"{}","ocr_model":{},"render_dpi":{},"ocr_confidence":{},"hosted_recommended":{},"timings":{{"render_ms":{},"ocr_ms":{},"layout_ms":{},"assembly_ms":{}}},"warnings":[{}]}}"#,
|
||||
provenance.page,
|
||||
source,
|
||||
json_escape(&page.markdown),
|
||||
model,
|
||||
optional_json_number(provenance.render_dpi),
|
||||
optional_json_number(provenance.ocr_confidence),
|
||||
provenance.hosted_recommended,
|
||||
provenance.timings.render_ms,
|
||||
provenance.timings.ocr_ms,
|
||||
provenance.timings.layout_ms,
|
||||
provenance.timings.assembly_ms,
|
||||
warnings,
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let table_pages = result
|
||||
.pages_with_tables
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let column_pages = result
|
||||
.pages_with_columns
|
||||
.iter()
|
||||
.map(u32::to_string)
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
format!(
|
||||
r#"{{"page_count":{},"processing_time_ms":{},"render_time_ms":{},"ocr_time_ms":{},"pages_recommended_for_ocr":[{}],"pages_routed_to_ocr":[{}],"pages_recommending_hosted":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"pages":[{}],"markdown":"{}"}}"#,
|
||||
result.page_count,
|
||||
result.processing_time_ms,
|
||||
result.render_time_ms,
|
||||
result.ocr_time_ms,
|
||||
recommended,
|
||||
routed,
|
||||
hosted,
|
||||
ocr_reasons,
|
||||
result.is_complex,
|
||||
table_pages,
|
||||
column_pages,
|
||||
pages,
|
||||
json_escape(&result.markdown),
|
||||
)
|
||||
}
|
||||
|
||||
fn argument_value<'a>(args: &'a [String], name: &str) -> Result<Option<&'a str>, String> {
|
||||
args.iter()
|
||||
.position(|argument| argument == name)
|
||||
.map(|index| {
|
||||
args.get(index + 1)
|
||||
.map(String::as_str)
|
||||
.ok_or_else(|| format!("{name} requires a value"))
|
||||
})
|
||||
.transpose()
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
fn float_argument(args: &[String], name: &str, default: f32) -> Result<f32, String> {
|
||||
argument_value(args, name)?
|
||||
.map(|value| {
|
||||
value
|
||||
.parse::<f32>()
|
||||
.map_err(|_| format!("{name} requires a number, got {value:?}"))
|
||||
})
|
||||
.transpose()
|
||||
.map(|value| value.unwrap_or(default))
|
||||
}
|
||||
|
||||
fn extract_items_json(
|
||||
pdf_path: &str,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
@@ -242,6 +375,12 @@ fn main() {
|
||||
eprintln!(" --password PW Password for an encrypted PDF");
|
||||
eprintln!(" --detect-only Only detect PDF type (no extraction)");
|
||||
eprintln!(" --analyze Detect + extract + layout analysis (no markdown)");
|
||||
eprintln!(" --ocr MODE OCR mode: off, auto, or force (requires feature `ocr`)");
|
||||
eprintln!(" --ocr-dpi N OCR render resolution (default: 150)");
|
||||
eprintln!(" --ocr-min-confidence N Drop OCR spans below N (default: 0)");
|
||||
eprintln!(" --ocr-hosted-threshold N Recommend hosted parsing below N (default: 0.5)");
|
||||
eprintln!(" --ocr-model-dir DIR Use a package-managed local model directory");
|
||||
eprintln!(" --ocr-offline Never download missing OCR models");
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
@@ -253,6 +392,10 @@ fn main() {
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
let analyze = args.iter().any(|a| a == "--analyze");
|
||||
let ocr_mode_argument = argument_value(&args, "--ocr").unwrap_or_else(|error| {
|
||||
eprintln!("Error: {error}");
|
||||
process::exit(1);
|
||||
});
|
||||
|
||||
// Parse --password value
|
||||
let password = args.iter().position(|a| a == "--password").map(|i| {
|
||||
@@ -283,6 +426,141 @@ fn main() {
|
||||
})
|
||||
});
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
.map(|s| s.as_str());
|
||||
|
||||
let has_ocr_only_option = [
|
||||
"--ocr-dpi",
|
||||
"--ocr-min-confidence",
|
||||
"--ocr-hosted-threshold",
|
||||
"--ocr-model-dir",
|
||||
"--ocr-offline",
|
||||
]
|
||||
.iter()
|
||||
.any(|option| args.iter().any(|argument| argument == option));
|
||||
if ocr_mode_argument.is_none() && has_ocr_only_option {
|
||||
eprintln!("Error: OCR options require --ocr off, --ocr auto, or --ocr force");
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
if let Some(mode) = ocr_mode_argument {
|
||||
if items_json_output || detect_only || analyze {
|
||||
eprintln!(
|
||||
"Error: --ocr cannot be combined with --items-json, --detect-only, or --analyze"
|
||||
);
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
#[cfg(not(all(feature = "ocr", not(target_arch = "wasm32"))))]
|
||||
{
|
||||
let _ = mode;
|
||||
eprintln!("Error: this pdf2md build does not include OCR; rebuild with --features ocr");
|
||||
process::exit(1);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
{
|
||||
let mode = match mode {
|
||||
"off" => OcrMode::Off,
|
||||
"auto" => OcrMode::Auto,
|
||||
"force" => OcrMode::Force,
|
||||
value => {
|
||||
eprintln!("Error: invalid --ocr mode {value:?}; expected off, auto, or force");
|
||||
process::exit(1);
|
||||
}
|
||||
};
|
||||
let dpi = float_argument(&args, "--ocr-dpi", 150.0).unwrap_or_else(|error| {
|
||||
eprintln!("Error: {error}");
|
||||
process::exit(1);
|
||||
});
|
||||
let minimum_confidence = float_argument(&args, "--ocr-min-confidence", 0.0)
|
||||
.unwrap_or_else(|error| {
|
||||
eprintln!("Error: {error}");
|
||||
process::exit(1);
|
||||
});
|
||||
let hosted_threshold = float_argument(&args, "--ocr-hosted-threshold", 0.5)
|
||||
.unwrap_or_else(|error| {
|
||||
eprintln!("Error: {error}");
|
||||
process::exit(1);
|
||||
});
|
||||
let model_directory =
|
||||
argument_value(&args, "--ocr-model-dir").unwrap_or_else(|error| {
|
||||
eprintln!("Error: {error}");
|
||||
process::exit(1);
|
||||
});
|
||||
|
||||
let mut ocr = OcrOptions::new()
|
||||
.mode(mode)
|
||||
.minimum_confidence(minimum_confidence);
|
||||
if let Some(directory) = model_directory {
|
||||
ocr = ocr.model_directory(directory);
|
||||
}
|
||||
if args.iter().any(|argument| argument == "--ocr-offline") {
|
||||
ocr = ocr.model_downloads(ModelDownloadPolicy::Offline);
|
||||
}
|
||||
let mut markdown = pdf_inspector::MarkdownOptions::default();
|
||||
if compact_output {
|
||||
markdown.profile = pdf_inspector::MarkdownProfile::Compact;
|
||||
}
|
||||
markdown.include_page_numbers = page_numbers;
|
||||
let mut pdf_options = OcrPdfOptions::new()
|
||||
.render(RenderOptions::new().dpi(dpi))
|
||||
.ocr(ocr)
|
||||
.markdown(markdown)
|
||||
.hosted_recommendation_confidence(hosted_threshold);
|
||||
if let Some(pages) = page_filter.clone() {
|
||||
pdf_options = pdf_options.pages(pages);
|
||||
}
|
||||
if let Some(password) = password.clone() {
|
||||
pdf_options = pdf_options.password(password);
|
||||
}
|
||||
|
||||
match process_pdf_with_ocr(pdf_path, pdf_options) {
|
||||
Ok(result) => {
|
||||
if json_output {
|
||||
println!("{}", format_ocr_json(&result));
|
||||
} else if raw_output {
|
||||
print!("{}", result.markdown);
|
||||
} else {
|
||||
eprintln!("PDF to Markdown Conversion (OCR)");
|
||||
eprintln!("======================================");
|
||||
eprintln!("File: {pdf_path}");
|
||||
eprintln!("Pages: {}", result.page_count);
|
||||
eprintln!("Pages routed to OCR: {:?}", result.pages_routed_to_ocr);
|
||||
if !result.pages_recommending_hosted.is_empty() {
|
||||
eprintln!(
|
||||
"Hosted parsing recommended for pages: {:?}",
|
||||
result.pages_recommending_hosted
|
||||
);
|
||||
}
|
||||
eprintln!("Processing time: {}ms", result.processing_time_ms);
|
||||
if let Some(output) = output_file {
|
||||
fs::write(output, &result.markdown)
|
||||
.expect("Failed to write output file");
|
||||
eprintln!("Markdown written to: {output}");
|
||||
} else {
|
||||
eprintln!();
|
||||
eprintln!("--- Markdown Output ---");
|
||||
eprintln!();
|
||||
print!("{}", result.markdown);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(error) => {
|
||||
if json_output {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&error.to_string()));
|
||||
} else {
|
||||
eprintln!("Error: {error}");
|
||||
}
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if items_json_output {
|
||||
match extract_items_json(pdf_path, page_filter.as_ref(), password.as_deref()) {
|
||||
Ok(json) => println!("{}", json),
|
||||
@@ -294,11 +572,6 @@ fn main() {
|
||||
return;
|
||||
}
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
.map(|s| s.as_str());
|
||||
|
||||
let process_mode = if detect_only {
|
||||
ProcessMode::DetectOnly
|
||||
} else if analyze {
|
||||
|
||||
+162
-11
@@ -43,6 +43,7 @@ mod text_quality;
|
||||
pub mod text_utils;
|
||||
pub mod tounicode;
|
||||
pub mod types;
|
||||
pub mod vision;
|
||||
|
||||
pub use detector::{
|
||||
detect_pdf_type, detect_pdf_type_mem, detect_pdf_type_mem_with_config,
|
||||
@@ -458,8 +459,35 @@ pub fn extract_pages_markdown_mem(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<PagesExtractionResult, PdfError> {
|
||||
extract_pages_markdown_mem_impl(buffer, pages, None, &MarkdownOptions::default(), false)
|
||||
.map(|(result, _)| result)
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
pub(crate) fn extract_pages_markdown_mem_for_ocr(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
password: Option<&str>,
|
||||
markdown_options: &MarkdownOptions,
|
||||
) -> Result<(PagesExtractionResult, u32), PdfError> {
|
||||
extract_pages_markdown_mem_impl(
|
||||
buffer,
|
||||
pages,
|
||||
password,
|
||||
markdown_options,
|
||||
markdown_options.strip_headers_footers,
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_pages_markdown_mem_impl(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
password: Option<&str>,
|
||||
markdown_options: &MarkdownOptions,
|
||||
strip_repeated_headers_footers: bool,
|
||||
) -> Result<(PagesExtractionResult, u32), PdfError> {
|
||||
validate_pdf_bytes(buffer)?;
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let (doc, page_count) = load_document_from_mem_with_password(buffer, password)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
@@ -502,6 +530,11 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
let repeated_header_footer_items = if strip_repeated_headers_footers {
|
||||
repeated_header_footer_item_keys(&all_items, &page_thresholds, &chart_regions, page_count)
|
||||
} else {
|
||||
HashSet::new()
|
||||
};
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -537,7 +570,10 @@ pub fn extract_pages_markdown_mem(
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.filter(|(item, _)| {
|
||||
item.page == page_1idx
|
||||
&& !repeated_header_footer_items.contains(&HeaderFooterItemKey::from(*item))
|
||||
})
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
@@ -574,7 +610,7 @@ pub fn extract_pages_markdown_mem(
|
||||
base_font_size: Some(font_stats.most_common_size),
|
||||
include_page_numbers: false,
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
..markdown_options.clone()
|
||||
};
|
||||
|
||||
let md = if has_text_quality_issue {
|
||||
@@ -633,14 +669,129 @@ pub fn extract_pages_markdown_mem(
|
||||
});
|
||||
}
|
||||
|
||||
Ok(PagesExtractionResult {
|
||||
pages: results,
|
||||
pages_with_tables: complexity.pages_with_tables,
|
||||
pages_with_columns: complexity.pages_with_columns,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(ocr_reasons_by_page),
|
||||
is_complex: complexity.is_complex,
|
||||
})
|
||||
Ok((
|
||||
PagesExtractionResult {
|
||||
pages: results,
|
||||
pages_with_tables: complexity.pages_with_tables,
|
||||
pages_with_columns: complexity.pages_with_columns,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(ocr_reasons_by_page),
|
||||
is_complex: complexity.is_complex,
|
||||
},
|
||||
page_count,
|
||||
))
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
|
||||
struct HeaderFooterItemKey {
|
||||
page: u32,
|
||||
x: u32,
|
||||
y: u32,
|
||||
text: String,
|
||||
}
|
||||
|
||||
impl From<&TextItem> for HeaderFooterItemKey {
|
||||
fn from(item: &TextItem) -> Self {
|
||||
Self {
|
||||
page: item.page,
|
||||
x: item.x.to_bits(),
|
||||
y: item.y.to_bits(),
|
||||
text: item.text.clone(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn repeated_header_footer_item_keys(
|
||||
items: &[TextItem],
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
page_count: u32,
|
||||
) -> HashSet<HeaderFooterItemKey> {
|
||||
let candidates = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
matches!(
|
||||
item.item_type,
|
||||
types::ItemType::Text | types::ItemType::FormField
|
||||
)
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
let lines = extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
candidates,
|
||||
page_thresholds,
|
||||
&HashSet::new(),
|
||||
chart_regions,
|
||||
);
|
||||
let all_items: HashSet<_> = lines
|
||||
.iter()
|
||||
.flat_map(|line| line.items.iter().map(HeaderFooterItemKey::from))
|
||||
.collect();
|
||||
let kept = markdown::strip_repeated_header_footer_lines(lines, page_count);
|
||||
let kept_items: HashSet<_> = kept
|
||||
.iter()
|
||||
.flat_map(|line| line.items.iter().map(HeaderFooterItemKey::from))
|
||||
.collect();
|
||||
all_items.difference(&kept_items).cloned().collect()
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "ocr", not(target_arch = "wasm32")))]
|
||||
mod ocr_header_footer_tests {
|
||||
use super::*;
|
||||
|
||||
fn item(page: u32, text: &str, y: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x: 10.0,
|
||||
y,
|
||||
width: 120.0,
|
||||
height: 10.0,
|
||||
font: "Test".to_string(),
|
||||
font_size: 10.0,
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: types::ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn local_pipeline_prefilters_document_wide_repeated_headers() {
|
||||
let mut items = Vec::new();
|
||||
let mut thresholds = HashMap::new();
|
||||
for page in 1..=3 {
|
||||
items.push(item(page, "Repeated report header", 800.0));
|
||||
for line in 0..12 {
|
||||
items.push(item(
|
||||
page,
|
||||
&format!("Page {page} paragraph {line} unique content"),
|
||||
700.0 - line as f32 * 40.0,
|
||||
));
|
||||
}
|
||||
thresholds.insert(page, 0.1);
|
||||
}
|
||||
|
||||
let removed = repeated_header_footer_item_keys(&items, &thresholds, &HashMap::new(), 3);
|
||||
assert_eq!(removed.len(), 2);
|
||||
for page in 1..=3 {
|
||||
assert_eq!(
|
||||
removed.contains(&HeaderFooterItemKey::from(&item(
|
||||
page,
|
||||
"Repeated report header",
|
||||
800.0,
|
||||
))),
|
||||
page > 1,
|
||||
);
|
||||
assert!(!removed.contains(&HeaderFooterItemKey::from(&item(
|
||||
page,
|
||||
&format!("Page {page} paragraph 5 unique content"),
|
||||
500.0,
|
||||
))));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Path-based wrapper for [`extract_pages_markdown_mem`].
|
||||
|
||||
@@ -1144,6 +1144,14 @@ pub fn to_markdown(text: &str, options: MarkdownOptions) -> String {
|
||||
output
|
||||
}
|
||||
|
||||
/// Applies the document-wide repeated header/footer classifier to grouped lines.
|
||||
pub(crate) fn strip_repeated_header_footer_lines(
|
||||
lines: Vec<crate::types::TextLine>,
|
||||
page_count: u32,
|
||||
) -> Vec<crate::types::TextLine> {
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown with structure detection
|
||||
pub fn to_markdown_from_items(items: Vec<TextItem>, options: MarkdownOptions) -> String {
|
||||
to_markdown_from_items_with_rects(items, options, &[])
|
||||
|
||||
+2
-2
@@ -604,7 +604,7 @@ mod tests {
|
||||
let items: Vec<(usize, &TextItem)> = vec![];
|
||||
assert_eq!(
|
||||
find_column_boundaries(&items, TableDetectionMode::SmallFont),
|
||||
vec![]
|
||||
Vec::<f32>::new()
|
||||
);
|
||||
}
|
||||
|
||||
@@ -661,7 +661,7 @@ mod tests {
|
||||
#[test]
|
||||
fn test_find_row_boundaries_empty() {
|
||||
let items: Vec<(usize, &TextItem)> = vec![];
|
||||
assert_eq!(find_row_boundaries(&items), vec![]);
|
||||
assert_eq!(find_row_boundaries(&items), Vec::<f32>::new());
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -0,0 +1,414 @@
|
||||
//! Public contracts between rendering, OCR, layout, and orchestration.
|
||||
|
||||
use std::error::Error;
|
||||
use std::path::PathBuf;
|
||||
|
||||
use super::{RenderOptions, RenderedPage};
|
||||
|
||||
/// Selects when OCR may run.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrMode {
|
||||
/// Never run OCR. This is the default and preserves existing behavior.
|
||||
#[default]
|
||||
Off,
|
||||
/// Run OCR only on pages selected by pdf-inspector's OCR routing signals.
|
||||
Auto,
|
||||
/// Run OCR on every selected page, including pages with native text.
|
||||
Force,
|
||||
}
|
||||
|
||||
/// Resource/quality profile for the OCR engine.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrProfile {
|
||||
/// Lowest latency and memory footprint.
|
||||
Edge,
|
||||
/// OCR-oriented balance of quality and CPU cost.
|
||||
#[default]
|
||||
Balanced,
|
||||
/// Highest quality within the lightweight model family.
|
||||
Quality,
|
||||
}
|
||||
|
||||
/// Controls whether missing model artifacts may be fetched.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum ModelDownloadPolicy {
|
||||
/// Fetch a pinned artifact only after OCR has actually been selected.
|
||||
#[default]
|
||||
IfMissing,
|
||||
/// Never access the network; require an override or a warm model cache.
|
||||
Offline,
|
||||
}
|
||||
|
||||
/// OCR engine configuration independent of a particular runtime.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct OcrOptions {
|
||||
/// Page-level routing behavior.
|
||||
pub mode: OcrMode,
|
||||
/// Local quality/resource profile.
|
||||
pub profile: OcrProfile,
|
||||
/// Drop recognition spans below this confidence threshold.
|
||||
pub minimum_confidence: f32,
|
||||
/// Optional language hints understood by the selected engine.
|
||||
pub languages: Vec<String>,
|
||||
/// Optional directory containing an offline model set.
|
||||
pub model_directory: Option<PathBuf>,
|
||||
/// Whether a missing pinned artifact may be downloaded.
|
||||
pub model_downloads: ModelDownloadPolicy,
|
||||
}
|
||||
|
||||
impl Default for OcrOptions {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
mode: OcrMode::Off,
|
||||
profile: OcrProfile::Balanced,
|
||||
minimum_confidence: 0.0,
|
||||
languages: Vec::new(),
|
||||
model_directory: None,
|
||||
model_downloads: ModelDownloadPolicy::IfMissing,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl OcrOptions {
|
||||
/// Creates OCR options with OCR disabled.
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Sets page-level OCR routing.
|
||||
pub fn mode(mut self, mode: OcrMode) -> Self {
|
||||
self.mode = mode;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the local resource/quality profile.
|
||||
pub fn profile(mut self, profile: OcrProfile) -> Self {
|
||||
self.profile = profile;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the minimum accepted recognition confidence.
|
||||
pub fn minimum_confidence(mut self, minimum_confidence: f32) -> Self {
|
||||
self.minimum_confidence = minimum_confidence;
|
||||
self
|
||||
}
|
||||
|
||||
/// Replaces the language hints passed to the OCR engine.
|
||||
pub fn languages(mut self, languages: impl IntoIterator<Item = impl Into<String>>) -> Self {
|
||||
self.languages = languages.into_iter().map(Into::into).collect();
|
||||
self
|
||||
}
|
||||
|
||||
/// Uses an explicit model directory, suitable for offline packaging.
|
||||
pub fn model_directory(mut self, directory: impl Into<PathBuf>) -> Self {
|
||||
self.model_directory = Some(directory.into());
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the missing-model download policy.
|
||||
pub fn model_downloads(mut self, policy: ModelDownloadPolicy) -> Self {
|
||||
self.model_downloads = policy;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// Configuration for an optional learned layout engine.
|
||||
///
|
||||
/// Layout inference is disabled by default. Existing deterministic layout,
|
||||
/// table, and Markdown logic remains the assembly path when this is disabled.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct LayoutOptions {
|
||||
/// Whether the learned layout extension may run.
|
||||
pub enabled: bool,
|
||||
/// Drop layout regions below this confidence threshold.
|
||||
pub minimum_confidence: f32,
|
||||
/// Optional directory containing an offline layout model set.
|
||||
pub model_directory: Option<PathBuf>,
|
||||
}
|
||||
|
||||
impl Default for LayoutOptions {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
enabled: false,
|
||||
minimum_confidence: 0.0,
|
||||
model_directory: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl LayoutOptions {
|
||||
/// Creates layout options with learned layout disabled.
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Enables or disables learned layout inference.
|
||||
pub fn enabled(mut self, enabled: bool) -> Self {
|
||||
self.enabled = enabled;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the minimum accepted region confidence.
|
||||
pub fn minimum_confidence(mut self, minimum_confidence: f32) -> Self {
|
||||
self.minimum_confidence = minimum_confidence;
|
||||
self
|
||||
}
|
||||
|
||||
/// Uses an explicit layout model directory.
|
||||
pub fn model_directory(mut self, directory: impl Into<PathBuf>) -> Self {
|
||||
self.model_directory = Some(directory.into());
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// A point in bitmap space, measured from the top-left in pixels.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq)]
|
||||
pub struct ImagePoint {
|
||||
/// Horizontal pixel coordinate.
|
||||
pub x: f32,
|
||||
/// Vertical pixel coordinate, increasing downward.
|
||||
pub y: f32,
|
||||
}
|
||||
|
||||
impl ImagePoint {
|
||||
/// Creates a bitmap-space point.
|
||||
pub fn new(x: f32, y: f32) -> Self {
|
||||
Self { x, y }
|
||||
}
|
||||
}
|
||||
|
||||
/// Four-point polygon in bitmap coordinates.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq)]
|
||||
pub struct ImageQuad {
|
||||
/// Polygon points in engine-provided order.
|
||||
pub points: [ImagePoint; 4],
|
||||
}
|
||||
|
||||
impl ImageQuad {
|
||||
/// Creates a four-point bitmap polygon.
|
||||
pub fn new(points: [ImagePoint; 4]) -> Self {
|
||||
Self { points }
|
||||
}
|
||||
}
|
||||
|
||||
/// Stable identity for an inference model used in output provenance.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct ModelIdentity {
|
||||
/// Model family/name, for example `pp-ocrv6-small`.
|
||||
pub name: String,
|
||||
/// Immutable model or artifact-set revision.
|
||||
pub revision: String,
|
||||
}
|
||||
|
||||
impl ModelIdentity {
|
||||
/// Creates a model identity.
|
||||
pub fn new(name: impl Into<String>, revision: impl Into<String>) -> Self {
|
||||
Self {
|
||||
name: name.into(),
|
||||
revision: revision.into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// One positioned OCR recognition result in bitmap coordinates.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct OcrSpan {
|
||||
/// Recognized text.
|
||||
pub text: String,
|
||||
/// Detection polygon in the original rendered page's pixel space.
|
||||
pub polygon: ImageQuad,
|
||||
/// Recognition confidence in the inclusive range 0–1.
|
||||
pub confidence: f32,
|
||||
/// Optional text-line orientation in clockwise degrees.
|
||||
pub orientation_degrees: Option<f32>,
|
||||
}
|
||||
|
||||
/// OCR output for one 1-indexed page.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct OcrPage {
|
||||
/// 1-indexed PDF page number.
|
||||
pub page: u32,
|
||||
/// Positioned recognition spans.
|
||||
pub spans: Vec<OcrSpan>,
|
||||
/// Mean confidence across accepted spans, when available.
|
||||
pub mean_confidence: Option<f32>,
|
||||
/// Exact model identity used for this result.
|
||||
pub model: ModelIdentity,
|
||||
/// OCR wall time for this page.
|
||||
pub processing_time_ms: u64,
|
||||
/// Non-fatal engine warnings.
|
||||
pub warnings: Vec<String>,
|
||||
}
|
||||
|
||||
/// Normalized semantic class emitted by a learned layout engine.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum LayoutRegionKind {
|
||||
/// Body or other prose text.
|
||||
Text,
|
||||
/// Document heading or title.
|
||||
Heading,
|
||||
/// Table region.
|
||||
Table,
|
||||
/// Figure/image region.
|
||||
Figure,
|
||||
/// Figure or table caption.
|
||||
Caption,
|
||||
/// Header/footer/page furniture.
|
||||
Furniture,
|
||||
/// Model-specific class retained without changing the common taxonomy.
|
||||
Other(String),
|
||||
}
|
||||
|
||||
/// One learned layout region in bitmap coordinates.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct LayoutRegion {
|
||||
/// Normalized semantic class.
|
||||
pub kind: LayoutRegionKind,
|
||||
/// Region polygon in the original rendered page's pixel space.
|
||||
pub polygon: ImageQuad,
|
||||
/// Model confidence in the inclusive range 0–1.
|
||||
pub confidence: f32,
|
||||
/// Optional model-provided reading-order position.
|
||||
pub reading_order: Option<u32>,
|
||||
}
|
||||
|
||||
/// Learned layout output for one 1-indexed page.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct LayoutPage {
|
||||
/// 1-indexed PDF page number.
|
||||
pub page: u32,
|
||||
/// Semantic regions.
|
||||
pub regions: Vec<LayoutRegion>,
|
||||
/// Exact model identity used for this result.
|
||||
pub model: ModelIdentity,
|
||||
/// Layout inference wall time for this page.
|
||||
pub processing_time_ms: u64,
|
||||
/// Non-fatal engine warnings.
|
||||
pub warnings: Vec<String>,
|
||||
}
|
||||
|
||||
/// How final page content was sourced.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum PageContentSource {
|
||||
/// Trusted native PDF text only.
|
||||
Native,
|
||||
/// OCR output only.
|
||||
Ocr,
|
||||
/// Native and OCR spans were fused.
|
||||
Fused,
|
||||
}
|
||||
|
||||
/// Per-page local processing timings.
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
pub struct VisionTimings {
|
||||
/// Rasterization wall time.
|
||||
pub render_ms: u64,
|
||||
/// OCR wall time.
|
||||
pub ocr_ms: u64,
|
||||
/// Optional learned layout wall time.
|
||||
pub layout_ms: u64,
|
||||
/// Native/OCR fusion and assembly wall time.
|
||||
pub assembly_ms: u64,
|
||||
}
|
||||
|
||||
/// Source and model metadata retained for one processed page.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct PageProvenance {
|
||||
/// 1-indexed PDF page number.
|
||||
pub page: u32,
|
||||
/// Final page-content source.
|
||||
pub source: PageContentSource,
|
||||
/// OCR model, when OCR ran.
|
||||
pub ocr_model: Option<ModelIdentity>,
|
||||
/// Learned layout model, when layout inference ran.
|
||||
pub layout_model: Option<ModelIdentity>,
|
||||
/// Render resolution used for local vision.
|
||||
pub render_dpi: Option<f32>,
|
||||
/// Mean accepted OCR confidence, when available.
|
||||
pub ocr_confidence: Option<f32>,
|
||||
/// Stage timings.
|
||||
pub timings: VisionTimings,
|
||||
/// Non-fatal warnings surfaced to downstream users.
|
||||
pub warnings: Vec<String>,
|
||||
/// True when this lightweight local path detected a case better suited to
|
||||
/// Firecrawl's hosted document pipeline.
|
||||
pub hosted_recommended: bool,
|
||||
}
|
||||
|
||||
/// Converts selected PDF pages into renderer-neutral owned bitmaps.
|
||||
pub trait PageRenderer: Send + Sync {
|
||||
/// Renderer-specific failure type.
|
||||
type Error: Error + Send + Sync + 'static;
|
||||
|
||||
/// Renders selected 1-indexed pages in the same order as `pages`.
|
||||
fn render_pages(
|
||||
&self,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
options: &RenderOptions,
|
||||
) -> Result<Vec<RenderedPage>, Self::Error>;
|
||||
}
|
||||
|
||||
/// Recognizes positioned text from rendered pages.
|
||||
pub trait OcrEngine: Send + Sync {
|
||||
/// Engine-specific failure type.
|
||||
type Error: Error + Send + Sync + 'static;
|
||||
|
||||
/// Exact model identity used by this engine instance.
|
||||
fn model(&self) -> &ModelIdentity;
|
||||
|
||||
/// Recognizes pages in batch and returns results in input order.
|
||||
fn recognize(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
options: &OcrOptions,
|
||||
) -> Result<Vec<OcrPage>, Self::Error>;
|
||||
}
|
||||
|
||||
/// Optional learned semantic layout extension.
|
||||
pub trait LayoutEngine: Send + Sync {
|
||||
/// Engine-specific failure type.
|
||||
type Error: Error + Send + Sync + 'static;
|
||||
|
||||
/// Exact model identity used by this engine instance.
|
||||
fn model(&self) -> &ModelIdentity;
|
||||
|
||||
/// Analyzes rendered pages, optionally using their OCR spans.
|
||||
fn analyze(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
ocr: &[OcrPage],
|
||||
options: &LayoutOptions,
|
||||
) -> Result<Vec<LayoutPage>, Self::Error>;
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn ocr_defaults_never_enable_recognition() {
|
||||
let options = OcrOptions::default();
|
||||
assert_eq!(options.mode, OcrMode::Off);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn offline_model_override_is_explicit() {
|
||||
let options = OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.model_directory("/models/pp-ocr")
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
assert_eq!(options.mode, OcrMode::Auto);
|
||||
assert_eq!(options.model_downloads, ModelDownloadPolicy::Offline);
|
||||
assert_eq!(
|
||||
options.model_directory,
|
||||
Some(PathBuf::from("/models/pp-ocr"))
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
//! HTTPS acquisition for pinned local model artifacts.
|
||||
|
||||
use std::fmt;
|
||||
use std::io::Read;
|
||||
use std::time::Duration;
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{ModelArtifact, ModelDownloader};
|
||||
|
||||
/// Default end-to-end timeout for one model artifact request.
|
||||
pub const DEFAULT_MODEL_DOWNLOAD_TIMEOUT: Duration = Duration::from_secs(5 * 60);
|
||||
|
||||
/// Streaming HTTPS downloader used by lazy model resolution.
|
||||
#[derive(Clone)]
|
||||
pub struct HttpModelDownloader {
|
||||
agent: ureq::Agent,
|
||||
}
|
||||
|
||||
impl fmt::Debug for HttpModelDownloader {
|
||||
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
formatter
|
||||
.debug_struct("HttpModelDownloader")
|
||||
.finish_non_exhaustive()
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for HttpModelDownloader {
|
||||
fn default() -> Self {
|
||||
Self::new(DEFAULT_MODEL_DOWNLOAD_TIMEOUT)
|
||||
}
|
||||
}
|
||||
|
||||
impl HttpModelDownloader {
|
||||
/// Creates an HTTPS-only downloader with an end-to-end request timeout.
|
||||
pub fn new(timeout: Duration) -> Self {
|
||||
let config = ureq::Agent::config_builder()
|
||||
.https_only(true)
|
||||
.timeout_global(Some(timeout))
|
||||
.user_agent(concat!("pdf-inspector/", env!("CARGO_PKG_VERSION")))
|
||||
.build();
|
||||
Self {
|
||||
agent: ureq::Agent::new_with_config(config),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl ModelDownloader for HttpModelDownloader {
|
||||
type Error = HttpModelDownloadError;
|
||||
|
||||
fn open(&self, artifact: &ModelArtifact) -> Result<Box<dyn Read + Send>, Self::Error> {
|
||||
let response = self.agent.get(artifact.url).call()?;
|
||||
if let Some(actual) = response.body().content_length() {
|
||||
if actual != artifact.size {
|
||||
return Err(HttpModelDownloadError::ContentLength {
|
||||
expected: artifact.size,
|
||||
actual,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// One extra byte lets ModelStore report an exact size mismatch while
|
||||
// preventing a malicious or broken server from filling the disk.
|
||||
let limit = artifact.size.saturating_add(1);
|
||||
Ok(Box::new(response.into_body().into_reader().take(limit)))
|
||||
}
|
||||
}
|
||||
|
||||
/// Failures before a response stream reaches [`super::ModelStore`].
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum HttpModelDownloadError {
|
||||
/// DNS, TLS, redirect, HTTP status, or response-stream setup failed.
|
||||
#[error(transparent)]
|
||||
Request(#[from] ureq::Error),
|
||||
/// The server declared a size that disagrees with the pinned manifest.
|
||||
#[error("server declared {actual} bytes; manifest requires {expected}")]
|
||||
ContentLength {
|
||||
/// Pinned artifact size.
|
||||
expected: u64,
|
||||
/// Server-declared size.
|
||||
actual: u64,
|
||||
},
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,66 @@
|
||||
//! Optional native vision primitives used by OCR pipelines.
|
||||
//!
|
||||
//! The existing lopdf extractor remains the default path. Native page
|
||||
//! rendering is available only with the `render-pdfium` feature. Engine
|
||||
//! contracts are available with `vision`, while checksum-verified model
|
||||
//! resolution is a separate `model-cache` feature. The `ocr-oar` feature adds
|
||||
//! a CPU PP-OCRv6 Small implementation of [`OcrEngine`]. These remain separate
|
||||
//! so browser WASM, text-only consumers, and renderer-only users take on no
|
||||
//! model-management or inference dependencies.
|
||||
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod contracts;
|
||||
#[cfg(all(feature = "model-download", not(target_arch = "wasm32")))]
|
||||
mod download;
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod fusion;
|
||||
#[cfg(all(feature = "model-cache", not(target_arch = "wasm32")))]
|
||||
mod models;
|
||||
#[cfg(all(feature = "ocr-oar", not(target_arch = "wasm32")))]
|
||||
mod oar;
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
mod pipeline;
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod render;
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
mod routing;
|
||||
|
||||
#[cfg(all(feature = "render-pdfium", not(target_arch = "wasm32")))]
|
||||
mod pdfium;
|
||||
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use contracts::{
|
||||
ImagePoint, ImageQuad, LayoutEngine, LayoutOptions, LayoutPage, LayoutRegion, LayoutRegionKind,
|
||||
ModelDownloadPolicy, ModelIdentity, OcrEngine, OcrMode, OcrOptions, OcrPage, OcrProfile,
|
||||
OcrSpan, PageContentSource, PageProvenance, PageRenderer, VisionTimings,
|
||||
};
|
||||
#[cfg(all(feature = "model-download", not(target_arch = "wasm32")))]
|
||||
pub use download::{HttpModelDownloadError, HttpModelDownloader, DEFAULT_MODEL_DOWNLOAD_TIMEOUT};
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use fusion::{
|
||||
fuse_ocr_pages, ocr_page_to_markdown, FusedPageMarkdown, FusedPages, OcrFusionError,
|
||||
OcrFusionOptions,
|
||||
};
|
||||
#[cfg(all(feature = "model-cache", not(target_arch = "wasm32")))]
|
||||
pub use models::{
|
||||
ModelAcquireError, ModelArtifact, ModelArtifactKind, ModelDownloader, ModelManifest,
|
||||
ModelPaths, ModelStore, ModelStoreError, PP_OCR_V6_SMALL,
|
||||
};
|
||||
#[cfg(all(feature = "ocr-oar", not(target_arch = "wasm32")))]
|
||||
pub use oar::{OarOcrEngine, OarOcrError, ONNX_RUNTIME_LIBRARY_ENV};
|
||||
#[cfg(all(feature = "ocr", not(target_arch = "wasm32")))]
|
||||
pub use pipeline::{
|
||||
process_pdf_with_ocr, process_pdf_with_ocr_mem, OcrPdfOptions, OcrPdfResult, OcrPipelineError,
|
||||
};
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use render::{
|
||||
PagePoint, PageTransform, RenderBufferError, RenderOptions, RenderPixelFormat, RenderedPage,
|
||||
DEFAULT_RENDER_DPI,
|
||||
};
|
||||
#[cfg(all(feature = "vision", not(target_arch = "wasm32")))]
|
||||
pub use routing::{
|
||||
route_ocr_pages, run_ocr_pages, OcrRoutingError, OcrRun, OcrRunError, RoutedOcrPage,
|
||||
};
|
||||
|
||||
#[cfg(all(feature = "render-pdfium", not(target_arch = "wasm32")))]
|
||||
pub use pdfium::{PdfiumRenderer, RenderError};
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,469 @@
|
||||
//! PP-OCRv6 Small implementation backed by OAR and ONNX Runtime.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::Instant;
|
||||
|
||||
use image::RgbImage;
|
||||
use oar_ocr::oarocr::{OAROCRBuilder, OAROCR};
|
||||
use oar_ocr::processors::BoundingBox;
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{
|
||||
ImagePoint, ImageQuad, ModelArtifactKind, ModelIdentity, ModelPaths, OcrEngine, OcrMode,
|
||||
OcrOptions, OcrPage, OcrSpan, RenderPixelFormat, RenderedPage,
|
||||
};
|
||||
|
||||
/// Environment variable selecting the ONNX Runtime shared library.
|
||||
pub const ONNX_RUNTIME_LIBRARY_ENV: &str = "ORT_DYLIB_PATH";
|
||||
|
||||
/// Failures while constructing or running the OAR OCR backend.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OarOcrError {
|
||||
/// A required file is missing from the resolved model set.
|
||||
#[error("resolved OCR model set is missing {kind:?}")]
|
||||
MissingModelArtifact {
|
||||
/// Missing artifact role.
|
||||
kind: ModelArtifactKind,
|
||||
},
|
||||
/// OCR was invoked while the caller explicitly disabled it.
|
||||
#[error("OCR is disabled; select Auto or Force before invoking the engine")]
|
||||
OcrDisabled,
|
||||
/// Confidence thresholds must match the normalized engine output range.
|
||||
#[error("minimum OCR confidence must be finite and between 0 and 1, got {value}")]
|
||||
InvalidMinimumConfidence {
|
||||
/// Invalid threshold.
|
||||
value: f32,
|
||||
},
|
||||
/// Bitmap dimension arithmetic exceeded the host address space.
|
||||
#[error("rendered page {page} bitmap dimensions overflow the host address space")]
|
||||
ImageSizeOverflow {
|
||||
/// 1-indexed page number.
|
||||
page: u32,
|
||||
},
|
||||
/// A validated renderer buffer could not be represented as an RGB image.
|
||||
#[error("rendered page {page} could not be converted to an RGB image")]
|
||||
InvalidImageBuffer {
|
||||
/// 1-indexed page number.
|
||||
page: u32,
|
||||
},
|
||||
/// The external ONNX Runtime shared library could not be loaded.
|
||||
#[error("failed to load ONNX Runtime from {path}: {source}")]
|
||||
OnnxRuntimeLoad {
|
||||
/// Requested shared-library path or platform library name.
|
||||
path: PathBuf,
|
||||
/// Dynamic-loader failure.
|
||||
#[source]
|
||||
source: ort::LoadDynamicError,
|
||||
},
|
||||
/// OAR returned no result for a submitted page.
|
||||
#[error("OAR returned no result for rendered page {page}")]
|
||||
MissingPageResult {
|
||||
/// 1-indexed page number.
|
||||
page: u32,
|
||||
},
|
||||
/// OAR or ONNX Runtime rejected the models or failed during inference.
|
||||
#[error(transparent)]
|
||||
Backend(#[from] oar_ocr::core::OCRError),
|
||||
}
|
||||
|
||||
/// CPU PP-OCRv6 Small engine using OAR's detection and recognition pipeline.
|
||||
///
|
||||
/// Construction accepts only [`ModelPaths`] that have already passed
|
||||
/// pdf-inspector's manifest size and SHA-256 verification. OAR's independent
|
||||
/// model auto-download feature is deliberately not enabled.
|
||||
#[derive(Debug)]
|
||||
pub struct OarOcrEngine {
|
||||
pipeline: OAROCR,
|
||||
model: ModelIdentity,
|
||||
}
|
||||
|
||||
impl OarOcrEngine {
|
||||
/// Loads PP-OCRv6 Small from a resolved, verified model set.
|
||||
pub fn from_models(models: &ModelPaths) -> Result<Self, OarOcrError> {
|
||||
load_onnx_runtime()?;
|
||||
let detection = required_model(models, ModelArtifactKind::TextDetection)?;
|
||||
let recognition = required_model(models, ModelArtifactKind::TextRecognition)?;
|
||||
let dictionary = required_model(models, ModelArtifactKind::CharacterDictionary)?;
|
||||
|
||||
let pipeline = OAROCRBuilder::new(detection, recognition, dictionary).build()?;
|
||||
let model = ModelIdentity::new(models.manifest_id(), models.revision());
|
||||
Ok(Self { pipeline, model })
|
||||
}
|
||||
|
||||
fn recognize_page(
|
||||
&self,
|
||||
page: &RenderedPage,
|
||||
options: &OcrOptions,
|
||||
) -> Result<OcrPage, OarOcrError> {
|
||||
let started = Instant::now();
|
||||
let image = rendered_page_to_rgb(page)?;
|
||||
let result = self
|
||||
.pipeline
|
||||
.predict(vec![image])?
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or(OarOcrError::MissingPageResult { page: page.page() })?;
|
||||
|
||||
let mut spans = Vec::with_capacity(result.text_regions.len());
|
||||
let mut invalid_geometry = 0usize;
|
||||
let mut missing_recognition = 0usize;
|
||||
for region in result.text_regions {
|
||||
let (Some(text), Some(confidence)) = (region.text, region.confidence) else {
|
||||
missing_recognition += 1;
|
||||
continue;
|
||||
};
|
||||
if text.trim().is_empty() || !confidence.is_finite() {
|
||||
missing_recognition += 1;
|
||||
continue;
|
||||
}
|
||||
let confidence = confidence.clamp(0.0, 1.0);
|
||||
if confidence < options.minimum_confidence {
|
||||
continue;
|
||||
}
|
||||
|
||||
let polygon = region.dt_poly.as_ref().unwrap_or(®ion.bounding_box);
|
||||
let Some(polygon) = bounding_box_to_quad(polygon, page.width(), page.height()) else {
|
||||
invalid_geometry += 1;
|
||||
continue;
|
||||
};
|
||||
spans.push(OcrSpan {
|
||||
text: text.to_string(),
|
||||
polygon,
|
||||
confidence,
|
||||
orientation_degrees: region.orientation_angle,
|
||||
});
|
||||
}
|
||||
|
||||
let mut warnings = Vec::new();
|
||||
if !options.languages.is_empty() {
|
||||
warnings
|
||||
.push("language hints are not used by the PP-OCRv6 Small OAR backend".to_string());
|
||||
}
|
||||
if missing_recognition > 0 {
|
||||
warnings.push(format!(
|
||||
"discarded {missing_recognition} regions without usable recognition output"
|
||||
));
|
||||
}
|
||||
if invalid_geometry > 0 {
|
||||
warnings.push(format!(
|
||||
"discarded {invalid_geometry} recognized regions with invalid geometry"
|
||||
));
|
||||
}
|
||||
|
||||
let mean_confidence = if spans.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(spans.iter().map(|span| span.confidence).sum::<f32>() / spans.len() as f32)
|
||||
};
|
||||
let processing_time_ms = u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX);
|
||||
|
||||
Ok(OcrPage {
|
||||
page: page.page(),
|
||||
spans,
|
||||
mean_confidence,
|
||||
model: self.model.clone(),
|
||||
processing_time_ms,
|
||||
warnings,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn load_onnx_runtime() -> Result<(), OarOcrError> {
|
||||
let path = std::env::var_os(ONNX_RUNTIME_LIBRARY_ENV)
|
||||
.filter(|path| !path.is_empty())
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(default_onnx_runtime_library);
|
||||
drop(
|
||||
ort::init_from(&path).map_err(|source| OarOcrError::OnnxRuntimeLoad {
|
||||
path: path.clone(),
|
||||
source,
|
||||
})?,
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn default_onnx_runtime_library() -> PathBuf {
|
||||
#[cfg(target_os = "windows")]
|
||||
const NAME: &str = "onnxruntime.dll";
|
||||
#[cfg(any(target_os = "linux", target_os = "android", target_os = "freebsd"))]
|
||||
const NAME: &str = "libonnxruntime.so";
|
||||
#[cfg(any(target_os = "macos", target_os = "ios"))]
|
||||
const NAME: &str = "libonnxruntime.dylib";
|
||||
PathBuf::from(NAME)
|
||||
}
|
||||
|
||||
impl OcrEngine for OarOcrEngine {
|
||||
type Error = OarOcrError;
|
||||
|
||||
fn model(&self) -> &ModelIdentity {
|
||||
&self.model
|
||||
}
|
||||
|
||||
fn recognize(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
options: &OcrOptions,
|
||||
) -> Result<Vec<OcrPage>, Self::Error> {
|
||||
validate_options(options)?;
|
||||
|
||||
pages
|
||||
.iter()
|
||||
.map(|page| self.recognize_page(page, options))
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_options(options: &OcrOptions) -> Result<(), OarOcrError> {
|
||||
if options.mode == OcrMode::Off {
|
||||
return Err(OarOcrError::OcrDisabled);
|
||||
}
|
||||
if !options.minimum_confidence.is_finite() || !(0.0..=1.0).contains(&options.minimum_confidence)
|
||||
{
|
||||
return Err(OarOcrError::InvalidMinimumConfidence {
|
||||
value: options.minimum_confidence,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn required_model(
|
||||
models: &ModelPaths,
|
||||
kind: ModelArtifactKind,
|
||||
) -> Result<&std::path::Path, OarOcrError> {
|
||||
models
|
||||
.get(kind)
|
||||
.ok_or(OarOcrError::MissingModelArtifact { kind })
|
||||
}
|
||||
|
||||
fn rendered_page_to_rgb(page: &RenderedPage) -> Result<RgbImage, OarOcrError> {
|
||||
let width = usize::try_from(page.width())
|
||||
.map_err(|_| OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let height = usize::try_from(page.height())
|
||||
.map_err(|_| OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let output_len = width
|
||||
.checked_mul(height)
|
||||
.and_then(|pixels| pixels.checked_mul(3))
|
||||
.ok_or(OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let input_bpp = page.format().bytes_per_pixel();
|
||||
let active_input_row = width
|
||||
.checked_mul(input_bpp)
|
||||
.ok_or(OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
let output_row = width
|
||||
.checked_mul(3)
|
||||
.ok_or(OarOcrError::ImageSizeOverflow { page: page.page() })?;
|
||||
|
||||
let mut rgb = vec![0u8; output_len];
|
||||
for row in 0..height {
|
||||
let input_start = row * page.stride();
|
||||
let input = &page.pixels()[input_start..input_start + active_input_row];
|
||||
let output_start = row * output_row;
|
||||
let output = &mut rgb[output_start..output_start + output_row];
|
||||
match page.format() {
|
||||
RenderPixelFormat::Rgb8 => output.copy_from_slice(input),
|
||||
RenderPixelFormat::Rgba8 => {
|
||||
for (rgba, rgb) in input.chunks_exact(4).zip(output.chunks_exact_mut(3)) {
|
||||
rgb.copy_from_slice(&rgba[..3]);
|
||||
}
|
||||
}
|
||||
RenderPixelFormat::Gray8 => {
|
||||
for (&gray, rgb) in input.iter().zip(output.chunks_exact_mut(3)) {
|
||||
rgb.fill(gray);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
RgbImage::from_raw(page.width(), page.height(), rgb)
|
||||
.ok_or(OarOcrError::InvalidImageBuffer { page: page.page() })
|
||||
}
|
||||
|
||||
fn bounding_box_to_quad(bounding_box: &BoundingBox, width: u32, height: u32) -> Option<ImageQuad> {
|
||||
let points: Vec<ImagePoint> = bounding_box
|
||||
.points
|
||||
.iter()
|
||||
.filter(|point| point.x.is_finite() && point.y.is_finite())
|
||||
.map(|point| {
|
||||
ImagePoint::new(
|
||||
point.x.clamp(0.0, width as f32),
|
||||
point.y.clamp(0.0, height as f32),
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
if bounding_box.points.len() == 4 && points.len() == 4 && is_ordered_convex_quad(&points) {
|
||||
return Some(ImageQuad::new([points[0], points[1], points[2], points[3]]));
|
||||
}
|
||||
if points.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let min_x = points
|
||||
.iter()
|
||||
.map(|point| point.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_x = points
|
||||
.iter()
|
||||
.map(|point| point.x)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let min_y = points
|
||||
.iter()
|
||||
.map(|point| point.y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_y = points
|
||||
.iter()
|
||||
.map(|point| point.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if max_x <= min_x || max_y <= min_y {
|
||||
return None;
|
||||
}
|
||||
Some(ImageQuad::new([
|
||||
ImagePoint::new(min_x, min_y),
|
||||
ImagePoint::new(max_x, min_y),
|
||||
ImagePoint::new(max_x, max_y),
|
||||
ImagePoint::new(min_x, max_y),
|
||||
]))
|
||||
}
|
||||
|
||||
fn is_ordered_convex_quad(points: &[ImagePoint]) -> bool {
|
||||
if points.len() != 4 {
|
||||
return false;
|
||||
}
|
||||
let mut orientation = 0.0_f32;
|
||||
for index in 0..4 {
|
||||
let first = points[index];
|
||||
let second = points[(index + 1) % 4];
|
||||
let third = points[(index + 2) % 4];
|
||||
let cross = (second.x - first.x) * (third.y - second.y)
|
||||
- (second.y - first.y) * (third.x - second.x);
|
||||
if cross.abs() <= f32::EPSILON {
|
||||
return false;
|
||||
}
|
||||
if orientation == 0.0 {
|
||||
orientation = cross.signum();
|
||||
} else if cross.signum() != orientation {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use oar_ocr::processors::Point;
|
||||
|
||||
use super::*;
|
||||
use crate::vision::PageTransform;
|
||||
|
||||
fn page(format: RenderPixelFormat, stride: usize, pixels: Vec<u8>) -> RenderedPage {
|
||||
let transform =
|
||||
PageTransform::from_corners(2, 2, (0.0, 2.0), (2.0, 2.0), (0.0, 0.0)).unwrap();
|
||||
RenderedPage::new(1, 2.0, 2.0, 2, 2, stride, format, pixels, transform).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_padded_rgb_without_exposing_padding() {
|
||||
let page = page(
|
||||
RenderPixelFormat::Rgb8,
|
||||
8,
|
||||
vec![1, 2, 3, 4, 5, 6, 99, 99, 7, 8, 9, 10, 11, 12, 99, 99],
|
||||
);
|
||||
let image = rendered_page_to_rgb(&page).unwrap();
|
||||
assert_eq!(image.as_raw(), &[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_rgba_and_gray_to_rgb() {
|
||||
let rgba = page(
|
||||
RenderPixelFormat::Rgba8,
|
||||
8,
|
||||
vec![1, 2, 3, 44, 4, 5, 6, 55, 7, 8, 9, 66, 10, 11, 12, 77],
|
||||
);
|
||||
assert_eq!(
|
||||
rendered_page_to_rgb(&rgba).unwrap().as_raw(),
|
||||
&[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]
|
||||
);
|
||||
|
||||
let gray = page(RenderPixelFormat::Gray8, 2, vec![1, 2, 3, 4]);
|
||||
assert_eq!(
|
||||
rendered_page_to_rgb(&gray).unwrap().as_raw(),
|
||||
&[1, 1, 1, 2, 2, 2, 3, 3, 3, 4, 4, 4]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserves_quads_and_clamps_them_to_the_bitmap() {
|
||||
let bbox = BoundingBox::new(vec![
|
||||
Point::new(-1.0, 2.0),
|
||||
Point::new(11.0, 2.0),
|
||||
Point::new(11.0, 9.0),
|
||||
Point::new(-1.0, 9.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&bbox, 10, 8).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(0.0, 2.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(10.0, 8.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reduces_polygons_to_a_stable_axis_aligned_quad() {
|
||||
let bbox = BoundingBox::new(vec![
|
||||
Point::new(2.0, 1.0),
|
||||
Point::new(7.0, 2.0),
|
||||
Point::new(8.0, 6.0),
|
||||
Point::new(5.0, 9.0),
|
||||
Point::new(1.0, 5.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&bbox, 10, 10).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(1.0, 1.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(8.0, 9.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalizes_unordered_or_partially_invalid_quads() {
|
||||
let unordered = BoundingBox::new(vec![
|
||||
Point::new(1.0, 1.0),
|
||||
Point::new(8.0, 8.0),
|
||||
Point::new(8.0, 1.0),
|
||||
Point::new(1.0, 8.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&unordered, 10, 10).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(1.0, 1.0));
|
||||
assert_eq!(quad.points[1], ImagePoint::new(8.0, 1.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(8.0, 8.0));
|
||||
|
||||
let partially_invalid = BoundingBox::new(vec![
|
||||
Point::new(8.0, 8.0),
|
||||
Point::new(f32::NAN, 4.0),
|
||||
Point::new(1.0, 8.0),
|
||||
Point::new(8.0, 1.0),
|
||||
Point::new(1.0, 1.0),
|
||||
]);
|
||||
let quad = bounding_box_to_quad(&partially_invalid, 10, 10).unwrap();
|
||||
assert_eq!(quad.points[0], ImagePoint::new(1.0, 1.0));
|
||||
assert_eq!(quad.points[1], ImagePoint::new(8.0, 1.0));
|
||||
assert_eq!(quad.points[2], ImagePoint::new(8.0, 8.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn refuses_disabled_or_invalid_options_before_inference() {
|
||||
assert!(matches!(
|
||||
validate_options(&OcrOptions::new()),
|
||||
Err(OarOcrError::OcrDisabled)
|
||||
));
|
||||
for value in [-0.1, 1.1, f32::NAN, f32::INFINITY] {
|
||||
let options = OcrOptions::new()
|
||||
.mode(OcrMode::Force)
|
||||
.minimum_confidence(value);
|
||||
assert!(matches!(
|
||||
validate_options(&options),
|
||||
Err(OarOcrError::InvalidMinimumConfidence { .. })
|
||||
));
|
||||
}
|
||||
assert!(validate_options(
|
||||
&OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.minimum_confidence(1.0)
|
||||
)
|
||||
.is_ok());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,266 @@
|
||||
//! PDFium-backed implementation of the renderer-neutral page contract.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use firecrawl_pdfium::{Pdfium, PixelFormat, PixelPoint, RenderConfig};
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{
|
||||
PageRenderer, PageTransform, RenderBufferError, RenderOptions, RenderPixelFormat, RenderedPage,
|
||||
};
|
||||
|
||||
impl RenderPixelFormat {
|
||||
fn pdfium_format(self) -> PixelFormat {
|
||||
match self {
|
||||
// PDFium produces BGR directly; `rendered_page_from_pdfium`
|
||||
// swaps the red and blue channels in place.
|
||||
Self::Rgb8 => PixelFormat::Bgr8,
|
||||
Self::Rgba8 => PixelFormat::Rgba8,
|
||||
Self::Gray8 => PixelFormat::Gray8,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl RenderOptions {
|
||||
fn pdfium_config(&self) -> RenderConfig {
|
||||
RenderConfig::new()
|
||||
.dpi(self.dpi)
|
||||
.pixel_format(self.pixel_format.pdfium_format())
|
||||
.annotations(self.annotations)
|
||||
.form_fields(self.form_fields)
|
||||
.max_output_bytes(self.max_output_bytes_per_page)
|
||||
}
|
||||
}
|
||||
|
||||
/// Errors produced by the optional local renderer.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum RenderError {
|
||||
/// Page numbers in pdf-inspector APIs are 1-indexed, so zero is invalid.
|
||||
#[error("page numbers are 1-indexed; page 0 is invalid")]
|
||||
InvalidPageNumber,
|
||||
/// The requested 1-indexed page is not present in the document.
|
||||
#[error("page {page} is out of bounds for a {page_count}-page document")]
|
||||
PageOutOfBounds {
|
||||
/// Requested 1-indexed page.
|
||||
page: u32,
|
||||
/// Number of pages in the document.
|
||||
page_count: usize,
|
||||
},
|
||||
/// PDFium loading, document parsing, form setup, or rendering failed.
|
||||
#[error(transparent)]
|
||||
Pdfium(#[from] firecrawl_pdfium::Error),
|
||||
/// PDFium returned an internally inconsistent bitmap or transform.
|
||||
#[error(transparent)]
|
||||
Buffer(#[from] RenderBufferError),
|
||||
}
|
||||
|
||||
/// Loaded PDFium renderer used to prepare pages for OCR.
|
||||
///
|
||||
/// PDFium calls are safe from concurrent threads but serialize inside the
|
||||
/// underlying binding. Returned [`RenderedPage`] values are ordinary owned
|
||||
/// data and can be processed concurrently after rendering.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct PdfiumRenderer {
|
||||
pdfium: Pdfium,
|
||||
}
|
||||
|
||||
impl PdfiumRenderer {
|
||||
/// Loads PDFium using `firecrawl-pdfium`'s documented discovery chain.
|
||||
pub fn load() -> Result<Self, RenderError> {
|
||||
Ok(Self {
|
||||
pdfium: Pdfium::load()?,
|
||||
})
|
||||
}
|
||||
|
||||
/// Loads PDFium from an explicit native library path.
|
||||
pub fn load_from_path(path: impl AsRef<Path>) -> Result<Self, RenderError> {
|
||||
Ok(Self {
|
||||
pdfium: Pdfium::load_from_path(path)?,
|
||||
})
|
||||
}
|
||||
|
||||
/// Path of the active PDFium library, if it was loaded from a concrete
|
||||
/// file rather than through the system loader.
|
||||
pub fn loaded_from(&self) -> Option<&Path> {
|
||||
self.pdfium.loaded_from()
|
||||
}
|
||||
|
||||
/// Renders selected 1-indexed pages in the same order as `pages`.
|
||||
///
|
||||
/// This inherent method mirrors [`PageRenderer`] so existing callers do
|
||||
/// not need to import the trait.
|
||||
pub fn render_pages(
|
||||
&self,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
options: &RenderOptions,
|
||||
) -> Result<Vec<RenderedPage>, RenderError> {
|
||||
self.render_pages_impl(pdf_bytes, pages, password, options)
|
||||
}
|
||||
|
||||
fn render_pages_impl(
|
||||
&self,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
options: &RenderOptions,
|
||||
) -> Result<Vec<RenderedPage>, RenderError> {
|
||||
if pages.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
|
||||
if pages.contains(&0) {
|
||||
return Err(RenderError::InvalidPageNumber);
|
||||
}
|
||||
|
||||
let document = self.pdfium.load_document(pdf_bytes.to_vec(), password)?;
|
||||
let page_count = document.page_count();
|
||||
|
||||
if let Some(&page) = pages.iter().find(|&&page| page as usize > page_count) {
|
||||
return Err(RenderError::PageOutOfBounds { page, page_count });
|
||||
}
|
||||
|
||||
if options.form_fields {
|
||||
document.enable_form_rendering()?;
|
||||
}
|
||||
|
||||
let config = options.pdfium_config();
|
||||
let mut rendered_pages = Vec::with_capacity(pages.len());
|
||||
for &page_number in pages {
|
||||
let page = document.page(page_number as usize - 1)?;
|
||||
let size = page.size();
|
||||
let rendered = page.render(&config)?;
|
||||
rendered_pages.push(rendered_page_from_pdfium(
|
||||
page_number,
|
||||
size.width,
|
||||
size.height,
|
||||
options.pixel_format,
|
||||
rendered,
|
||||
)?);
|
||||
}
|
||||
|
||||
Ok(rendered_pages)
|
||||
}
|
||||
}
|
||||
|
||||
impl PageRenderer for PdfiumRenderer {
|
||||
type Error = RenderError;
|
||||
|
||||
fn render_pages(
|
||||
&self,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
options: &RenderOptions,
|
||||
) -> Result<Vec<RenderedPage>, Self::Error> {
|
||||
self.render_pages_impl(pdf_bytes, pages, password, options)
|
||||
}
|
||||
}
|
||||
|
||||
fn rendered_page_from_pdfium(
|
||||
page: u32,
|
||||
page_width: f32,
|
||||
page_height: f32,
|
||||
format: RenderPixelFormat,
|
||||
rendered: firecrawl_pdfium::RenderedPage,
|
||||
) -> Result<RenderedPage, RenderBufferError> {
|
||||
let width = rendered.width();
|
||||
let height = rendered.height();
|
||||
let stride = rendered.stride();
|
||||
let pdfium_transform = *rendered.transform();
|
||||
let corner = |x, y| {
|
||||
let point = pdfium_transform.pixel_to_page(PixelPoint::new(x, y));
|
||||
(point.x, point.y)
|
||||
};
|
||||
let transform = PageTransform::from_corners(
|
||||
width,
|
||||
height,
|
||||
corner(0.0, 0.0),
|
||||
corner(f64::from(width), 0.0),
|
||||
corner(0.0, f64::from(height)),
|
||||
)
|
||||
.ok_or(RenderBufferError::InvalidTransform)?;
|
||||
let mut pixels = rendered.into_pixels();
|
||||
|
||||
if format == RenderPixelFormat::Rgb8 {
|
||||
bgr_to_rgb_in_place(&mut pixels, width, height, stride)?;
|
||||
}
|
||||
|
||||
RenderedPage::new(
|
||||
page,
|
||||
page_width,
|
||||
page_height,
|
||||
width,
|
||||
height,
|
||||
stride,
|
||||
format,
|
||||
pixels,
|
||||
transform,
|
||||
)
|
||||
}
|
||||
|
||||
fn bgr_to_rgb_in_place(
|
||||
pixels: &mut [u8],
|
||||
width: u32,
|
||||
height: u32,
|
||||
stride: usize,
|
||||
) -> Result<(), RenderBufferError> {
|
||||
let row_bytes = (width as usize)
|
||||
.checked_mul(RenderPixelFormat::Rgb8.bytes_per_pixel())
|
||||
.ok_or(RenderBufferError::SizeOverflow)?;
|
||||
if stride < row_bytes {
|
||||
return Err(RenderBufferError::InvalidStride {
|
||||
stride,
|
||||
minimum: row_bytes,
|
||||
});
|
||||
}
|
||||
let expected = stride
|
||||
.checked_mul(height as usize)
|
||||
.ok_or(RenderBufferError::SizeOverflow)?;
|
||||
if pixels.len() != expected {
|
||||
return Err(RenderBufferError::InvalidBufferLength {
|
||||
actual: pixels.len(),
|
||||
expected,
|
||||
});
|
||||
}
|
||||
|
||||
for row in pixels.chunks_exact_mut(stride) {
|
||||
for pixel in row[..row_bytes].chunks_exact_mut(3) {
|
||||
pixel.swap(0, 2);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn bgr_pixels_are_converted_to_rgb_in_place() {
|
||||
let mut pixels = vec![1, 2, 3, 4, 5, 6];
|
||||
bgr_to_rgb_in_place(&mut pixels, 2, 1, 6).unwrap();
|
||||
assert_eq!(pixels, [3, 2, 1, 6, 5, 4]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bgr_conversion_skips_row_padding() {
|
||||
let mut pixels = vec![1, 2, 3, 9, 7, 8, 9, 6];
|
||||
bgr_to_rgb_in_place(&mut pixels, 1, 2, 4).unwrap();
|
||||
assert_eq!(pixels, [3, 2, 1, 9, 9, 8, 7, 6]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn malformed_bgr_buffers_return_errors() {
|
||||
assert!(matches!(
|
||||
bgr_to_rgb_in_place(&mut [0; 6], 2, 1, 5),
|
||||
Err(RenderBufferError::InvalidStride { .. })
|
||||
));
|
||||
assert!(matches!(
|
||||
bgr_to_rgb_in_place(&mut [0; 5], 1, 2, 3),
|
||||
Err(RenderBufferError::InvalidBufferLength { .. })
|
||||
));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,426 @@
|
||||
//! One-call native extraction and OCR pipeline.
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
use std::path::Path;
|
||||
use std::time::Instant;
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
use crate::{MarkdownOptions, PageOcrReasons, PdfError};
|
||||
|
||||
use super::{
|
||||
fuse_ocr_pages, route_ocr_pages, run_ocr_pages, FusedPageMarkdown, HttpModelDownloadError,
|
||||
HttpModelDownloader, ModelAcquireError, ModelStore, ModelStoreError, OarOcrEngine, OarOcrError,
|
||||
OcrFusionError, OcrFusionOptions, OcrMode, OcrOptions, OcrRoutingError, OcrRun, OcrRunError,
|
||||
PdfiumRenderer, RenderError, RenderOptions, PP_OCR_V6_SMALL,
|
||||
};
|
||||
|
||||
/// Options for native extraction with optional OCR.
|
||||
#[derive(Clone)]
|
||||
pub struct OcrPdfOptions {
|
||||
/// Page rasterization settings used when OCR is routed.
|
||||
pub render: RenderOptions,
|
||||
/// OCR routing, model, and recognition settings.
|
||||
pub ocr: OcrOptions,
|
||||
/// Markdown formatting shared by native and OCR assembly.
|
||||
pub markdown: MarkdownOptions,
|
||||
/// Optional 1-indexed page selection. `None` processes the full document.
|
||||
pub page_filter: Option<BTreeSet<u32>>,
|
||||
/// Password for an encrypted PDF.
|
||||
pub password: Option<String>,
|
||||
/// Weak OCR threshold for recommending the hosted pipeline.
|
||||
pub hosted_recommendation_confidence: f32,
|
||||
}
|
||||
|
||||
impl Default for OcrPdfOptions {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
render: RenderOptions::default(),
|
||||
ocr: OcrOptions::default(),
|
||||
markdown: MarkdownOptions::default(),
|
||||
page_filter: None,
|
||||
password: None,
|
||||
hosted_recommendation_confidence: 0.5,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for OcrPdfOptions {
|
||||
fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
formatter
|
||||
.debug_struct("OcrPdfOptions")
|
||||
.field("render", &self.render)
|
||||
.field("ocr", &self.ocr)
|
||||
.field("markdown", &self.markdown)
|
||||
.field("page_filter", &self.page_filter)
|
||||
.field("password", &self.password.as_ref().map(|_| "[REDACTED]"))
|
||||
.field(
|
||||
"hosted_recommendation_confidence",
|
||||
&self.hosted_recommendation_confidence,
|
||||
)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl OcrPdfOptions {
|
||||
/// Creates options with OCR disabled, preserving the native-only path.
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Replaces page rasterization settings.
|
||||
pub fn render(mut self, render: RenderOptions) -> Self {
|
||||
self.render = render;
|
||||
self
|
||||
}
|
||||
|
||||
/// Replaces OCR routing and recognition settings.
|
||||
pub fn ocr(mut self, ocr: OcrOptions) -> Self {
|
||||
self.ocr = ocr;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets OCR routing without changing the remaining OCR settings.
|
||||
pub fn mode(mut self, mode: OcrMode) -> Self {
|
||||
self.ocr.mode = mode;
|
||||
self
|
||||
}
|
||||
|
||||
/// Replaces Markdown formatting options.
|
||||
pub fn markdown(mut self, markdown: MarkdownOptions) -> Self {
|
||||
self.markdown = markdown;
|
||||
self
|
||||
}
|
||||
|
||||
/// Restricts processing to 1-indexed pages in ascending order.
|
||||
pub fn pages(mut self, pages: impl IntoIterator<Item = u32>) -> Self {
|
||||
self.page_filter = Some(pages.into_iter().collect());
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the password used to decrypt the PDF.
|
||||
pub fn password(mut self, password: impl Into<String>) -> Self {
|
||||
self.password = Some(password.into());
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the weak-OCR threshold for recommending hosted document parsing.
|
||||
pub fn hosted_recommendation_confidence(mut self, confidence: f32) -> Self {
|
||||
self.hosted_recommendation_confidence = confidence;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// Complete native/OCR Markdown output for a PDF request.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct OcrPdfResult {
|
||||
/// Final document Markdown in selected-page order.
|
||||
pub markdown: String,
|
||||
/// Final per-page Markdown and provenance, using 1-indexed page numbers.
|
||||
pub pages: Vec<FusedPageMarkdown>,
|
||||
/// Total pages in the PDF, independent of page selection.
|
||||
pub page_count: u32,
|
||||
/// 1-indexed selected pages recommended for OCR by native extraction.
|
||||
pub pages_recommended_for_ocr: Vec<u32>,
|
||||
/// 1-indexed pages actually rendered and recognized.
|
||||
pub pages_routed_to_ocr: Vec<u32>,
|
||||
/// 1-indexed pages whose OCR result recommends hosted document parsing.
|
||||
pub pages_recommending_hosted: Vec<u32>,
|
||||
/// Original machine-readable OCR reasons for selected pages.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
/// Selected pages where deterministic table detection found tables.
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// Selected pages where deterministic layout found multiple columns.
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// Whether deterministic extraction found tables or columns.
|
||||
pub is_complex: bool,
|
||||
/// End-to-end processing time.
|
||||
pub processing_time_ms: u64,
|
||||
/// Batch page-rendering time; zero when no OCR work was routed.
|
||||
pub render_time_ms: u64,
|
||||
/// Batch OCR time; zero when no OCR work was routed.
|
||||
pub ocr_time_ms: u64,
|
||||
}
|
||||
|
||||
/// Processes a PDF file through native extraction and selective OCR.
|
||||
pub fn process_pdf_with_ocr(
|
||||
path: impl AsRef<Path>,
|
||||
options: OcrPdfOptions,
|
||||
) -> Result<OcrPdfResult, OcrPipelineError> {
|
||||
let bytes = std::fs::read(path).map_err(PdfError::from)?;
|
||||
process_pdf_with_ocr_mem(&bytes, options)
|
||||
}
|
||||
|
||||
/// Processes PDF bytes through native extraction and selective OCR.
|
||||
///
|
||||
/// Native extraction always runs first. `Auto` initializes PDFium, downloads
|
||||
/// models, and starts OAR only if the detector selected at least one page.
|
||||
/// `Off` therefore has no renderer, model-cache, network, or inference side
|
||||
/// effects even though the complete feature is compiled into the application.
|
||||
pub fn process_pdf_with_ocr_mem(
|
||||
buffer: &[u8],
|
||||
options: OcrPdfOptions,
|
||||
) -> Result<OcrPdfResult, OcrPipelineError> {
|
||||
OcrFusionOptions::new()
|
||||
.render_dpi(options.render.dpi)
|
||||
.hosted_recommendation_confidence(options.hosted_recommendation_confidence)
|
||||
.validate()?;
|
||||
let minimum_confidence = options.ocr.minimum_confidence;
|
||||
if !minimum_confidence.is_finite() || !(0.0..=1.0).contains(&minimum_confidence) {
|
||||
return Err(OcrPipelineError::InvalidMinimumConfidence {
|
||||
value: minimum_confidence,
|
||||
});
|
||||
}
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(OcrPipelineError::InvalidSelectedPage { page: 0 });
|
||||
}
|
||||
|
||||
let started = Instant::now();
|
||||
let selected_pages: Option<Vec<u32>> = options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.map(|pages| pages.iter().copied().collect());
|
||||
let selected_pages_zero_indexed: Option<Vec<u32>> = selected_pages
|
||||
.as_ref()
|
||||
.map(|pages| pages.iter().map(|page| page - 1).collect());
|
||||
|
||||
let mut page_markdown_options = options.markdown.clone();
|
||||
page_markdown_options.include_page_numbers = false;
|
||||
let (native, page_count) = crate::extract_pages_markdown_mem_for_ocr(
|
||||
buffer,
|
||||
selected_pages_zero_indexed.as_deref(),
|
||||
options.password.as_deref(),
|
||||
&page_markdown_options,
|
||||
)?;
|
||||
if let Some(invalid) = selected_pages
|
||||
.as_ref()
|
||||
.and_then(|pages| pages.iter().copied().find(|page| *page > page_count))
|
||||
{
|
||||
return Err(OcrPipelineError::InvalidSelectedPage { page: invalid });
|
||||
}
|
||||
|
||||
let routed = route_ocr_pages(
|
||||
options.ocr.mode,
|
||||
page_count,
|
||||
&native.pages_needing_ocr,
|
||||
selected_pages.as_deref(),
|
||||
)?;
|
||||
|
||||
let ocr_run = if routed.is_empty() {
|
||||
OcrRun {
|
||||
pages: Vec::new(),
|
||||
render_time_ms: 0,
|
||||
ocr_time_ms: 0,
|
||||
}
|
||||
} else {
|
||||
// Resolve the native renderer before any network request so a missing
|
||||
// PDFium installation cannot trigger a model download it cannot use.
|
||||
let renderer = PdfiumRenderer::load()?;
|
||||
let store = ModelStore::from_options(&options.ocr)?;
|
||||
let models = store.resolve_or_download(
|
||||
&PP_OCR_V6_SMALL,
|
||||
options.ocr.model_downloads,
|
||||
&HttpModelDownloader::default(),
|
||||
)?;
|
||||
let engine = OarOcrEngine::from_models(&models)?;
|
||||
run_ocr_pages(
|
||||
&renderer,
|
||||
&engine,
|
||||
buffer,
|
||||
&routed,
|
||||
options.password.as_deref(),
|
||||
&options.render,
|
||||
&options.ocr,
|
||||
)?
|
||||
};
|
||||
|
||||
let fusion_options = OcrFusionOptions::new()
|
||||
.markdown(page_markdown_options)
|
||||
.render_dpi(options.render.dpi)
|
||||
.hosted_recommendation_confidence(options.hosted_recommendation_confidence);
|
||||
let fused = fuse_ocr_pages(&native.pages, &ocr_run, page_count, &fusion_options)?;
|
||||
let pages_recommending_hosted = fused
|
||||
.pages
|
||||
.iter()
|
||||
.filter(|page| page.provenance.hosted_recommended)
|
||||
.map(|page| page.provenance.page)
|
||||
.collect();
|
||||
let markdown = assemble_document_markdown(&fused.pages, options.markdown.include_page_numbers);
|
||||
|
||||
Ok(OcrPdfResult {
|
||||
markdown,
|
||||
pages: fused.pages,
|
||||
page_count,
|
||||
pages_recommended_for_ocr: native.pages_needing_ocr,
|
||||
pages_routed_to_ocr: routed,
|
||||
pages_recommending_hosted,
|
||||
ocr_reasons_by_page: native.ocr_reasons_by_page,
|
||||
pages_with_tables: native.pages_with_tables,
|
||||
pages_with_columns: native.pages_with_columns,
|
||||
is_complex: native.is_complex,
|
||||
processing_time_ms: elapsed_ms(started),
|
||||
render_time_ms: fused.render_time_ms,
|
||||
ocr_time_ms: fused.ocr_time_ms,
|
||||
})
|
||||
}
|
||||
|
||||
fn assemble_document_markdown(pages: &[FusedPageMarkdown], include_page_numbers: bool) -> String {
|
||||
let mut document = String::new();
|
||||
for (index, page) in pages.iter().enumerate() {
|
||||
if index > 0 {
|
||||
document.push_str("\n\n");
|
||||
}
|
||||
if include_page_numbers {
|
||||
document.push_str(&format!("<!-- Page {} -->\n\n", page.page));
|
||||
}
|
||||
document.push_str(page.markdown.trim());
|
||||
}
|
||||
if !document.is_empty() {
|
||||
document.push('\n');
|
||||
}
|
||||
document
|
||||
}
|
||||
|
||||
fn elapsed_ms(started: Instant) -> u64 {
|
||||
u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
/// Failures from the complete OCR pipeline.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrPipelineError {
|
||||
/// PDF loading or native extraction failed.
|
||||
#[error(transparent)]
|
||||
Pdf(#[from] PdfError),
|
||||
/// Page routing rejected an invalid request.
|
||||
#[error(transparent)]
|
||||
Routing(#[from] OcrRoutingError),
|
||||
/// The model cache could not be located or initialized.
|
||||
#[error(transparent)]
|
||||
ModelStore(#[from] ModelStoreError),
|
||||
/// A pinned model set could not be resolved or acquired.
|
||||
#[error(transparent)]
|
||||
ModelAcquire(#[from] ModelAcquireError<HttpModelDownloadError>),
|
||||
/// PDFium could not load or rasterize the request.
|
||||
#[error(transparent)]
|
||||
Render(#[from] RenderError),
|
||||
/// The OAR engine could not initialize.
|
||||
#[error(transparent)]
|
||||
Oar(#[from] OarOcrError),
|
||||
/// Selective rendering or OCR execution failed.
|
||||
#[error(transparent)]
|
||||
Run(#[from] OcrRunError),
|
||||
/// OCR/native Markdown fusion failed.
|
||||
#[error(transparent)]
|
||||
Fusion(#[from] OcrFusionError),
|
||||
/// Page zero is invalid because public page selections are 1-indexed.
|
||||
#[error("selected page {page} is invalid; page numbers are 1-indexed")]
|
||||
InvalidSelectedPage {
|
||||
/// Invalid page number.
|
||||
page: u32,
|
||||
},
|
||||
/// OCR span confidence is outside the inclusive 0–1 range.
|
||||
#[error("minimum OCR confidence must be between 0 and 1, got {value}")]
|
||||
InvalidMinimumConfidence {
|
||||
/// Invalid value.
|
||||
value: f32,
|
||||
},
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn off_mode_extracts_native_text_without_runtime_side_effects() {
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let result = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new()).unwrap();
|
||||
|
||||
assert_eq!(result.page_count, 3);
|
||||
assert_eq!(result.pages.len(), 3);
|
||||
assert!(result.markdown.contains("Thermodynamic Properties"));
|
||||
assert!(result.pages_recommended_for_ocr.is_empty());
|
||||
assert!(result.pages_routed_to_ocr.is_empty());
|
||||
assert!(result.pages_recommending_hosted.is_empty());
|
||||
assert_eq!(result.render_time_ms, 0);
|
||||
assert_eq!(result.ocr_time_ms, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn auto_mode_does_not_load_pdfium_or_models_for_clean_pdf() {
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let result =
|
||||
process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().mode(OcrMode::Auto)).unwrap();
|
||||
|
||||
assert!(result.pages_routed_to_ocr.is_empty());
|
||||
assert!(result.markdown.contains("Freon 12"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn selection_and_page_markers_use_public_one_indexed_pages() {
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let mut markdown = MarkdownOptions::default();
|
||||
markdown.include_page_numbers = true;
|
||||
let result =
|
||||
process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().pages([2]).markdown(markdown))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert_eq!(result.pages[0].page, 2);
|
||||
assert_eq!(result.pages[0].page, result.pages[0].provenance.page);
|
||||
assert!(result.markdown.starts_with("<!-- Page 2 -->"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn off_mode_marks_unprocessed_scan_for_hosted_fallback() {
|
||||
let bytes = std::fs::read("tests/fixtures/scan_with_native_header_text.pdf").unwrap();
|
||||
let result = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new()).unwrap();
|
||||
|
||||
assert!(result.pages_routed_to_ocr.is_empty());
|
||||
assert_eq!(result.pages_recommending_hosted, vec![1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn password_is_redacted_and_used_for_native_extraction() {
|
||||
let options = OcrPdfOptions::new().password("secret123");
|
||||
assert!(!format!("{options:?}").contains("secret123"));
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/encrypted-secret123.pdf").unwrap();
|
||||
let result = process_pdf_with_ocr_mem(&bytes, options).unwrap();
|
||||
assert!(result.markdown.contains("Procurement"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_out_of_range_selection_even_with_ocr_off() {
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let error = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().pages([4])).unwrap_err();
|
||||
assert!(matches!(
|
||||
error,
|
||||
OcrPipelineError::InvalidSelectedPage { page: 4 }
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_expensive_options_fail_before_pdf_or_runtime_access() {
|
||||
let mut invalid_dpi = OcrPdfOptions::new();
|
||||
invalid_dpi.render.dpi = f32::NAN;
|
||||
assert!(matches!(
|
||||
process_pdf_with_ocr_mem(b"not a PDF", invalid_dpi),
|
||||
Err(OcrPipelineError::Fusion(
|
||||
OcrFusionError::InvalidRenderDpi { .. }
|
||||
))
|
||||
));
|
||||
|
||||
let invalid_hosted = OcrPdfOptions::new().hosted_recommendation_confidence(1.1);
|
||||
assert!(matches!(
|
||||
process_pdf_with_ocr_mem(b"not a PDF", invalid_hosted),
|
||||
Err(OcrPipelineError::Fusion(
|
||||
OcrFusionError::InvalidHostedConfidence { .. }
|
||||
))
|
||||
));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,552 @@
|
||||
//! Renderer-neutral page bitmap and coordinate types.
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
use crate::PdfRect;
|
||||
|
||||
/// Default rendering resolution for OCR.
|
||||
pub const DEFAULT_RENDER_DPI: f32 = 150.0;
|
||||
|
||||
/// Default maximum size of one rendered page: 256 MiB.
|
||||
pub const DEFAULT_MAX_OUTPUT_BYTES: u64 = 256 * 1024 * 1024;
|
||||
|
||||
/// Pixel layout returned by [`RenderedPage`].
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum RenderPixelFormat {
|
||||
/// Three bytes per pixel in red, green, blue order. This is the default
|
||||
/// because OCR preprocessors conventionally consume RGB images.
|
||||
#[default]
|
||||
Rgb8,
|
||||
/// Four bytes per pixel in red, green, blue, alpha order.
|
||||
Rgba8,
|
||||
/// One luminance byte per pixel.
|
||||
Gray8,
|
||||
}
|
||||
|
||||
impl RenderPixelFormat {
|
||||
/// Number of bytes used by one pixel.
|
||||
pub fn bytes_per_pixel(self) -> usize {
|
||||
match self {
|
||||
Self::Rgb8 => 3,
|
||||
Self::Rgba8 => 4,
|
||||
Self::Gray8 => 1,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Configuration for pages rendered as input to a local vision pipeline.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct RenderOptions {
|
||||
/// Output resolution. Defaults to 150 DPI.
|
||||
pub dpi: f32,
|
||||
/// Pixel layout. Defaults to three-channel RGB.
|
||||
pub pixel_format: RenderPixelFormat,
|
||||
/// Include PDF annotations in the rendered bitmap.
|
||||
pub annotations: bool,
|
||||
/// Include visible static AcroForm field appearances.
|
||||
pub form_fields: bool,
|
||||
/// Maximum allocation for each rendered page.
|
||||
pub max_output_bytes_per_page: u64,
|
||||
}
|
||||
|
||||
impl Default for RenderOptions {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
dpi: DEFAULT_RENDER_DPI,
|
||||
pixel_format: RenderPixelFormat::Rgb8,
|
||||
annotations: true,
|
||||
form_fields: true,
|
||||
max_output_bytes_per_page: DEFAULT_MAX_OUTPUT_BYTES,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl RenderOptions {
|
||||
/// Creates local-rendering options with OCR-oriented defaults.
|
||||
pub fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
|
||||
/// Sets the output resolution in dots per inch.
|
||||
pub fn dpi(mut self, dpi: f32) -> Self {
|
||||
self.dpi = dpi;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the output pixel layout.
|
||||
pub fn pixel_format(mut self, pixel_format: RenderPixelFormat) -> Self {
|
||||
self.pixel_format = pixel_format;
|
||||
self
|
||||
}
|
||||
|
||||
/// Toggles annotation rendering.
|
||||
pub fn annotations(mut self, annotations: bool) -> Self {
|
||||
self.annotations = annotations;
|
||||
self
|
||||
}
|
||||
|
||||
/// Toggles visible static form-field rendering.
|
||||
pub fn form_fields(mut self, form_fields: bool) -> Self {
|
||||
self.form_fields = form_fields;
|
||||
self
|
||||
}
|
||||
|
||||
/// Sets the maximum allocation for each rendered page.
|
||||
pub fn max_output_bytes_per_page(mut self, bytes: u64) -> Self {
|
||||
self.max_output_bytes_per_page = bytes;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
/// A point in PDF page space, measured in points from the bottom-left.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct PagePoint {
|
||||
/// Horizontal position in PDF points.
|
||||
pub x: f32,
|
||||
/// Vertical position in PDF points, increasing upward.
|
||||
pub y: f32,
|
||||
}
|
||||
|
||||
/// Affine transform between top-left pixel space and PDF page space.
|
||||
///
|
||||
/// Renderers create this from the page-space images of the bitmap corners.
|
||||
/// Keeping the coefficients in pdf-inspector makes [`RenderedPage`] neutral
|
||||
/// to the renderer implementation that produced it.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct PageTransform {
|
||||
forward: [f64; 6],
|
||||
inverse: [f64; 6],
|
||||
pixel_width: u32,
|
||||
pixel_height: u32,
|
||||
}
|
||||
|
||||
impl PageTransform {
|
||||
/// Builds a transform from the PDF-space images of device corners
|
||||
/// `(0, 0)`, `(pixel_width, 0)`, and `(0, pixel_height)`.
|
||||
pub fn from_corners(
|
||||
pixel_width: u32,
|
||||
pixel_height: u32,
|
||||
origin: (f64, f64),
|
||||
x_axis: (f64, f64),
|
||||
y_axis: (f64, f64),
|
||||
) -> Option<Self> {
|
||||
if pixel_width == 0 || pixel_height == 0 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let values = [origin.0, origin.1, x_axis.0, x_axis.1, y_axis.0, y_axis.1];
|
||||
if values.iter().any(|value| !value.is_finite()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let width = f64::from(pixel_width);
|
||||
let height = f64::from(pixel_height);
|
||||
let a = (x_axis.0 - origin.0) / width;
|
||||
let c = (x_axis.1 - origin.1) / width;
|
||||
let b = (y_axis.0 - origin.0) / height;
|
||||
let d = (y_axis.1 - origin.1) / height;
|
||||
let (e, f) = origin;
|
||||
let forward = [a, b, c, d, e, f];
|
||||
if forward.iter().any(|coefficient| !coefficient.is_finite()) {
|
||||
return None;
|
||||
}
|
||||
let determinant = a * d - b * c;
|
||||
if determinant == 0.0 || !determinant.is_finite() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let inverse_a = d / determinant;
|
||||
let inverse_b = -b / determinant;
|
||||
let inverse_c = -c / determinant;
|
||||
let inverse_d = a / determinant;
|
||||
let inverse_e = -(inverse_a * e + inverse_b * f);
|
||||
let inverse_f = -(inverse_c * e + inverse_d * f);
|
||||
let inverse = [
|
||||
inverse_a, inverse_b, inverse_c, inverse_d, inverse_e, inverse_f,
|
||||
];
|
||||
if inverse.iter().any(|coefficient| !coefficient.is_finite()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
Some(Self {
|
||||
forward,
|
||||
inverse,
|
||||
pixel_width,
|
||||
pixel_height,
|
||||
})
|
||||
}
|
||||
|
||||
/// Width of the bitmap this transform describes.
|
||||
pub fn pixel_width(&self) -> u32 {
|
||||
self.pixel_width
|
||||
}
|
||||
|
||||
/// Height of the bitmap this transform describes.
|
||||
pub fn pixel_height(&self) -> u32 {
|
||||
self.pixel_height
|
||||
}
|
||||
|
||||
/// Converts a bitmap point to PDF page space.
|
||||
pub fn pixel_to_page(&self, x: f64, y: f64) -> PagePoint {
|
||||
let [a, b, c, d, e, f] = self.forward;
|
||||
PagePoint {
|
||||
x: (a * x + b * y + e) as f32,
|
||||
y: (c * x + d * y + f) as f32,
|
||||
}
|
||||
}
|
||||
|
||||
/// Converts a PDF page-space point to bitmap coordinates.
|
||||
pub fn page_to_pixel(&self, x: f64, y: f64) -> (f64, f64) {
|
||||
let [a, b, c, d, e, f] = self.inverse;
|
||||
(a * x + b * y + e, c * x + d * y + f)
|
||||
}
|
||||
}
|
||||
|
||||
/// Invalid renderer output rejected by [`RenderedPage::new`].
|
||||
#[derive(Debug, Error, Clone, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum RenderBufferError {
|
||||
/// Page numbers are 1-indexed.
|
||||
#[error("rendered page number must be at least 1")]
|
||||
InvalidPageNumber,
|
||||
/// Bitmap dimensions must be non-zero.
|
||||
#[error("rendered bitmap dimensions must be non-zero")]
|
||||
InvalidDimensions,
|
||||
/// Page dimensions must be positive finite numbers.
|
||||
#[error("rendered PDF page dimensions must be positive and finite")]
|
||||
InvalidPageDimensions,
|
||||
/// Transform dimensions must match the bitmap dimensions.
|
||||
#[error("coordinate transform dimensions do not match the rendered bitmap")]
|
||||
TransformDimensions,
|
||||
/// Renderer did not provide an invertible finite coordinate transform.
|
||||
#[error("renderer returned an invalid coordinate transform")]
|
||||
InvalidTransform,
|
||||
/// The stride cannot hold one active row of pixels.
|
||||
#[error("pixel stride {stride} is shorter than the active row size {minimum}")]
|
||||
InvalidStride {
|
||||
/// Supplied bytes per row.
|
||||
stride: usize,
|
||||
/// Minimum bytes required for one row.
|
||||
minimum: usize,
|
||||
},
|
||||
/// Pixel buffer size is inconsistent with height and stride.
|
||||
#[error("pixel buffer has {actual} bytes; expected {expected}")]
|
||||
InvalidBufferLength {
|
||||
/// Actual byte count.
|
||||
actual: usize,
|
||||
/// Required byte count.
|
||||
expected: usize,
|
||||
},
|
||||
/// Dimension arithmetic overflowed the host address space.
|
||||
#[error("rendered bitmap dimensions overflow the host address space")]
|
||||
SizeOverflow,
|
||||
}
|
||||
|
||||
/// One rendered page with owned pixels and its pixel-to-PDF transform.
|
||||
///
|
||||
/// The value contains no live renderer, page, or document handles. It can be
|
||||
/// moved to an OCR worker and retained after rendering returns.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct RenderedPage {
|
||||
page: u32,
|
||||
page_width: f32,
|
||||
page_height: f32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
stride: usize,
|
||||
format: RenderPixelFormat,
|
||||
pixels: Vec<u8>,
|
||||
transform: PageTransform,
|
||||
}
|
||||
|
||||
impl RenderedPage {
|
||||
/// Creates a renderer-neutral owned page after validating its buffer.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn new(
|
||||
page: u32,
|
||||
page_width: f32,
|
||||
page_height: f32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
stride: usize,
|
||||
format: RenderPixelFormat,
|
||||
pixels: Vec<u8>,
|
||||
transform: PageTransform,
|
||||
) -> Result<Self, RenderBufferError> {
|
||||
if page == 0 {
|
||||
return Err(RenderBufferError::InvalidPageNumber);
|
||||
}
|
||||
if width == 0 || height == 0 {
|
||||
return Err(RenderBufferError::InvalidDimensions);
|
||||
}
|
||||
if page_width <= 0.0
|
||||
|| page_height <= 0.0
|
||||
|| !page_width.is_finite()
|
||||
|| !page_height.is_finite()
|
||||
{
|
||||
return Err(RenderBufferError::InvalidPageDimensions);
|
||||
}
|
||||
if transform.pixel_width() != width || transform.pixel_height() != height {
|
||||
return Err(RenderBufferError::TransformDimensions);
|
||||
}
|
||||
|
||||
let row_bytes = (width as usize)
|
||||
.checked_mul(format.bytes_per_pixel())
|
||||
.ok_or(RenderBufferError::SizeOverflow)?;
|
||||
if stride < row_bytes {
|
||||
return Err(RenderBufferError::InvalidStride {
|
||||
stride,
|
||||
minimum: row_bytes,
|
||||
});
|
||||
}
|
||||
let expected = stride
|
||||
.checked_mul(height as usize)
|
||||
.ok_or(RenderBufferError::SizeOverflow)?;
|
||||
if pixels.len() != expected {
|
||||
return Err(RenderBufferError::InvalidBufferLength {
|
||||
actual: pixels.len(),
|
||||
expected,
|
||||
});
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
page,
|
||||
page_width,
|
||||
page_height,
|
||||
width,
|
||||
height,
|
||||
stride,
|
||||
format,
|
||||
pixels,
|
||||
transform,
|
||||
})
|
||||
}
|
||||
|
||||
/// 1-indexed page number.
|
||||
pub fn page(&self) -> u32 {
|
||||
self.page
|
||||
}
|
||||
|
||||
/// Page width in PDF points after applying the page's rotation.
|
||||
pub fn page_width(&self) -> f32 {
|
||||
self.page_width
|
||||
}
|
||||
|
||||
/// Page height in PDF points after applying the page's rotation.
|
||||
pub fn page_height(&self) -> f32 {
|
||||
self.page_height
|
||||
}
|
||||
|
||||
/// Bitmap width in pixels.
|
||||
pub fn width(&self) -> u32 {
|
||||
self.width
|
||||
}
|
||||
|
||||
/// Bitmap height in pixels.
|
||||
pub fn height(&self) -> u32 {
|
||||
self.height
|
||||
}
|
||||
|
||||
/// Number of bytes between adjacent bitmap rows.
|
||||
pub fn stride(&self) -> usize {
|
||||
self.stride
|
||||
}
|
||||
|
||||
/// Pixel layout of [`pixels`](Self::pixels).
|
||||
pub fn format(&self) -> RenderPixelFormat {
|
||||
self.format
|
||||
}
|
||||
|
||||
/// Owned bitmap bytes, with rows ordered top-to-bottom.
|
||||
pub fn pixels(&self) -> &[u8] {
|
||||
&self.pixels
|
||||
}
|
||||
|
||||
/// Consumes the page and returns its pixel buffer.
|
||||
pub fn into_pixels(self) -> Vec<u8> {
|
||||
self.pixels
|
||||
}
|
||||
|
||||
/// Coordinate transform associated with the rendered page.
|
||||
pub fn transform(&self) -> PageTransform {
|
||||
self.transform
|
||||
}
|
||||
|
||||
/// Converts a bitmap point (top-left origin, y-down) to PDF page space
|
||||
/// (bottom-left origin, y-up).
|
||||
pub fn pixel_to_page(&self, x: f64, y: f64) -> PagePoint {
|
||||
self.transform.pixel_to_page(x, y)
|
||||
}
|
||||
|
||||
/// Converts a bitmap rectangle to the repository's existing PDF-space
|
||||
/// rectangle type. The returned page number remains 1-indexed.
|
||||
pub fn pixel_rect_to_pdf_rect(&self, x: f64, y: f64, width: f64, height: f64) -> PdfRect {
|
||||
let points = [
|
||||
self.transform.pixel_to_page(x, y),
|
||||
self.transform.pixel_to_page(x + width, y),
|
||||
self.transform.pixel_to_page(x, y + height),
|
||||
self.transform.pixel_to_page(x + width, y + height),
|
||||
];
|
||||
let left = points
|
||||
.iter()
|
||||
.map(|point| point.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let right = points
|
||||
.iter()
|
||||
.map(|point| point.x)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let bottom = points
|
||||
.iter()
|
||||
.map(|point| point.y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let top = points
|
||||
.iter()
|
||||
.map(|point| point.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
PdfRect {
|
||||
x: left,
|
||||
y: bottom,
|
||||
width: right - left,
|
||||
height: top - bottom,
|
||||
page: self.page,
|
||||
}
|
||||
}
|
||||
|
||||
/// Converts a PDF-space rectangle to bitmap coordinates
|
||||
/// `(x, y, width, height)` with a top-left origin.
|
||||
pub fn pdf_rect_to_pixel(&self, rect: &PdfRect) -> (f64, f64, f64, f64) {
|
||||
let left = f64::from(rect.x);
|
||||
let right = f64::from(rect.x + rect.width);
|
||||
let bottom = f64::from(rect.y);
|
||||
let top = f64::from(rect.y + rect.height);
|
||||
let points = [
|
||||
self.transform.page_to_pixel(left, bottom),
|
||||
self.transform.page_to_pixel(right, bottom),
|
||||
self.transform.page_to_pixel(left, top),
|
||||
self.transform.page_to_pixel(right, top),
|
||||
];
|
||||
let min_x = points
|
||||
.iter()
|
||||
.map(|point| point.0)
|
||||
.fold(f64::INFINITY, f64::min);
|
||||
let max_x = points
|
||||
.iter()
|
||||
.map(|point| point.0)
|
||||
.fold(f64::NEG_INFINITY, f64::max);
|
||||
let min_y = points
|
||||
.iter()
|
||||
.map(|point| point.1)
|
||||
.fold(f64::INFINITY, f64::min);
|
||||
let max_y = points
|
||||
.iter()
|
||||
.map(|point| point.1)
|
||||
.fold(f64::NEG_INFINITY, f64::max);
|
||||
(min_x, min_y, max_x - min_x, max_y - min_y)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn transform() -> PageTransform {
|
||||
PageTransform::from_corners(400, 200, (0.0, 100.0), (200.0, 100.0), (0.0, 0.0)).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transform_maps_both_directions_at_non_identity_scale() {
|
||||
let transform = transform();
|
||||
let point = transform.pixel_to_page(100.0, 50.0);
|
||||
assert!((point.x - 50.0).abs() < 1e-6);
|
||||
assert!((point.y - 75.0).abs() < 1e-6);
|
||||
let pixel = transform.page_to_pixel(f64::from(point.x), f64::from(point.y));
|
||||
assert!((pixel.0 - 100.0).abs() < 1e-6);
|
||||
assert!((pixel.1 - 50.0).abs() < 1e-6);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rendered_page_accepts_padding_and_validates_length() {
|
||||
let page = RenderedPage::new(
|
||||
1,
|
||||
200.0,
|
||||
100.0,
|
||||
400,
|
||||
200,
|
||||
1_204,
|
||||
RenderPixelFormat::Rgb8,
|
||||
vec![0; 1_204 * 200],
|
||||
transform(),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(page.stride(), 1_204);
|
||||
|
||||
assert!(matches!(
|
||||
RenderedPage::new(
|
||||
1,
|
||||
200.0,
|
||||
100.0,
|
||||
400,
|
||||
200,
|
||||
1_204,
|
||||
RenderPixelFormat::Rgb8,
|
||||
vec![0; 5],
|
||||
transform(),
|
||||
),
|
||||
Err(RenderBufferError::InvalidBufferLength { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rotated_transform_round_trips_rectangles() {
|
||||
let transform =
|
||||
PageTransform::from_corners(100, 200, (0.0, 0.0), (0.0, 100.0), (200.0, 0.0)).unwrap();
|
||||
let page = RenderedPage::new(
|
||||
1,
|
||||
200.0,
|
||||
100.0,
|
||||
100,
|
||||
200,
|
||||
300,
|
||||
RenderPixelFormat::Rgb8,
|
||||
vec![0; 300 * 200],
|
||||
transform,
|
||||
)
|
||||
.unwrap();
|
||||
let pdf = page.pixel_rect_to_pdf_rect(10.0, 20.0, 30.0, 40.0);
|
||||
let pixel = page.pdf_rect_to_pixel(&pdf);
|
||||
assert!((pixel.0 - 10.0).abs() < 1e-5);
|
||||
assert!((pixel.1 - 20.0).abs() < 1e-5);
|
||||
assert!((pixel.2 - 30.0).abs() < 1e-5);
|
||||
assert!((pixel.3 - 40.0).abs() < 1e-5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn skewed_transform_bounds_all_rectangle_corners() {
|
||||
let transform =
|
||||
PageTransform::from_corners(100, 100, (0.0, 100.0), (100.0, 125.0), (25.0, 0.0))
|
||||
.unwrap();
|
||||
let page = RenderedPage::new(
|
||||
1,
|
||||
125.0,
|
||||
125.0,
|
||||
100,
|
||||
100,
|
||||
300,
|
||||
RenderPixelFormat::Rgb8,
|
||||
vec![0; 30_000],
|
||||
transform,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let pdf = page.pixel_rect_to_pdf_rect(10.0, 20.0, 30.0, 40.0);
|
||||
assert!((pdf.x - 15.0).abs() < 1e-5);
|
||||
assert!((pdf.y - 42.5).abs() < 1e-5);
|
||||
assert!((pdf.width - 40.0).abs() < 1e-5);
|
||||
assert!((pdf.height - 47.5).abs() < 1e-5);
|
||||
|
||||
let pixels = page.pdf_rect_to_pixel(&pdf);
|
||||
assert!(pixels.0 <= 10.0 && pixels.1 <= 20.0);
|
||||
assert!(pixels.0 + pixels.2 >= 40.0 && pixels.1 + pixels.3 >= 60.0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,380 @@
|
||||
//! Page routing and renderer/OCR orchestration without Markdown fusion.
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
use std::error::Error;
|
||||
use std::time::Instant;
|
||||
|
||||
use thiserror::Error;
|
||||
|
||||
use super::{OcrEngine, OcrMode, OcrOptions, OcrPage, PageRenderer, RenderOptions, RenderedPage};
|
||||
|
||||
/// A rendered page paired with OCR output in the same bitmap coordinate space.
|
||||
#[derive(Debug)]
|
||||
pub struct RoutedOcrPage {
|
||||
/// Renderer-owned bitmap and pixel↔PDF transform.
|
||||
pub rendered: RenderedPage,
|
||||
/// Positioned OCR spans for the bitmap.
|
||||
pub ocr: OcrPage,
|
||||
}
|
||||
|
||||
/// Output of one selective OCR invocation.
|
||||
#[derive(Debug)]
|
||||
pub struct OcrRun {
|
||||
/// Pages processed in ascending document order.
|
||||
pub pages: Vec<RoutedOcrPage>,
|
||||
/// Total page-rendering wall time.
|
||||
pub render_time_ms: u64,
|
||||
/// Total engine wall time.
|
||||
pub ocr_time_ms: u64,
|
||||
}
|
||||
|
||||
/// Selects 1-indexed pages for OCR.
|
||||
///
|
||||
/// `recommended_pages` comes from pdf-inspector's existing detector/text
|
||||
/// quality signals. `selected_pages` is an optional user page filter. Results
|
||||
/// are validated, deduplicated, and returned in document order.
|
||||
pub fn route_ocr_pages(
|
||||
mode: OcrMode,
|
||||
page_count: u32,
|
||||
recommended_pages: &[u32],
|
||||
selected_pages: Option<&[u32]>,
|
||||
) -> Result<Vec<u32>, OcrRoutingError> {
|
||||
match mode {
|
||||
OcrMode::Off => Ok(Vec::new()),
|
||||
OcrMode::Auto => {
|
||||
let mut routed = validated_page_set("recommended", recommended_pages, page_count)?;
|
||||
if let Some(selected) = selected_pages {
|
||||
let selected = validated_page_set("selected", selected, page_count)?;
|
||||
routed.retain(|page| selected.contains(page));
|
||||
}
|
||||
Ok(routed.into_iter().collect())
|
||||
}
|
||||
OcrMode::Force => {
|
||||
let routed = selected_pages
|
||||
.map(|pages| validated_page_set("selected", pages, page_count))
|
||||
.transpose()?
|
||||
.unwrap_or_else(|| (1..=page_count).collect());
|
||||
Ok(routed.into_iter().collect())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Renders and recognizes already-routed pages while retaining transforms for
|
||||
/// the following fusion layer.
|
||||
///
|
||||
/// An empty page list returns without calling either dependency, which keeps
|
||||
/// model resolution and inference lazy when Auto routing finds no OCR work.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn run_ocr_pages<R, O>(
|
||||
renderer: &R,
|
||||
engine: &O,
|
||||
pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
password: Option<&str>,
|
||||
render_options: &RenderOptions,
|
||||
ocr_options: &OcrOptions,
|
||||
) -> Result<OcrRun, OcrRunError>
|
||||
where
|
||||
R: PageRenderer,
|
||||
O: OcrEngine,
|
||||
{
|
||||
if pages.is_empty() {
|
||||
return Ok(OcrRun {
|
||||
pages: Vec::new(),
|
||||
render_time_ms: 0,
|
||||
ocr_time_ms: 0,
|
||||
});
|
||||
}
|
||||
if ocr_options.mode == OcrMode::Off {
|
||||
return Err(OcrRunError::OcrDisabled);
|
||||
}
|
||||
|
||||
let render_started = Instant::now();
|
||||
let rendered = renderer
|
||||
.render_pages(pdf_bytes, pages, password, render_options)
|
||||
.map_err(|source| OcrRunError::Render {
|
||||
source: Box::new(source),
|
||||
})?;
|
||||
let render_time_ms = elapsed_ms(render_started);
|
||||
validate_page_order("renderer", pages, rendered.iter().map(RenderedPage::page))?;
|
||||
|
||||
let ocr_started = Instant::now();
|
||||
let recognized =
|
||||
engine
|
||||
.recognize(&rendered, ocr_options)
|
||||
.map_err(|source| OcrRunError::Ocr {
|
||||
source: Box::new(source),
|
||||
})?;
|
||||
let ocr_time_ms = elapsed_ms(ocr_started);
|
||||
validate_page_order("OCR engine", pages, recognized.iter().map(|page| page.page))?;
|
||||
|
||||
Ok(OcrRun {
|
||||
pages: rendered
|
||||
.into_iter()
|
||||
.zip(recognized)
|
||||
.map(|(rendered, ocr)| RoutedOcrPage { rendered, ocr })
|
||||
.collect(),
|
||||
render_time_ms,
|
||||
ocr_time_ms,
|
||||
})
|
||||
}
|
||||
|
||||
/// Invalid page routing or renderer/engine contract output.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrRoutingError {
|
||||
/// A page list contained zero or a page beyond the document.
|
||||
#[error("{source_name} OCR page {page} is outside the valid range 1..={page_count}")]
|
||||
InvalidPage {
|
||||
/// Page-list source.
|
||||
source_name: &'static str,
|
||||
/// Invalid 1-indexed page.
|
||||
page: u32,
|
||||
/// Document page count.
|
||||
page_count: u32,
|
||||
},
|
||||
}
|
||||
|
||||
/// Failures while rendering and recognizing a routed page set.
|
||||
#[derive(Debug, Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum OcrRunError {
|
||||
/// A non-empty route cannot execute with OCR disabled.
|
||||
#[error("cannot process routed pages while OCR mode is Off")]
|
||||
OcrDisabled,
|
||||
/// Page rasterization failed.
|
||||
#[error("page rendering failed: {source}")]
|
||||
Render {
|
||||
/// Renderer-specific failure.
|
||||
#[source]
|
||||
source: Box<dyn Error + Send + Sync>,
|
||||
},
|
||||
/// OCR inference failed.
|
||||
#[error("OCR inference failed: {source}")]
|
||||
Ocr {
|
||||
/// Engine-specific failure.
|
||||
#[source]
|
||||
source: Box<dyn Error + Send + Sync>,
|
||||
},
|
||||
/// A dependency returned the wrong count or order.
|
||||
#[error("{stage} returned pages {actual:?}; expected {expected:?}")]
|
||||
PageOrderMismatch {
|
||||
/// Dependency boundary that violated the contract.
|
||||
stage: &'static str,
|
||||
/// Requested 1-indexed pages.
|
||||
expected: Vec<u32>,
|
||||
/// Returned 1-indexed pages.
|
||||
actual: Vec<u32>,
|
||||
},
|
||||
}
|
||||
|
||||
fn validated_page_set(
|
||||
source_name: &'static str,
|
||||
pages: &[u32],
|
||||
page_count: u32,
|
||||
) -> Result<BTreeSet<u32>, OcrRoutingError> {
|
||||
let mut result = BTreeSet::new();
|
||||
for &page in pages {
|
||||
if page == 0 || page > page_count {
|
||||
return Err(OcrRoutingError::InvalidPage {
|
||||
source_name,
|
||||
page,
|
||||
page_count,
|
||||
});
|
||||
}
|
||||
result.insert(page);
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn validate_page_order(
|
||||
stage: &'static str,
|
||||
expected: &[u32],
|
||||
actual: impl IntoIterator<Item = u32>,
|
||||
) -> Result<(), OcrRunError> {
|
||||
let actual: Vec<u32> = actual.into_iter().collect();
|
||||
if actual == expected {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(OcrRunError::PageOrderMismatch {
|
||||
stage,
|
||||
expected: expected.to_vec(),
|
||||
actual,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(started: Instant) -> u64 {
|
||||
u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::vision::{
|
||||
ImagePoint, ImageQuad, ModelIdentity, OcrSpan, PageTransform, RenderBufferError,
|
||||
RenderPixelFormat,
|
||||
};
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
#[error("fake failure")]
|
||||
struct FakeError;
|
||||
|
||||
struct FakeRenderer;
|
||||
|
||||
impl PageRenderer for FakeRenderer {
|
||||
type Error = FakeError;
|
||||
|
||||
fn render_pages(
|
||||
&self,
|
||||
_pdf_bytes: &[u8],
|
||||
pages: &[u32],
|
||||
_password: Option<&str>,
|
||||
_options: &RenderOptions,
|
||||
) -> Result<Vec<RenderedPage>, Self::Error> {
|
||||
pages
|
||||
.iter()
|
||||
.copied()
|
||||
.map(rendered_page)
|
||||
.collect::<Result<_, _>>()
|
||||
.map_err(|_| FakeError)
|
||||
}
|
||||
}
|
||||
|
||||
struct FakeEngine {
|
||||
model: ModelIdentity,
|
||||
}
|
||||
|
||||
impl FakeEngine {
|
||||
fn new() -> Self {
|
||||
Self {
|
||||
model: ModelIdentity::new("fake", "v1"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl OcrEngine for FakeEngine {
|
||||
type Error = FakeError;
|
||||
|
||||
fn model(&self) -> &ModelIdentity {
|
||||
&self.model
|
||||
}
|
||||
|
||||
fn recognize(
|
||||
&self,
|
||||
pages: &[RenderedPage],
|
||||
_options: &OcrOptions,
|
||||
) -> Result<Vec<OcrPage>, Self::Error> {
|
||||
Ok(pages
|
||||
.iter()
|
||||
.map(|page| OcrPage {
|
||||
page: page.page(),
|
||||
spans: vec![OcrSpan {
|
||||
text: format!("page {}", page.page()),
|
||||
polygon: ImageQuad::new([
|
||||
ImagePoint::new(0.0, 0.0),
|
||||
ImagePoint::new(1.0, 0.0),
|
||||
ImagePoint::new(1.0, 1.0),
|
||||
ImagePoint::new(0.0, 1.0),
|
||||
]),
|
||||
confidence: 0.9,
|
||||
orientation_degrees: None,
|
||||
}],
|
||||
mean_confidence: Some(0.9),
|
||||
model: self.model.clone(),
|
||||
processing_time_ms: 1,
|
||||
warnings: Vec::new(),
|
||||
})
|
||||
.collect())
|
||||
}
|
||||
}
|
||||
|
||||
fn rendered_page(page: u32) -> Result<RenderedPage, RenderBufferError> {
|
||||
let transform =
|
||||
PageTransform::from_corners(1, 1, (0.0, 1.0), (1.0, 1.0), (0.0, 0.0)).unwrap();
|
||||
RenderedPage::new(
|
||||
page,
|
||||
1.0,
|
||||
1.0,
|
||||
1,
|
||||
1,
|
||||
3,
|
||||
RenderPixelFormat::Rgb8,
|
||||
vec![255; 3],
|
||||
transform,
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn off_auto_and_force_route_expected_pages() {
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Off, 0, &[99], Some(&[0])).unwrap(),
|
||||
Vec::<u32>::new()
|
||||
);
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Auto, 5, &[5, 3, 3, 1], Some(&[2, 3, 5])).unwrap(),
|
||||
vec![3, 5]
|
||||
);
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Force, 4, &[], None).unwrap(),
|
||||
vec![1, 2, 3, 4]
|
||||
);
|
||||
assert_eq!(
|
||||
route_ocr_pages(OcrMode::Force, 4, &[], Some(&[4, 2, 2])).unwrap(),
|
||||
vec![2, 4]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn routing_rejects_invalid_page_numbers() {
|
||||
assert!(matches!(
|
||||
route_ocr_pages(OcrMode::Auto, 2, &[0], None),
|
||||
Err(OcrRoutingError::InvalidPage { page: 0, .. })
|
||||
));
|
||||
assert!(matches!(
|
||||
route_ocr_pages(OcrMode::Force, 2, &[], Some(&[3])),
|
||||
Err(OcrRoutingError::InvalidPage { page: 3, .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn run_retains_render_transforms_and_input_order() {
|
||||
let options = OcrOptions::new().mode(OcrMode::Auto);
|
||||
let run = run_ocr_pages(
|
||||
&FakeRenderer,
|
||||
&FakeEngine::new(),
|
||||
b"pdf",
|
||||
&[2, 4],
|
||||
None,
|
||||
&RenderOptions::new(),
|
||||
&options,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
run.pages
|
||||
.iter()
|
||||
.map(|page| page.rendered.page())
|
||||
.collect::<Vec<_>>(),
|
||||
vec![2, 4]
|
||||
);
|
||||
assert_eq!(run.pages[1].ocr.spans[0].text, "page 4");
|
||||
assert_eq!(run.pages[1].rendered.pixel_to_page(0.0, 0.0).y, 1.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_route_is_a_noop_even_when_ocr_is_off() {
|
||||
let run = run_ocr_pages(
|
||||
&FakeRenderer,
|
||||
&FakeEngine::new(),
|
||||
b"pdf",
|
||||
&[],
|
||||
None,
|
||||
&RenderOptions::new(),
|
||||
&OcrOptions::new(),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(run.pages.is_empty());
|
||||
assert_eq!(run.render_time_ms, 0);
|
||||
assert_eq!(run.ocr_time_ms, 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
#![cfg(all(feature = "render-pdfium", not(target_arch = "wasm32")))]
|
||||
|
||||
use pdf_inspector::vision::{PdfiumRenderer, RenderError, RenderOptions, RenderPixelFormat};
|
||||
|
||||
fn load_renderer() -> Option<PdfiumRenderer> {
|
||||
match PdfiumRenderer::load() {
|
||||
Ok(renderer) => Some(renderer),
|
||||
Err(RenderError::Pdfium(firecrawl_pdfium::Error::Load(
|
||||
firecrawl_pdfium::LoadError::LibraryNotFound { .. },
|
||||
))) => {
|
||||
eprintln!("skipping PDFium runtime test because no native library is installed");
|
||||
None
|
||||
}
|
||||
Err(error) => panic!("failed to load PDFium: {error}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn renders_owned_rgb_page_and_round_trips_coordinates() {
|
||||
let Some(renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let pages = renderer
|
||||
.render_pages(
|
||||
&bytes,
|
||||
&[1],
|
||||
None,
|
||||
&RenderOptions::new().dpi(150.0).form_fields(false),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(pages.len(), 1);
|
||||
let page = &pages[0];
|
||||
assert_eq!(page.page(), 1);
|
||||
assert_eq!(page.format(), RenderPixelFormat::Rgb8);
|
||||
assert_eq!(page.stride(), page.width() as usize * 3);
|
||||
assert_eq!(page.pixels().len(), page.stride() * page.height() as usize);
|
||||
assert!((page.width() as f32 - page.page_width()).abs() > 1.0);
|
||||
|
||||
let pdf_rect = page.pixel_rect_to_pdf_rect(10.0, 10.0, 20.0, 12.0);
|
||||
let pixel_rect = page.pdf_rect_to_pixel(&pdf_rect);
|
||||
assert!((pixel_rect.0 - 10.0).abs() < 0.01);
|
||||
assert!((pixel_rect.1 - 10.0).abs() < 0.01);
|
||||
assert!((pixel_rect.2 - 20.0).abs() < 0.01);
|
||||
assert!((pixel_rect.3 - 12.0).abs() < 0.01);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_zero_and_out_of_range_page_numbers() {
|
||||
let Some(renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
|
||||
assert!(matches!(
|
||||
renderer.render_pages(&bytes, &[0], None, &RenderOptions::new()),
|
||||
Err(RenderError::InvalidPageNumber)
|
||||
));
|
||||
|
||||
assert!(matches!(
|
||||
renderer.render_pages(&bytes, &[u32::MAX], None, &RenderOptions::new()),
|
||||
Err(RenderError::PageOutOfBounds { .. })
|
||||
));
|
||||
}
|
||||
@@ -0,0 +1,159 @@
|
||||
#![cfg(all(feature = "ocr-oar", not(target_arch = "wasm32")))]
|
||||
|
||||
#[cfg(feature = "ocr")]
|
||||
use pdf_inspector::vision::{
|
||||
process_pdf_with_ocr_mem, ModelDownloadPolicy, OcrPdfOptions, PageContentSource,
|
||||
};
|
||||
use pdf_inspector::vision::{
|
||||
ModelStore, OarOcrEngine, OcrEngine, OcrMode, OcrOptions, PageTransform, RenderPixelFormat,
|
||||
RenderedPage, PP_OCR_V6_SMALL,
|
||||
};
|
||||
#[cfg(feature = "render-pdfium")]
|
||||
use pdf_inspector::vision::{PdfiumRenderer, RenderError, RenderOptions};
|
||||
|
||||
const MODEL_DIRECTORY_ENV: &str = "PDF_INSPECTOR_OCR_TEST_MODELS";
|
||||
const IMAGE_ENV: &str = "PDF_INSPECTOR_OCR_TEST_IMAGE";
|
||||
const EXPECTED_TEXT_ENV: &str = "PDF_INSPECTOR_OCR_TEST_EXPECTED";
|
||||
|
||||
#[cfg(feature = "render-pdfium")]
|
||||
fn load_renderer() -> Option<PdfiumRenderer> {
|
||||
match PdfiumRenderer::load() {
|
||||
Ok(renderer) => Some(renderer),
|
||||
Err(RenderError::Pdfium(firecrawl_pdfium::Error::Load(
|
||||
firecrawl_pdfium::LoadError::LibraryNotFound { .. },
|
||||
))) => {
|
||||
eprintln!("skipping OCR runtime test because no native PDFium library is installed");
|
||||
None
|
||||
}
|
||||
Err(error) => panic!("failed to load PDFium: {error}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recognizes_an_rgb_image_with_verified_models() {
|
||||
let Some(model_directory) = std::env::var_os(MODEL_DIRECTORY_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {MODEL_DIRECTORY_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
let Some(image_path) = std::env::var_os(IMAGE_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {IMAGE_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
|
||||
let image = image::open(image_path).unwrap().into_rgb8();
|
||||
let (width, height) = image.dimensions();
|
||||
let transform = PageTransform::from_corners(
|
||||
width,
|
||||
height,
|
||||
(0.0, f64::from(height)),
|
||||
(f64::from(width), f64::from(height)),
|
||||
(0.0, 0.0),
|
||||
)
|
||||
.unwrap();
|
||||
let page = RenderedPage::new(
|
||||
1,
|
||||
width as f32,
|
||||
height as f32,
|
||||
width,
|
||||
height,
|
||||
width as usize * 3,
|
||||
RenderPixelFormat::Rgb8,
|
||||
image.into_raw(),
|
||||
transform,
|
||||
)
|
||||
.unwrap();
|
||||
let results = recognize(&model_directory, &[page]);
|
||||
assert_usable_result(&results);
|
||||
|
||||
let text = results[0]
|
||||
.spans
|
||||
.iter()
|
||||
.map(|span| span.text.as_str())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
eprintln!("recognized: {text}");
|
||||
if let Ok(expected) = std::env::var(EXPECTED_TEXT_ENV) {
|
||||
assert!(
|
||||
text.to_lowercase().contains(&expected.to_lowercase()),
|
||||
"expected OCR output to contain {expected:?}, got {text:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "render-pdfium")]
|
||||
#[test]
|
||||
fn recognizes_a_pdfium_rendered_fixture_with_verified_models() {
|
||||
let Some(model_directory) = std::env::var_os(MODEL_DIRECTORY_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {MODEL_DIRECTORY_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
let Some(renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let pages = renderer
|
||||
.render_pages(
|
||||
&bytes,
|
||||
&[1],
|
||||
None,
|
||||
&RenderOptions::new().dpi(150.0).form_fields(false),
|
||||
)
|
||||
.unwrap();
|
||||
let results = recognize(&model_directory, &pages);
|
||||
assert_usable_result(&results);
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "ocr", feature = "render-pdfium"))]
|
||||
#[test]
|
||||
fn complete_ocr_pipeline_routes_and_assembles_a_scanned_fixture() {
|
||||
let Some(model_directory) = std::env::var_os(MODEL_DIRECTORY_ENV) else {
|
||||
eprintln!("skipping OCR runtime test because {MODEL_DIRECTORY_ENV} is not set");
|
||||
return;
|
||||
};
|
||||
let Some(_renderer) = load_renderer() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let bytes = std::fs::read("tests/fixtures/scan_with_native_header_text.pdf").unwrap();
|
||||
let ocr = OcrOptions::new()
|
||||
.mode(OcrMode::Auto)
|
||||
.minimum_confidence(0.3)
|
||||
.model_directory(model_directory)
|
||||
.model_downloads(ModelDownloadPolicy::Offline);
|
||||
let result = process_pdf_with_ocr_mem(&bytes, OcrPdfOptions::new().ocr(ocr)).unwrap();
|
||||
|
||||
assert_eq!(result.pages_routed_to_ocr, vec![1]);
|
||||
assert!(!result.markdown.trim().is_empty());
|
||||
assert_eq!(result.pages[0].provenance.source, PageContentSource::Ocr);
|
||||
assert_eq!(
|
||||
result.pages[0].provenance.ocr_model.as_ref().unwrap().name,
|
||||
PP_OCR_V6_SMALL.id
|
||||
);
|
||||
}
|
||||
|
||||
fn recognize(
|
||||
model_directory: &std::ffi::OsStr,
|
||||
pages: &[RenderedPage],
|
||||
) -> Vec<pdf_inspector::vision::OcrPage> {
|
||||
let store = ModelStore::new(model_directory).override_root(model_directory);
|
||||
let models = store.resolve(&PP_OCR_V6_SMALL).unwrap();
|
||||
let engine = OarOcrEngine::from_models(&models).unwrap();
|
||||
engine
|
||||
.recognize(
|
||||
pages,
|
||||
&OcrOptions::new()
|
||||
.mode(OcrMode::Force)
|
||||
.minimum_confidence(0.3),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn assert_usable_result(results: &[pdf_inspector::vision::OcrPage]) {
|
||||
assert_eq!(results.len(), 1);
|
||||
assert_eq!(results[0].page, 1);
|
||||
assert_eq!(results[0].model.name, PP_OCR_V6_SMALL.id);
|
||||
assert_eq!(results[0].model.revision, PP_OCR_V6_SMALL.revision);
|
||||
assert!(!results[0].spans.is_empty());
|
||||
assert!(results[0].spans.iter().all(|span| span.confidence >= 0.3));
|
||||
}
|
||||
Reference in New Issue
Block a user