Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
062ede203e | ||
|
|
16fbe29b52 |
@@ -180,31 +180,18 @@ jobs:
|
||||
test -n "$ort_path"
|
||||
echo "ORT_DYLIB_PATH=$ort_path" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Configure isolated model cache
|
||||
shell: bash
|
||||
run: echo "PDF_INSPECTOR_MODEL_CACHE=$RUNNER_TEMP/pdf-inspector-models" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build OCR CLI
|
||||
run: cargo build --features ocr --bin pdf2md
|
||||
|
||||
- name: Test PDFium runtime
|
||||
run: cargo test --features ocr --test local_render_tests
|
||||
|
||||
- name: Provision OCR model cache
|
||||
shell: bash
|
||||
run: |
|
||||
target/debug/pdf2md \
|
||||
tests/fixtures/scan_with_native_header_text.pdf \
|
||||
--ocr force \
|
||||
--json > /dev/null
|
||||
|
||||
- name: Run OCR CLI
|
||||
shell: bash
|
||||
run: |
|
||||
target/debug/pdf2md \
|
||||
tests/fixtures/scan_with_native_header_text.pdf \
|
||||
--ocr auto \
|
||||
--ocr-offline \
|
||||
--json > "$RUNNER_TEMP/ocr-result.json"
|
||||
|
||||
- name: Validate OCR JSON contract
|
||||
@@ -226,12 +213,6 @@ jobs:
|
||||
assert "layout_ms" not in result["pages"][0]["timings"]
|
||||
PY
|
||||
|
||||
- name: Run OCR launch smoke set
|
||||
shell: bash
|
||||
run: |
|
||||
export PDF_INSPECTOR_OCR_TEST_MODELS="$PDF_INSPECTOR_MODEL_CACHE/pp-ocrv6-small/oar-ocr-v0.7.0"
|
||||
cargo test --features ocr --test ocr_tests -- --nocapture
|
||||
|
||||
- name: Build Node binding
|
||||
working-directory: napi
|
||||
run: |
|
||||
|
||||
@@ -86,14 +86,12 @@ jobs:
|
||||
name: Build ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
docker-options: -e CFLAGS_aarch64_unknown_linux_gnu=-D__ARM_ARCH=8
|
||||
# macos-13 was retired by GitHub; macos-15-intel is the remaining
|
||||
# Intel runner label (available through 2027).
|
||||
- os: macos-15-intel
|
||||
@@ -115,9 +113,6 @@ jobs:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
# The manylinux AArch64 GCC omits this macro while preprocessing
|
||||
# ring's assembly. AArch64 is ARMv8 by definition.
|
||||
docker-options: ${{ matrix.docker-options }}
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
|
||||
@@ -60,7 +60,6 @@ jobs:
|
||||
name: Build ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
@@ -71,7 +70,6 @@ jobs:
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
build-flags: --use-napi-cross
|
||||
cflags: -D__ARM_ARCH=8
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
@@ -128,10 +126,6 @@ jobs:
|
||||
|
||||
- name: Build native addon
|
||||
working-directory: napi
|
||||
env:
|
||||
# napi-cross's old AArch64 GCC omits this predefined macro while
|
||||
# preprocessing ring's assembly. AArch64 is ARMv8 by definition.
|
||||
CFLAGS_aarch64_unknown_linux_gnu: ${{ matrix.cflags }}
|
||||
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
|
||||
|
||||
- name: Upload native binary
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
|
||||
@@ -48,7 +48,8 @@ Use the [paired benchmark harness](docs/benchmarking.md) to compare two local bu
|
||||
### Python
|
||||
|
||||
```bash
|
||||
pip install pdf-inspector
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
```
|
||||
|
||||
```python
|
||||
@@ -181,9 +182,8 @@ and confidence, warnings, and pages recommended for the hosted document
|
||||
pipeline. Native Python and Node packages expose the same pipeline without a
|
||||
source-build feature. All native entry points still require separately
|
||||
installed PDFium and ONNX Runtime libraries only when OCR is routed. See the
|
||||
[OCR runtime setup guide](docs/ocr-runtime.md) for pinned downloads, platform
|
||||
support, model-cache behavior, and hosted-fallback integration. See the
|
||||
[Rust API guide](docs/rust-api.md#complete-ocr-api) for lower-level controls.
|
||||
[Rust API guide](docs/rust-api.md#complete-ocr-api) for model cache and offline
|
||||
configuration.
|
||||
|
||||
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
||||
|
||||
|
||||
@@ -1,97 +0,0 @@
|
||||
# OCR runtime setup
|
||||
|
||||
Selective OCR is available from the Rust library and CLI, Python, and Node.js.
|
||||
Clean native-text documents do not load an OCR dependency or download a model.
|
||||
When `auto` routes at least one page, the process needs PDFium, ONNX Runtime,
|
||||
and the pinned PP-OCRv6 Small model set.
|
||||
|
||||
## Validated versions
|
||||
|
||||
The reproducible runtime path uses these builds:
|
||||
|
||||
- [Firecrawl PDFium `native-v7988`](https://github.com/firecrawl/pdfium-rs/releases/tag/native-v7988),
|
||||
containing PDFium `153.0.7988.0`
|
||||
- [ONNX Runtime `1.27.0`](https://github.com/microsoft/onnxruntime/releases/tag/v1.27.0)
|
||||
- PP-OCRv6 Small artifact revision `oar-ocr-v0.7.0`
|
||||
|
||||
Use these versions for the reproducible path. Other compatible shared-library
|
||||
builds may work, but are not part of the release smoke test.
|
||||
|
||||
## Install the shared libraries
|
||||
|
||||
Download and extract the matching archives:
|
||||
|
||||
| Platform | PDFium asset | ONNX Runtime asset |
|
||||
|---|---|---|
|
||||
| Linux x64 | `firecrawl-pdfium-linux-x64.tgz` | `onnxruntime-linux-x64-1.27.0.tgz` |
|
||||
| Linux ARM64 | `firecrawl-pdfium-linux-arm64.tgz` | `onnxruntime-linux-aarch64-1.27.0.tgz` |
|
||||
| macOS Apple Silicon | `firecrawl-pdfium-mac-arm64.tgz` | `onnxruntime-osx-arm64-1.27.0.tgz` |
|
||||
| Windows x64 | `firecrawl-pdfium-win-x64.tgz` | `onnxruntime-win-x64-1.27.0.zip` |
|
||||
|
||||
The PDFium release publishes `SHA256SUMS`, build provenance, license files,
|
||||
and an SPDX document for every platform archive. GitHub publishes a SHA-256
|
||||
digest with each ONNX Runtime asset.
|
||||
|
||||
Point pdf-inspector at the extracted shared libraries when they are not on the
|
||||
platform library search path:
|
||||
|
||||
```bash
|
||||
export PDFIUM_LIB_PATH=/absolute/path/to/libpdfium.so
|
||||
export ORT_DYLIB_PATH=/absolute/path/to/libonnxruntime.so
|
||||
pdf2md scan.pdf --ocr auto --json
|
||||
```
|
||||
|
||||
On macOS the filenames end in `.dylib`. On Windows, use PowerShell and point
|
||||
the variables at `pdfium.dll` and `onnxruntime.dll`:
|
||||
|
||||
```powershell
|
||||
$env:PDFIUM_LIB_PATH = "C:\absolute\path\to\pdfium.dll"
|
||||
$env:ORT_DYLIB_PATH = "C:\absolute\path\to\onnxruntime.dll"
|
||||
pdf2md scan.pdf --ocr auto --json
|
||||
```
|
||||
|
||||
The native extraction packages also support platforms without these exact
|
||||
runtime assets. In particular, the Python package has an Intel macOS wheel,
|
||||
but ONNX Runtime 1.27.0 does not publish an Intel macOS archive; local OCR on
|
||||
that target requires a compatible custom ONNX Runtime build.
|
||||
|
||||
The full OCR path is exercised end to end on Linux x64 in CI. macOS and
|
||||
Windows compile and run the feature's platform-independent tests, while their
|
||||
external-runtime paths should be treated as preview until equivalent smoke
|
||||
jobs are added.
|
||||
|
||||
## Model cache and offline mode
|
||||
|
||||
The first routed page downloads and SHA-256-verifies three pinned artifacts:
|
||||
the detection model, recognition model, and character dictionary. Together
|
||||
they are about 31 MB. They are stored below the platform cache directory.
|
||||
Set `PDF_INSPECTOR_MODEL_CACHE` to choose a managed cache root.
|
||||
|
||||
For hermetic deployments, populate the model directory ahead of time and use
|
||||
the language-specific offline option:
|
||||
|
||||
- CLI: `--ocr-offline --ocr-model-dir /models/pp-ocrv6-small`
|
||||
- Rust: `ModelDownloadPolicy::Offline` with `OcrOptions::model_directory`
|
||||
- Python: `offline=True, model_directory="/models/pp-ocrv6-small"`
|
||||
- Node.js: `offline: true, modelDirectory: "/models/pp-ocrv6-small"`
|
||||
|
||||
The model artifacts come from
|
||||
[`GreatV/oar-ocr`](https://github.com/GreatV/oar-ocr/releases/tag/v0.7.0),
|
||||
whose OCR implementation and upstream PaddleOCR project use Apache-2.0
|
||||
licensing. Models are downloaded at runtime and are not embedded in any
|
||||
pdf-inspector package.
|
||||
|
||||
## Hosted fallback boundary
|
||||
|
||||
`pages_recommending_hosted` is available after the local pipeline completes.
|
||||
It marks pages whose completed OCR result is empty, low-confidence, or still
|
||||
appears incomplete.
|
||||
|
||||
Setup and execution failures happen before that result exists. A missing or
|
||||
incompatible PDFium/ONNX Runtime library, failed model acquisition, or OCR
|
||||
execution error is returned as an error. A downstream integration that has a
|
||||
hosted parser should catch that error and route the document to the hosted
|
||||
path. This keeps deployment problems distinct from page-quality judgments.
|
||||
|
||||
In `auto`, documents with no routed pages return successfully without touching
|
||||
PDFium, ONNX Runtime, the model cache, or the network.
|
||||
+1
-3
@@ -44,9 +44,7 @@ OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||
platform library search path. The pinned OCR model set is downloaded and
|
||||
checksum-verified on the first routed page; use `offline=True` with a warm
|
||||
cache or `model_directory` to prohibit network access. See the
|
||||
[OCR runtime setup guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md)
|
||||
for pinned downloads, supported platforms, and hosted-fallback behavior.
|
||||
cache or `model_directory` to prohibit network access.
|
||||
|
||||
## Usage
|
||||
|
||||
|
||||
@@ -383,10 +383,6 @@ shape; `Force` renders every selected page. OCR uses the existing deterministic
|
||||
table, column, reading-order, and Markdown assembly path; no learned layout
|
||||
model is included.
|
||||
|
||||
The [OCR runtime setup guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md)
|
||||
lists the pinned PDFium and ONNX Runtime builds, environment variables, model
|
||||
cache behavior, and the error boundary downstream hosted fallbacks should use.
|
||||
|
||||
For ambiguous mixed pages, `Auto` privately retains clean native fragments
|
||||
instead of discarding them when OCR is selected. After recognition it compares
|
||||
script-agnostic text quality, OCR confidence, character overlap, and material
|
||||
|
||||
Generated
+2
-2
@@ -2089,7 +2089,7 @@ checksum = "2ee67f1008b1ba2321834326597b8e186293b049a023cdef258527550b9935b4"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
dependencies = [
|
||||
"dirs",
|
||||
"env_logger",
|
||||
@@ -2114,7 +2114,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
+1
-3
@@ -41,9 +41,7 @@ OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||
platform library search path. The pinned OCR model set is downloaded and
|
||||
checksum-verified on the first routed page; use `offline: true` with a warm
|
||||
cache or `modelDirectory` to prohibit network access. See the
|
||||
[OCR runtime setup guide](https://github.com/firecrawl/pdf-inspector/blob/main/docs/ocr-runtime.md)
|
||||
for pinned downloads, supported platforms, and hosted-fallback behavior.
|
||||
cache or `modelDirectory` to prohibit network access.
|
||||
|
||||
## API
|
||||
|
||||
|
||||
+6
-6
@@ -8,12 +8,12 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.2",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+7
-7
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.15.0",
|
||||
"version": "1.14.2",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -52,11 +52,11 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.15.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.15.0"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.2",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.2"
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ build-backend = "maturin"
|
||||
name = "pdf-inspector"
|
||||
# Keep package versions in sync with `python3 scripts/version.py <version>`.
|
||||
# CI publishes automatically when the synchronized change lands on main.
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
+1
-1
@@ -975,7 +975,7 @@ result = pdf_inspector.<span class="fn">process_pdf</span>(<span class="str">"do
|
||||
<script>
|
||||
(() => {
|
||||
const MAX_FILE_SIZE = 25 * 1024 * 1024;
|
||||
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@1.15.0/pdf_inspector_wasm.js";
|
||||
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@1.14.2/pdf_inspector_wasm.js";
|
||||
const input = document.querySelector("#pdf-input");
|
||||
const dropZone = document.querySelector("#drop-zone");
|
||||
const filePanel = document.querySelector("#demo-file");
|
||||
|
||||
+12
-150
@@ -403,15 +403,8 @@ pub(crate) fn detect_from_document(
|
||||
&& !(analysis.has_decodable_text_fonts && analysis.text_operator_count >= 10);
|
||||
let looks_like_scan =
|
||||
analysis.image_count <= 1 && analysis.text_operator_count < 50 && alphanum_low;
|
||||
// A template-image page below the `pages_with_text` floor is
|
||||
// a scan with incidental chrome (masthead, stamp, date line)
|
||||
// even when that chrome is diverse, decodable text — keep
|
||||
// this in sync with `page_ocr_signals`.
|
||||
let sparse_text_over_scan = analysis.has_template_image
|
||||
&& analysis.text_operator_count < config.min_text_ops_per_page.max(10);
|
||||
if (analysis.has_template_image && looks_like_scan)
|
||||
|| analysis.has_vector_text
|
||||
|| sparse_text_over_scan
|
||||
|| (analysis.text_operator_count < config.min_text_ops_per_page
|
||||
&& analysis.has_images)
|
||||
{
|
||||
@@ -1802,24 +1795,17 @@ pub(crate) fn analyze_page_images(doc: &Document, page_id: ObjectId) -> (bool, u
|
||||
/// low alphanumeric diversity in raw string operands (unless decodable
|
||||
/// CID/ToUnicode fonts explain that away) — the gate used for
|
||||
/// `pages_with_template_images` and Mixed-type per-page routing.
|
||||
/// 2. Insufficient real text volume, using the same `effective_min_ops`
|
||||
/// floor (`min_text_ops_per_page.max(10)`) that `pages_with_text`
|
||||
/// applies to image-bearing pages. That floor is a per-page judgment,
|
||||
/// not part of the cross-page aggregate: classification counts a
|
||||
/// template-image page with fewer ops as textless and routes it to OCR,
|
||||
/// so this function must agree. The lower bare threshold (3) let a
|
||||
/// full-page scan carrying a small native masthead — a newspaper
|
||||
/// header, stamp, or date line of ~4 diverse, decodable text ops —
|
||||
/// extract as "a text page" here while whole-document classification
|
||||
/// called the same page scanned, silently dropping the page body from
|
||||
/// OCR routing. `alphanum_low` can't catch that case: masthead chrome
|
||||
/// is real text, so its byte diversity is high.
|
||||
///
|
||||
/// This function always evaluates against `DetectionConfig::default()` —
|
||||
/// it has no config parameter, and the per-page extraction path that calls
|
||||
/// it never carries one. A caller passing a custom `min_text_ops_per_page`
|
||||
/// to `detect_from_document` affects whole-document detection only; the
|
||||
/// two paths agree under the default configuration.
|
||||
/// 2. Insufficient real text volume, using `DetectionConfig::default()`'s
|
||||
/// `min_text_ops_per_page` (3) — the same threshold Mixed-type per-page
|
||||
/// routing applies via `text_operator_count < config.min_text_ops_per_page
|
||||
/// && has_images` (simplified here since a template image implies
|
||||
/// `has_images`). Deliberately *not* the higher `effective_min_ops`
|
||||
/// floor (`min_text_ops_per_page.max(10)`) that whole-document
|
||||
/// `PdfType::ImageBased`/`Scanned` classification uses for
|
||||
/// `pages_with_text` — that's a cross-page aggregate decision this
|
||||
/// per-page function has no way to replicate exactly, and the lower
|
||||
/// per-page threshold is the one a single page's own signals can
|
||||
/// actually agree with.
|
||||
///
|
||||
/// `has_vector_text` is true when a page has vector-outlined text (glyphs
|
||||
/// drawn as paths rather than shown via text-showing operators) —
|
||||
@@ -1841,7 +1827,7 @@ pub(crate) fn page_ocr_signals(doc: &Document, page_id: ObjectId) -> (bool, bool
|
||||
let looks_like_scan =
|
||||
analysis.image_count <= 1 && analysis.text_operator_count < 50 && alphanum_low;
|
||||
let insufficient_text =
|
||||
analysis.text_operator_count < DetectionConfig::default().min_text_ops_per_page.max(10);
|
||||
analysis.text_operator_count < DetectionConfig::default().min_text_ops_per_page;
|
||||
looks_like_scan || insufficient_text
|
||||
};
|
||||
|
||||
@@ -3009,130 +2995,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
// ---------- masthead-over-scan tests: template image + sparse chrome ----------
|
||||
|
||||
/// Builds a page whose only image is a full-page scan inside a Form
|
||||
/// XObject, plus `masthead_ops` native text-show ops of diverse,
|
||||
/// decodable chrome (newspaper masthead / date line style).
|
||||
fn masthead_scan_page(masthead_lines: &[&str]) -> (Document, ObjectId) {
|
||||
use lopdf::dictionary;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let pages_id = doc.new_object_id();
|
||||
let page_id = doc.new_object_id();
|
||||
|
||||
let image_id = doc.add_object(Object::Stream(lopdf::Stream::new(
|
||||
dictionary! {
|
||||
"Type" => "XObject",
|
||||
"Subtype" => Object::Name(b"Image".to_vec()),
|
||||
"Width" => Object::Integer(1500),
|
||||
"Height" => Object::Integer(2383),
|
||||
},
|
||||
Vec::new(),
|
||||
)));
|
||||
let form_id = doc.add_object(Object::Stream(lopdf::Stream::new(
|
||||
dictionary! {
|
||||
"Type" => "XObject",
|
||||
"Subtype" => Object::Name(b"Form".to_vec()),
|
||||
"Resources" => dictionary! {
|
||||
"XObject" => dictionary! {
|
||||
"Im0" => Object::Reference(image_id),
|
||||
},
|
||||
},
|
||||
},
|
||||
b"1500 0 0 2383 0 0 cm /Im0 Do".to_vec(),
|
||||
)));
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => Object::Name(b"Type1".to_vec()),
|
||||
"BaseFont" => Object::Name(b"Helvetica".to_vec()),
|
||||
});
|
||||
|
||||
let mut content = b"q /Fm0 Do Q BT /F1 12 Tf ".to_vec();
|
||||
for line in masthead_lines {
|
||||
content.extend_from_slice(format!("({line}) Tj ").as_bytes());
|
||||
}
|
||||
content.extend_from_slice(b"ET");
|
||||
let content_id =
|
||||
doc.add_object(Object::Stream(lopdf::Stream::new(dictionary! {}, content)));
|
||||
|
||||
doc.objects.insert(
|
||||
page_id,
|
||||
Object::Dictionary(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Parent" => Object::Reference(pages_id),
|
||||
"MediaBox" => vec![0.into(), 0.into(), 1500.into(), 2383.into()],
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
"XObject" => dictionary! { "Fm0" => Object::Reference(form_id) },
|
||||
},
|
||||
"Contents" => Object::Reference(content_id),
|
||||
}),
|
||||
);
|
||||
doc.objects.insert(
|
||||
pages_id,
|
||||
Object::Dictionary(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
"Count" => Object::Integer(1),
|
||||
}),
|
||||
);
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_masthead_over_form_wrapped_scan_needs_ocr() {
|
||||
// A full-page scan wrapped in a Form XObject with ~4 ops of real,
|
||||
// diverse masthead text. `alphanum_low` can't flag it (the chrome is
|
||||
// genuine text), so the sparse-text floor must: without OCR the page
|
||||
// body is silently lost while classification calls the page scanned.
|
||||
let (doc, page_id) = masthead_scan_page(&[
|
||||
"18",
|
||||
"FINANCIAL EXPRESS",
|
||||
"WWW.FINANCIALEXPRESS.COM",
|
||||
"FRIDAY, DECEMBER 13, 2024",
|
||||
]);
|
||||
let analysis = analyze_page_content(&doc, page_id);
|
||||
assert!(
|
||||
analysis.has_template_image,
|
||||
"sanity: full-page image inside the form must be found"
|
||||
);
|
||||
assert!(
|
||||
analysis.unique_alphanum_chars >= 10,
|
||||
"sanity: masthead text is diverse, alphanum_low cannot fire"
|
||||
);
|
||||
let (needs_ocr, _) = page_ocr_signals(&doc, page_id);
|
||||
assert!(
|
||||
needs_ocr,
|
||||
"template image + text below the pages_with_text floor is a scan"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_text_page_over_background_image_stays_native() {
|
||||
// Counterpart: a real text page over a full-page background image
|
||||
// (letterhead/watermark) has enough text ops to clear the
|
||||
// `pages_with_text` floor and must NOT be routed to OCR.
|
||||
let lines: Vec<String> = (0..12)
|
||||
.map(|i| format!("Paragraph line {i} with ordinary body text"))
|
||||
.collect();
|
||||
let refs: Vec<&str> = lines.iter().map(String::as_str).collect();
|
||||
let (doc, page_id) = masthead_scan_page(&refs);
|
||||
let analysis = analyze_page_content(&doc, page_id);
|
||||
assert!(
|
||||
analysis.has_template_image,
|
||||
"sanity: background image found"
|
||||
);
|
||||
assert!(
|
||||
analysis.text_operator_count >= 10,
|
||||
"sanity: body text clears the floor"
|
||||
);
|
||||
let (needs_ocr, _) = page_ocr_signals(&doc, page_id);
|
||||
assert!(
|
||||
!needs_ocr,
|
||||
"a text page with a background image must stay native"
|
||||
);
|
||||
}
|
||||
|
||||
// ---------- P2 tests: Form XObject font traversal ----------
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -561,11 +561,7 @@ pub(crate) fn extract_page_text_items(
|
||||
y,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: crate::extractor::fonts::item_font_name(
|
||||
¤t_font,
|
||||
base_font,
|
||||
)
|
||||
.to_string(),
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
@@ -749,11 +745,7 @@ pub(crate) fn extract_page_text_items(
|
||||
y,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: crate::extractor::fonts::item_font_name(
|
||||
¤t_font,
|
||||
base_font,
|
||||
)
|
||||
.to_string(),
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
@@ -860,11 +852,7 @@ pub(crate) fn extract_page_text_items(
|
||||
y,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: crate::extractor::fonts::item_font_name(
|
||||
¤t_font,
|
||||
base_font,
|
||||
)
|
||||
.to_string(),
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
@@ -1017,11 +1005,7 @@ pub(crate) fn extract_page_text_items(
|
||||
y,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: crate::extractor::fonts::item_font_name(
|
||||
¤t_font,
|
||||
base_font,
|
||||
)
|
||||
.to_string(),
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
|
||||
@@ -225,26 +225,6 @@ pub(crate) fn build_type3_scales(
|
||||
scales
|
||||
}
|
||||
|
||||
/// The name a `TextItem` carries for its font: the `/BaseFont` family name
|
||||
/// ("ABCDEF+CMMI10"), which identifies the actual face, rather than the
|
||||
/// arbitrary per-page resource tag ("F2").
|
||||
///
|
||||
/// Exception: resource names using Distiller's CID convention (`C2_0`,
|
||||
/// `C0_1`) are kept as-is — `text_utils::is_cid_font` keys on that prefix
|
||||
/// for micro-gap joining, and the family name carries no CID marker to
|
||||
/// replace it. This is a known, deliberate wart: `TextItem::font` is the
|
||||
/// face name except for this one producer convention. The clean fix is an
|
||||
/// explicit CID flag on `TextItem`, which touches its ~29 construction
|
||||
/// sites; do that migration when `TextItem` next changes shape, and delete
|
||||
/// this carve-out with it.
|
||||
pub(crate) fn item_font_name<'a>(resource_name: &'a str, base_font: &'a str) -> &'a str {
|
||||
if crate::text_utils::is_cid_font(resource_name) {
|
||||
resource_name
|
||||
} else {
|
||||
base_font
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse font widths from a font dictionary, dispatching by Subtype
|
||||
pub(crate) fn parse_font_widths(
|
||||
doc: &Document,
|
||||
@@ -1684,17 +1664,6 @@ fn score_text(text: &str) -> i32 {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn item_font_name_prefers_family_over_resource_tag() {
|
||||
use super::item_font_name;
|
||||
assert_eq!(item_font_name("F2", "ABCDEF+CMMI10"), "ABCDEF+CMMI10");
|
||||
assert_eq!(item_font_name("T22", "Times-Roman"), "Times-Roman");
|
||||
// Distiller CID-convention resources keep the resource name:
|
||||
// is_cid_font keys on the C2_/C0_ prefix for micro-gap joining.
|
||||
assert_eq!(item_font_name("C2_0", "ABCDEE+SimSun"), "C2_0");
|
||||
assert_eq!(item_font_name("C0_1", "ABCDEE+MSMincho"), "C0_1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_resolves_indirect_matrix_and_bbox_numbers() {
|
||||
use lopdf::{dictionary, Document, Object};
|
||||
|
||||
@@ -620,11 +620,7 @@ fn extract_form_xobject_text_inner(
|
||||
y,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: crate::extractor::fonts::item_font_name(
|
||||
¤t_font,
|
||||
base_font,
|
||||
)
|
||||
.to_string(),
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
@@ -779,11 +775,7 @@ fn extract_form_xobject_text_inner(
|
||||
y,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: crate::extractor::fonts::item_font_name(
|
||||
¤t_font,
|
||||
base_font,
|
||||
)
|
||||
.to_string(),
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
|
||||
@@ -203,42 +203,9 @@ pub(crate) fn is_code_like(text: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// True when a line's text is essentially all monospace (≥90% by character
|
||||
/// count). Code lines are wholly monospace; anything less is prose carrying
|
||||
/// mono-styled fragments — a URL sidebar, or a sentence quoting an inline
|
||||
/// code literal — and fencing it would split paragraphs mid-sentence.
|
||||
/// Any-item matching was safe only while items carried opaque font resource
|
||||
/// names that never matched the monospace patterns; items now carry real
|
||||
/// family names.
|
||||
pub(crate) fn line_is_monospace(line: &crate::types::TextLine) -> bool {
|
||||
let mut monospace_chars = 0usize;
|
||||
let mut total_chars = 0usize;
|
||||
for item in &line.items {
|
||||
let text = item.text.trim();
|
||||
let chars = text.chars().count();
|
||||
total_chars += chars;
|
||||
// Hyperlinks and underlined text set in a mono face are link
|
||||
// styling, not code — a URL sidebar must not fence lyric lines.
|
||||
let looks_like_link = item.is_underline
|
||||
|| matches!(item.item_type, crate::types::ItemType::Link(_))
|
||||
|| text.contains("://")
|
||||
|| text.starts_with("www.");
|
||||
if is_monospace_font(&item.font) && !looks_like_link {
|
||||
monospace_chars += chars;
|
||||
}
|
||||
}
|
||||
total_chars > 0 && monospace_chars * 10 >= total_chars * 9
|
||||
}
|
||||
|
||||
/// Check if font name indicates monospace
|
||||
pub(crate) fn is_monospace_font(font_name: &str) -> bool {
|
||||
let lower = font_name.to_lowercase();
|
||||
// "Monotype" is a foundry prefix on proportional faces (Monotype
|
||||
// Corsiva, Monotype Garamond) — it must not satisfy the generic "mono"
|
||||
// token below.
|
||||
if lower.contains("monotype") {
|
||||
return false;
|
||||
}
|
||||
let patterns = [
|
||||
"courier",
|
||||
"consolas",
|
||||
@@ -263,17 +230,6 @@ pub(crate) fn is_monospace_font(font_name: &str) -> bool {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn monotype_foundry_faces_are_not_monospace() {
|
||||
// "Monotype" is a foundry prefix on proportional faces; the generic
|
||||
// "mono" token must not classify them as code fonts.
|
||||
assert!(!is_monospace_font("MonotypeCorsiva"));
|
||||
assert!(!is_monospace_font("ABCDEF+Monotype-Garamond"));
|
||||
assert!(is_monospace_font("RobotoMono-Regular"));
|
||||
assert!(is_monospace_font("PTMono"));
|
||||
assert!(is_monospace_font("Courier"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_plain_bullet() {
|
||||
assert_eq!(format_list_item("● Item"), "- Item");
|
||||
|
||||
+28
-52
@@ -11,7 +11,9 @@ use super::analysis::{
|
||||
detect_header_level, font_size_rarity, has_dot_leaders, is_heading_fragment, is_toc_entry_line,
|
||||
is_toc_marker_heading,
|
||||
};
|
||||
use super::classify::{format_list_item, is_caption_line, is_list_item, starts_with_bullet_marker};
|
||||
use super::classify::{
|
||||
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
||||
};
|
||||
use super::heading::classify_heading_sequences;
|
||||
use super::postprocess::clean_markdown;
|
||||
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
||||
@@ -769,27 +771,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
let mut in_list = false;
|
||||
let mut in_paragraph = false;
|
||||
let mut last_list_x: Option<f32> = None;
|
||||
// Code lines accumulate here and the fence is emitted only when the
|
||||
// block flushes with content — an empty ``` ``` pair can never appear.
|
||||
fn flush_code_block(output: &mut String, pending_code: &mut String) {
|
||||
let trimmed = pending_code.trim();
|
||||
// A fragment too short to be code — a lone ® or stray glyph set in
|
||||
// a mono face — reads better as plain text than as a fenced block.
|
||||
if trimmed.chars().count() < 3 {
|
||||
if !trimmed.is_empty() {
|
||||
output.push_str(trimmed);
|
||||
output.push_str("\n\n");
|
||||
}
|
||||
} else {
|
||||
output.push_str("```\n");
|
||||
output.push_str(pending_code);
|
||||
output.push_str("```\n");
|
||||
}
|
||||
pending_code.clear();
|
||||
}
|
||||
|
||||
let mut in_code_block = false;
|
||||
let mut pending_code = String::new();
|
||||
let mut prev_had_dot_leaders = false;
|
||||
let mut paragraph_in_wrapped_bold_run = false;
|
||||
let mut toc_suppress_page: Option<u32> = None;
|
||||
@@ -823,7 +805,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// Flush current page's remaining tables and images
|
||||
if current_page > 0 {
|
||||
if in_code_block {
|
||||
flush_code_block(&mut output, &mut pending_code);
|
||||
output.push_str("```\n");
|
||||
in_code_block = false;
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
@@ -885,14 +867,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
PositionedBlockKind::Image => inserted_images.contains(&(current_page, idx)),
|
||||
};
|
||||
if positioned_block_precedes_line(block, line) && !already_inserted {
|
||||
// Code lines buffer until their block closes; flush them
|
||||
// first so this block cannot jump ahead of code that
|
||||
// precedes it in reading order. A code line after the
|
||||
// block reopens a new fence naturally.
|
||||
if in_code_block {
|
||||
flush_code_block(&mut output, &mut pending_code);
|
||||
in_code_block = false;
|
||||
}
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
@@ -963,22 +937,15 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// These should be on their own line followed by a paragraph break
|
||||
let struct_role = struct_roles.and_then(|roles| resolve_line_struct_role(line, roles));
|
||||
|
||||
// Determine if this line is code (struct-tree or font-based) for
|
||||
// block accumulation. Font-based detection only opens a block at a
|
||||
// paragraph boundary: a mono-set line that continues an open prose
|
||||
// paragraph is the producer smearing an inline code literal's style
|
||||
// across a wrapped line (HTML-to-PDF exports do this), and fencing
|
||||
// it would cut the sentence in three.
|
||||
// Determine if this line is code (struct-tree or font-based) for block accumulation
|
||||
let is_code_line = struct_role
|
||||
.as_ref()
|
||||
.is_some_and(|r| matches!(r, StructRole::Code))
|
||||
|| (options.detect_code
|
||||
&& (in_code_block || !in_paragraph)
|
||||
&& super::classify::line_is_monospace(line));
|
||||
|| (options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)));
|
||||
|
||||
// Close code block when transitioning to non-code
|
||||
if in_code_block && !is_code_line {
|
||||
flush_code_block(&mut output, &mut pending_code);
|
||||
output.push_str("```\n");
|
||||
in_code_block = false;
|
||||
}
|
||||
|
||||
@@ -1212,9 +1179,12 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
in_code_block = true;
|
||||
pending_code.push_str(plain_trimmed);
|
||||
pending_code.push('\n');
|
||||
if !in_code_block {
|
||||
output.push_str("```\n");
|
||||
in_code_block = true;
|
||||
}
|
||||
output.push_str(plain_trimmed);
|
||||
output.push('\n');
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -1239,7 +1209,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
|
||||
// Close any trailing code block
|
||||
if in_code_block {
|
||||
flush_code_block(&mut output, &mut pending_code);
|
||||
output.push_str("```\n");
|
||||
}
|
||||
|
||||
// Flush current page and any remaining pages with tables/images
|
||||
@@ -1400,7 +1370,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
&& !is_toc_entry_line(plain_trimmed)
|
||||
&& !is_heading_fragment(plain_trimmed)
|
||||
&& toc_suppress_page != Some(line.page)
|
||||
&& !(options.detect_code && super::classify::line_is_monospace(line))
|
||||
&& !(options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)))
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
if let Some(header_level) = detect_header_level(
|
||||
@@ -1501,13 +1471,19 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
}
|
||||
}
|
||||
|
||||
// Detect code blocks by font. Only at a paragraph boundary — a
|
||||
// mono-set line continuing an open prose paragraph is an inline
|
||||
// code literal's style smeared across a wrapped line, not code.
|
||||
if options.detect_code && !in_paragraph && super::classify::line_is_monospace(line) {
|
||||
// Use plain text for code blocks
|
||||
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
|
||||
continue;
|
||||
// Detect code blocks by font
|
||||
if options.detect_code {
|
||||
let is_mono = line.items.iter().any(|i| is_monospace_font(&i.font));
|
||||
if is_mono {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
// Use plain text for code blocks
|
||||
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Regular text - join lines within same paragraph with space
|
||||
|
||||
@@ -237,13 +237,6 @@ pub trait OcrEngine: Send + Sync {
|
||||
pages: &[RenderedPage],
|
||||
options: &OcrOptions,
|
||||
) -> Result<Vec<OcrPage>, Self::Error>;
|
||||
|
||||
/// Number of pages this engine can process concurrently in one
|
||||
/// `recognize` call. The pipeline sizes its page batches from this so a
|
||||
/// parallel engine is not starved by small chunks; `1` means sequential.
|
||||
fn preferred_page_concurrency(&self) -> usize {
|
||||
1
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
+36
-429
@@ -1,14 +1,11 @@
|
||||
//! PP-OCRv6 Small implementation backed by OAR and ONNX Runtime.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::sync::Arc;
|
||||
use std::time::Instant;
|
||||
|
||||
use image::RgbImage;
|
||||
use oar_ocr::core::config::onnx::OrtSessionConfig;
|
||||
use oar_ocr::domain::tasks::TextDetectionConfig;
|
||||
use oar_ocr::oarocr::{EdgeProcessor, TextCroppingProcessor};
|
||||
use oar_ocr::predictors::{TextDetectionPredictor, TextRecognitionPredictor};
|
||||
use oar_ocr::oarocr::{OAROCRBuilder, OAROCR};
|
||||
use oar_ocr::processors::BoundingBox;
|
||||
use thiserror::Error;
|
||||
|
||||
@@ -73,170 +70,17 @@ pub enum OarOcrError {
|
||||
Backend(#[from] oar_ocr::core::OCRError),
|
||||
}
|
||||
|
||||
/// Standard detection input cap. PP-OCR detection resizes each page so its
|
||||
/// longest side fits this before inference; it is the PaddleOCR default and
|
||||
/// is sufficient for ordinary body text at 150 DPI.
|
||||
const DETECTION_LIMIT_STANDARD: u32 = 960;
|
||||
|
||||
/// Escalated detection input cap for dense fine-print pages. Beyond this the
|
||||
/// measured recall plateaus while inference cost keeps growing.
|
||||
const DETECTION_LIMIT_ESCALATED: u32 = 2560;
|
||||
|
||||
/// Hard ceiling protecting detection from out-of-memory on giant renders.
|
||||
const DETECTION_MAXIMUM_SIDE: u32 = 4000;
|
||||
|
||||
/// Escalate only for pages dense with small text: at least this many detected
|
||||
/// regions in the standard pass...
|
||||
const ESCALATION_MINIMUM_REGIONS: usize = 80;
|
||||
|
||||
/// ...whose median height, at detection scale, is below this. Calibrated at
|
||||
/// `unclip_ratio` 2.0 (the expansion inflates measured heights, so this
|
||||
/// constant is coupled to [`detection_config`]): dense fine-print pages that
|
||||
/// gain from escalation measure 12.0–14.2 px with 144+ regions; the nearest
|
||||
/// non-gaining page above the region gate (an engineering drawing) measures
|
||||
/// 15.7 px, and prose/typewriter pages measure 14.5 px+ with too few
|
||||
/// regions to qualify at all.
|
||||
const ESCALATION_MAXIMUM_MEDIAN_HEIGHT: f32 = 15.0;
|
||||
|
||||
/// One worker's model sessions: a standard-limit detector plus a recognizer,
|
||||
/// and that worker's own lazily built escalated-limit detector.
|
||||
/// Staged (detect, crop, recognize as separate calls) rather than OAROCR's
|
||||
/// combined `predict` so an escalated page replaces only its detection pass —
|
||||
/// recognition runs exactly once, on the final region set.
|
||||
struct OcrWorker {
|
||||
detector: TextDetectionPredictor,
|
||||
recognizer: TextRecognitionPredictor,
|
||||
/// Built on this worker's first dense fine-print page. `None` inside the
|
||||
/// cell records a failed build so it is not retried per page.
|
||||
escalated: std::sync::OnceLock<Option<TextDetectionPredictor>>,
|
||||
}
|
||||
|
||||
/// CPU PP-OCRv6 Small engine using OAR's detection and recognition components.
|
||||
/// CPU PP-OCRv6 Small engine using OAR's detection and recognition pipeline.
|
||||
///
|
||||
/// Construction accepts only [`ModelPaths`] that have already passed
|
||||
/// pdf-inspector's manifest size and SHA-256 verification. OAR's independent
|
||||
/// model auto-download feature is deliberately not enabled.
|
||||
#[derive(Debug)]
|
||||
pub struct OarOcrEngine {
|
||||
workers: Vec<OcrWorker>,
|
||||
detection_path: PathBuf,
|
||||
intra_threads: usize,
|
||||
/// Present only when more than one worker exists; sized to match.
|
||||
pool: Option<rayon::ThreadPool>,
|
||||
pipeline: OAROCR,
|
||||
model: ModelIdentity,
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for OarOcrEngine {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("OarOcrEngine")
|
||||
.field("workers", &self.workers.len())
|
||||
.field("parallel", &self.pool.is_some())
|
||||
.field("model", &self.model)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
/// Pages processed concurrently: one OAROCR pipeline (and its ONNX sessions)
|
||||
/// per worker, because oar-ocr serializes each session behind a mutex.
|
||||
/// Measured on CPU: workers beyond 3 stop scaling (memory-bandwidth bound)
|
||||
/// and each worker is fastest with 2 intra-op threads.
|
||||
fn pipeline_concurrency() -> usize {
|
||||
let cores = std::thread::available_parallelism()
|
||||
.map(std::num::NonZeroUsize::get)
|
||||
.unwrap_or(1);
|
||||
(cores / 4).clamp(1, 3)
|
||||
}
|
||||
|
||||
fn intra_threads_per_pipeline(concurrency: usize) -> usize {
|
||||
let cores = std::thread::available_parallelism()
|
||||
.map(std::num::NonZeroUsize::get)
|
||||
.unwrap_or(1);
|
||||
if concurrency > 1 {
|
||||
2
|
||||
} else {
|
||||
cores.min(4)
|
||||
}
|
||||
}
|
||||
|
||||
/// True when a standard-limit detection pass over a downscaled page shows
|
||||
/// dense, small text: the page deserves a second pass at the escalated limit.
|
||||
fn should_escalate_detection(
|
||||
median_detection_height: f32,
|
||||
region_count: usize,
|
||||
downscale: f32,
|
||||
) -> bool {
|
||||
downscale < 1.0
|
||||
&& region_count >= ESCALATION_MINIMUM_REGIONS
|
||||
&& median_detection_height < ESCALATION_MAXIMUM_MEDIAN_HEIGHT
|
||||
}
|
||||
|
||||
/// Median detected-region height in detection-input pixels: original-image
|
||||
/// heights multiplied by the downscale detection applied.
|
||||
fn median_detection_height(heights: &mut [f32], downscale: f32) -> f32 {
|
||||
if heights.is_empty() {
|
||||
return f32::MAX;
|
||||
}
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
let middle = heights.len() / 2;
|
||||
let median = if heights.len().is_multiple_of(2) {
|
||||
(heights[middle - 1] + heights[middle]) / 2.0
|
||||
} else {
|
||||
heights[middle]
|
||||
};
|
||||
median * downscale
|
||||
}
|
||||
|
||||
/// Detection preprocessing config at a given input cap.
|
||||
///
|
||||
/// Supplying an explicit config suppresses OAROCR's "general" text-type
|
||||
/// overrides, so every field the override would have set must be pinned
|
||||
/// here to match what the combined pipeline ran with before the staged
|
||||
/// split: score 0.3 and box 0.6 (equal to [`TextDetectionConfig`]'s
|
||||
/// defaults) and unclip 2.0 (the default is 1.5 — leaving it would
|
||||
/// silently shrink detection-box expansion and risk clipping edge glyphs).
|
||||
fn detection_config(detection_limit: u32) -> TextDetectionConfig {
|
||||
TextDetectionConfig {
|
||||
limit_side_len: Some(detection_limit),
|
||||
limit_type: Some(oar_ocr::processors::LimitType::Max),
|
||||
max_side_len: Some(DETECTION_MAXIMUM_SIDE),
|
||||
unclip_ratio: 2.0,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn build_detector(
|
||||
detection: &std::path::Path,
|
||||
detection_limit: u32,
|
||||
intra_threads: usize,
|
||||
) -> Result<TextDetectionPredictor, OarOcrError> {
|
||||
Ok(TextDetectionPredictor::builder()
|
||||
.with_config(detection_config(detection_limit))
|
||||
.with_ort_config(ocr_session_config(intra_threads))
|
||||
.build(detection)?)
|
||||
}
|
||||
|
||||
fn build_workers(
|
||||
detection: &std::path::Path,
|
||||
recognition: &std::path::Path,
|
||||
dictionary: &std::path::Path,
|
||||
count: usize,
|
||||
intra_threads: usize,
|
||||
) -> Result<Vec<OcrWorker>, OarOcrError> {
|
||||
let mut workers = Vec::with_capacity(count);
|
||||
for _ in 0..count {
|
||||
let detector = build_detector(detection, DETECTION_LIMIT_STANDARD, intra_threads)?;
|
||||
let recognizer = TextRecognitionPredictor::builder()
|
||||
.dict_path(dictionary)
|
||||
.with_ort_config(ocr_session_config(intra_threads))
|
||||
.build(recognition)?;
|
||||
workers.push(OcrWorker {
|
||||
detector,
|
||||
recognizer,
|
||||
escalated: std::sync::OnceLock::new(),
|
||||
});
|
||||
}
|
||||
Ok(workers)
|
||||
}
|
||||
|
||||
impl OarOcrEngine {
|
||||
/// Loads PP-OCRv6 Small from a resolved, verified model set.
|
||||
pub fn from_models(models: &ModelPaths) -> Result<Self, OarOcrError> {
|
||||
@@ -245,169 +89,36 @@ impl OarOcrEngine {
|
||||
let recognition = required_model(models, ModelArtifactKind::TextRecognition)?;
|
||||
let dictionary = required_model(models, ModelArtifactKind::CharacterDictionary)?;
|
||||
|
||||
let concurrency = pipeline_concurrency();
|
||||
let intra_threads = intra_threads_per_pipeline(concurrency);
|
||||
let workers = build_workers(
|
||||
detection,
|
||||
recognition,
|
||||
dictionary,
|
||||
concurrency,
|
||||
intra_threads,
|
||||
)?;
|
||||
let pool = if concurrency > 1 {
|
||||
rayon::ThreadPoolBuilder::new()
|
||||
.num_threads(concurrency)
|
||||
.build()
|
||||
.ok()
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let pipeline = OAROCRBuilder::new(detection, recognition, dictionary)
|
||||
.ort_session(ocr_session_config())
|
||||
// Document line crops often have very different widths. Keeping
|
||||
// CPU recognition batches at one avoids padding every crop to the
|
||||
// widest line, reducing both inference work and peak memory.
|
||||
.region_batch_size(1)
|
||||
.build()?;
|
||||
let model = ModelIdentity::new(models.manifest_id(), models.revision());
|
||||
Ok(Self {
|
||||
workers,
|
||||
detection_path: detection.to_path_buf(),
|
||||
intra_threads,
|
||||
pool,
|
||||
model,
|
||||
})
|
||||
}
|
||||
|
||||
/// This worker's escalated-limit detector, built on first use.
|
||||
fn escalated_detector<'w>(&self, worker: &'w OcrWorker) -> Option<&'w TextDetectionPredictor> {
|
||||
worker
|
||||
.escalated
|
||||
.get_or_init(|| {
|
||||
match build_detector(
|
||||
&self.detection_path,
|
||||
DETECTION_LIMIT_ESCALATED,
|
||||
self.intra_threads,
|
||||
) {
|
||||
Ok(detector) => Some(detector),
|
||||
Err(error) => {
|
||||
log::warn!(
|
||||
"escalated OCR detection unavailable, keeping standard pass: {error}"
|
||||
);
|
||||
None
|
||||
}
|
||||
}
|
||||
})
|
||||
.as_ref()
|
||||
}
|
||||
|
||||
/// Detects text regions for one page: a standard-limit pass first, then —
|
||||
/// for pages the standard limit demonstrably under-resolves — a second
|
||||
/// pass at the escalated limit whose boxes replace the first. Pages whose
|
||||
/// render dwarfs even the escalated limit skip the standard pass outright.
|
||||
fn detect_boxes(
|
||||
&self,
|
||||
page: &RenderedPage,
|
||||
image: &Arc<RgbImage>,
|
||||
worker: &OcrWorker,
|
||||
) -> Result<Vec<BoundingBox>, OarOcrError> {
|
||||
let longest_side = page.width().max(page.height()) as f32;
|
||||
|
||||
// A page more than twice the standard limit loses over half its
|
||||
// resolution before detection even runs; go straight to the escalated
|
||||
// detector instead of paying a doomed standard pass.
|
||||
if longest_side > (DETECTION_LIMIT_STANDARD * 2) as f32 {
|
||||
if let Some(escalated) = self.escalated_detector(worker) {
|
||||
log::debug!(
|
||||
"page {}: direct escalated detection (render {longest_side}px)",
|
||||
page.page(),
|
||||
);
|
||||
match detect_with(escalated, image, page.page()) {
|
||||
Ok(boxes) => return Ok(boxes),
|
||||
Err(error) => {
|
||||
// Same degradation as the adaptive branch below: a
|
||||
// failing escalated pass falls back to standard
|
||||
// detection instead of failing the page outright.
|
||||
// Return the standard boxes directly — the adaptive
|
||||
// trigger would only re-invoke the detector that
|
||||
// just failed (repeating an OOM on a dense page).
|
||||
log::warn!(
|
||||
"page {}: direct escalated detection failed, using standard pass: {error}",
|
||||
page.page()
|
||||
);
|
||||
return detect_with(&worker.detector, image, page.page());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let detections = detect_with(&worker.detector, image, page.page())?;
|
||||
|
||||
// Dense fine-print pages (broadsheets, pricing sheets) lose most of
|
||||
// their text when detection downscales them to the standard limit.
|
||||
// When the standard pass shows many regions of tiny detection-scale
|
||||
// height, rerun detection at the escalated limit.
|
||||
let downscale = (DETECTION_LIMIT_STANDARD as f32 / longest_side).min(1.0);
|
||||
let mut heights: Vec<f32> = detections.iter().map(polygon_height).collect();
|
||||
let median = median_detection_height(&mut heights, downscale);
|
||||
log::trace!(
|
||||
"page {}: standard pass {} regions, median height {:.1}px at detection scale",
|
||||
page.page(),
|
||||
detections.len(),
|
||||
median
|
||||
);
|
||||
if should_escalate_detection(median, detections.len(), downscale) {
|
||||
log::debug!(
|
||||
"page {}: escalating detection ({} regions, median height {:.1}px at detection scale)",
|
||||
page.page(),
|
||||
detections.len(),
|
||||
median
|
||||
);
|
||||
if let Some(escalated) = self.escalated_detector(worker) {
|
||||
match detect_with(escalated, image, page.page()) {
|
||||
Ok(escalated_boxes) => return Ok(escalated_boxes),
|
||||
Err(error) => {
|
||||
log::warn!(
|
||||
"page {}: escalated detection failed, keeping standard pass: {error}",
|
||||
page.page()
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(detections)
|
||||
Ok(Self { pipeline, model })
|
||||
}
|
||||
|
||||
fn recognize_page(
|
||||
&self,
|
||||
page: &RenderedPage,
|
||||
options: &OcrOptions,
|
||||
worker: usize,
|
||||
) -> Result<OcrPage, OarOcrError> {
|
||||
let started = Instant::now();
|
||||
let worker = &self.workers[worker % self.workers.len()];
|
||||
let image = Arc::new(rendered_page_to_rgb(page)?);
|
||||
let boxes = self.detect_boxes(page, &image, worker)?;
|
||||
// Reading order, matching what the combined pipeline produced.
|
||||
let boxes = oar_ocr::processors::sort_quad_boxes(&boxes);
|
||||
let image = rendered_page_to_rgb(page)?;
|
||||
let result = self
|
||||
.pipeline
|
||||
.predict(vec![image])?
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or(OarOcrError::MissingPageResult { page: page.page() })?;
|
||||
|
||||
// Same rotation-aware cropping the combined pipeline uses.
|
||||
let crops =
|
||||
TextCroppingProcessor::new(true).process((Arc::clone(&image), boxes.clone()))?;
|
||||
drop(image);
|
||||
|
||||
let recognizer = &worker.recognizer;
|
||||
let mut spans = Vec::with_capacity(boxes.len());
|
||||
let mut spans = Vec::with_capacity(result.text_regions.len());
|
||||
let mut invalid_geometry = 0usize;
|
||||
let mut missing_recognition = 0usize;
|
||||
for (bounding_box, crop) in boxes.iter().zip(crops) {
|
||||
let Some(crop) = crop else {
|
||||
invalid_geometry += 1;
|
||||
continue;
|
||||
};
|
||||
// One crop per call: document line crops often have very
|
||||
// different widths, and batching pads every crop to the widest
|
||||
// line. Measured on CPU, batched recognition (even width-sorted)
|
||||
// is 2–3× slower than per-crop calls.
|
||||
let crop = Arc::try_unwrap(crop).unwrap_or_else(|shared| (*shared).clone());
|
||||
let recognized = recognizer.predict(vec![crop])?;
|
||||
let (Some(text), Some(confidence)) = (
|
||||
recognized.texts.into_iter().next(),
|
||||
recognized.scores.into_iter().next(),
|
||||
) else {
|
||||
for region in result.text_regions {
|
||||
let (Some(text), Some(confidence)) = (region.text, region.confidence) else {
|
||||
missing_recognition += 1;
|
||||
continue;
|
||||
};
|
||||
@@ -420,21 +131,16 @@ impl OarOcrEngine {
|
||||
continue;
|
||||
}
|
||||
|
||||
let Some(polygon) = bounding_box_to_quad(bounding_box, page.width(), page.height())
|
||||
else {
|
||||
let polygon = region.dt_poly.as_ref().unwrap_or(®ion.bounding_box);
|
||||
let Some(polygon) = bounding_box_to_quad(polygon, page.width(), page.height()) else {
|
||||
invalid_geometry += 1;
|
||||
continue;
|
||||
};
|
||||
spans.push(OcrSpan {
|
||||
text,
|
||||
text: text.to_string(),
|
||||
polygon,
|
||||
confidence,
|
||||
// The combined pipeline's orientation_angle came from the
|
||||
// text-line-orientation classifier, a model this engine has
|
||||
// never loaded — it was structurally None before the staged
|
||||
// split too (the staged/combined A/B was byte-identical).
|
||||
// Region rotation is still carried by the polygon itself.
|
||||
orientation_degrees: None,
|
||||
orientation_degrees: region.orientation_angle,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -468,42 +174,12 @@ impl OarOcrEngine {
|
||||
}
|
||||
}
|
||||
|
||||
/// Runs one detector over one page image and returns its region polygons.
|
||||
fn detect_with(
|
||||
detector: &TextDetectionPredictor,
|
||||
image: &Arc<RgbImage>,
|
||||
page_number: u32,
|
||||
) -> Result<Vec<BoundingBox>, OarOcrError> {
|
||||
let mut result = detector.predict(vec![(**image).clone()])?;
|
||||
if result.detections.is_empty() {
|
||||
return Err(OarOcrError::MissingPageResult { page: page_number });
|
||||
}
|
||||
Ok(result
|
||||
.detections
|
||||
.swap_remove(0)
|
||||
.into_iter()
|
||||
.map(|detection| detection.bbox)
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// Vertical extent of a detection polygon in original-image pixels.
|
||||
fn polygon_height(polygon: &BoundingBox) -> f32 {
|
||||
let mut min_y = f32::MAX;
|
||||
let mut max_y = f32::MIN;
|
||||
for point in &polygon.points {
|
||||
min_y = min_y.min(point.y);
|
||||
max_y = max_y.max(point.y);
|
||||
}
|
||||
if max_y > min_y {
|
||||
max_y - min_y
|
||||
} else {
|
||||
0.0
|
||||
}
|
||||
}
|
||||
|
||||
fn ocr_session_config(intra_threads: usize) -> OrtSessionConfig {
|
||||
fn ocr_session_config() -> OrtSessionConfig {
|
||||
let available = std::thread::available_parallelism()
|
||||
.map(std::num::NonZeroUsize::get)
|
||||
.unwrap_or(1);
|
||||
OrtSessionConfig::new()
|
||||
.with_intra_threads(intra_threads.max(1))
|
||||
.with_intra_threads(available.min(4))
|
||||
.with_inter_threads(1)
|
||||
.with_parallel_execution(false)
|
||||
}
|
||||
@@ -550,33 +226,10 @@ impl OcrEngine for OarOcrEngine {
|
||||
) -> Result<Vec<OcrPage>, Self::Error> {
|
||||
validate_options(options)?;
|
||||
|
||||
let Some(pool) = self.pool.as_ref().filter(|_| pages.len() > 1) else {
|
||||
return pages
|
||||
.iter()
|
||||
.map(|page| self.recognize_page(page, options, 0))
|
||||
.collect();
|
||||
};
|
||||
pool.install(|| {
|
||||
use rayon::prelude::*;
|
||||
pages
|
||||
.par_iter()
|
||||
.map(|page| {
|
||||
let worker = rayon::current_thread_index().unwrap_or(0);
|
||||
self.recognize_page(page, options, worker)
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
}
|
||||
|
||||
fn preferred_page_concurrency(&self) -> usize {
|
||||
// Without a pool, recognition runs sequentially regardless of worker
|
||||
// count — report that honestly so the pipeline doesn't render
|
||||
// oversized page batches for parallelism that isn't there.
|
||||
if self.pool.is_some() {
|
||||
self.workers.len()
|
||||
} else {
|
||||
1
|
||||
}
|
||||
pages
|
||||
.iter()
|
||||
.map(|page| self.recognize_page(page, options))
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -723,13 +376,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn cpu_session_budget_is_bounded_for_small_ocr_models() {
|
||||
let concurrency = pipeline_concurrency();
|
||||
let config = ocr_session_config(intra_threads_per_pipeline(concurrency));
|
||||
let config = ocr_session_config();
|
||||
assert!((1..=4).contains(&config.intra_threads.unwrap()));
|
||||
assert_eq!(config.inter_threads, Some(1));
|
||||
assert_eq!(config.parallel_execution, Some(false));
|
||||
// Zero requests are clamped so a session always has a thread.
|
||||
assert_eq!(ocr_session_config(0).intra_threads, Some(1));
|
||||
}
|
||||
|
||||
fn page(format: RenderPixelFormat, stride: usize, pixels: Vec<u8>) -> RenderedPage {
|
||||
@@ -843,47 +493,4 @@ mod tests {
|
||||
)
|
||||
.is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn escalation_fires_for_dense_fine_print_pages() {
|
||||
// Measured cases (at unclip 2.0) that gain from escalation: dense
|
||||
// tiled ad pages (12.3–14.2px, 158–286 regions) and a dense pricing
|
||||
// sheet (12.0px, 144 regions), all downscaled by the standard limit.
|
||||
assert!(should_escalate_detection(14.2, 186, 0.55));
|
||||
assert!(should_escalate_detection(13.1, 286, 0.55));
|
||||
assert!(should_escalate_detection(12.3, 158, 0.55));
|
||||
assert!(should_escalate_detection(12.0, 144, 0.55));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn escalation_skips_ordinary_pages() {
|
||||
// Academic prose: too few regions (and tall enough at unclip 2.0).
|
||||
assert!(!should_escalate_detection(14.5, 47, 0.58));
|
||||
// Engineering drawing: many regions but tall enough text.
|
||||
assert!(!should_escalate_detection(15.7, 205, 0.58));
|
||||
// Typewriter scan: tall text, few regions.
|
||||
assert!(!should_escalate_detection(17.5, 77, 0.55));
|
||||
// Page not downscaled at all: escalation cannot add pixels.
|
||||
assert!(!should_escalate_detection(9.0, 300, 1.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn median_detection_height_scales_and_handles_empty() {
|
||||
let mut heights = vec![30.0, 10.0, 20.0];
|
||||
assert_eq!(median_detection_height(&mut heights, 0.5), 10.0);
|
||||
// Even counts average the two middle values instead of picking the
|
||||
// upper one, so borderline pages don't skew away from escalation.
|
||||
let mut even = vec![10.0, 12.0, 14.0, 30.0];
|
||||
assert_eq!(median_detection_height(&mut even, 1.0), 13.0);
|
||||
let mut empty: Vec<f32> = Vec::new();
|
||||
assert_eq!(median_detection_height(&mut empty, 0.5), f32::MAX);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn concurrency_derivations_stay_in_bounds() {
|
||||
let concurrency = pipeline_concurrency();
|
||||
assert!((1..=3).contains(&concurrency));
|
||||
assert!(intra_threads_per_pipeline(2) == 2);
|
||||
assert!((1..=4).contains(&intra_threads_per_pipeline(1)));
|
||||
}
|
||||
}
|
||||
|
||||
+1
-25
@@ -31,17 +31,6 @@ use super::{
|
||||
/// Bounds live rendered-page memory while preserving small OCR batches.
|
||||
const OCR_PAGE_CHUNK_SIZE: usize = 4;
|
||||
|
||||
/// Pages rendered and held in memory per OCR batch. A parallel engine gets
|
||||
/// three waves of work per batch so its workers are not starved at chunk
|
||||
/// barriers; a sequential engine keeps the small memory-bounding default.
|
||||
/// Engine-reported concurrency is a trait hook, so it is clamped before
|
||||
/// sizing anything from it — this helper exists to bound rendered-page
|
||||
/// memory and must not let an engine inflate it arbitrarily.
|
||||
fn ocr_page_chunk_size(engine_concurrency: usize) -> usize {
|
||||
const MAX_ENGINE_CONCURRENCY: usize = 8;
|
||||
(engine_concurrency.clamp(1, MAX_ENGINE_CONCURRENCY) * 3).max(OCR_PAGE_CHUNK_SIZE)
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
struct OcrEngineCacheKey {
|
||||
model_root: PathBuf,
|
||||
@@ -454,7 +443,7 @@ where
|
||||
let mut render_time_ms = 0u64;
|
||||
let mut ocr_time_ms = 0u64;
|
||||
|
||||
for chunk in routed_pages.chunks(ocr_page_chunk_size(engine.preferred_page_concurrency())) {
|
||||
for chunk in routed_pages.chunks(OCR_PAGE_CHUNK_SIZE) {
|
||||
let native_chunk = chunk
|
||||
.iter()
|
||||
.map(|page_number| {
|
||||
@@ -938,19 +927,6 @@ pub enum OcrPipelineError {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn chunk_size_scales_with_engine_concurrency() {
|
||||
// Sequential engines keep the memory-bounding default.
|
||||
assert_eq!(ocr_page_chunk_size(1), OCR_PAGE_CHUNK_SIZE);
|
||||
// Parallel engines get three waves of work per batch.
|
||||
assert_eq!(ocr_page_chunk_size(3), 9);
|
||||
assert_eq!(ocr_page_chunk_size(2), 6);
|
||||
// Engine-reported concurrency is untrusted: clamp before sizing so a
|
||||
// misbehaving engine cannot inflate rendered-page memory or overflow.
|
||||
assert_eq!(ocr_page_chunk_size(0), OCR_PAGE_CHUNK_SIZE);
|
||||
assert_eq!(ocr_page_chunk_size(usize::MAX), 24);
|
||||
}
|
||||
|
||||
struct TrackingRenderer {
|
||||
batches: Mutex<Vec<Vec<u32>>>,
|
||||
}
|
||||
|
||||
Generated
+2
-2
@@ -724,7 +724,7 @@ checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
@@ -740,7 +740,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
dependencies = [
|
||||
"console_error_panic_hook",
|
||||
"js-sys",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "1.15.0"
|
||||
version = "1.14.2"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
|
||||
Reference in New Issue
Block a user