Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
35e244fde9 | ||
|
|
3327ccddab | ||
|
|
228a3bfd33 | ||
|
|
37c3670bdd | ||
|
|
04324bfeb4 |
@@ -25,9 +25,6 @@ jobs:
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
|
||||
- name: Test developer scripts
|
||||
run: python3 -m unittest discover -s scripts/tests
|
||||
|
||||
fmt:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
@@ -42,9 +39,6 @@ jobs:
|
||||
- name: Check formatting
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Check WASM formatting
|
||||
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
|
||||
|
||||
clippy:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
@@ -83,37 +77,3 @@ jobs:
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
key: wasm
|
||||
|
||||
- name: Check WebAssembly bindings
|
||||
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
|
||||
|
||||
- name: Check root package for WebAssembly
|
||||
run: cargo check --target wasm32-unknown-unknown
|
||||
|
||||
- name: Lint WebAssembly bindings
|
||||
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Test WebAssembly package
|
||||
run: wasm-pack test --node --release wasm
|
||||
|
||||
@@ -1,111 +0,0 @@
|
||||
name: Publish WebAssembly package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['wasm/Cargo.toml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
package_exists: ${{ steps.check.outputs.package_exists }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
|
||||
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
|
||||
echo "package_exists=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
echo "The initial package must be published once before trusted publishing can be configured."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "package_exists=true" >> "$GITHUB_OUTPUT"
|
||||
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
publish:
|
||||
name: Build and publish
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Build browser package
|
||||
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
|
||||
- name: Prepare package metadata
|
||||
run: |
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const path = "wasm/pkg/package.json"
|
||||
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
|
||||
pkg.name = "@firecrawl/pdf-inspector-wasm"
|
||||
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
|
||||
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
|
||||
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
|
||||
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
|
||||
pkg.publishConfig = { access: "public" }
|
||||
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
- name: Inspect package contents
|
||||
run: npm pack --dry-run ./wasm/pkg
|
||||
|
||||
- name: Publish package
|
||||
run: npm publish ./wasm/pkg --provenance --access public
|
||||
+1
-3
@@ -1,13 +1,10 @@
|
||||
# Rust build artifacts
|
||||
/target/
|
||||
/wasm/target/
|
||||
/wasm/pkg/
|
||||
debug/
|
||||
*.pdb
|
||||
|
||||
# Cargo lock (optional for libraries)
|
||||
Cargo.lock
|
||||
!/wasm/Cargo.lock
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
@@ -42,3 +39,4 @@ test_output/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
+12
-18
@@ -12,13 +12,13 @@ readme = "docs/rust-api.md"
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"/src/**",
|
||||
"/external/bcmaps/**",
|
||||
"/docs/rust-api.md",
|
||||
"/LICENSE",
|
||||
"src/**",
|
||||
"external/bcmaps/**",
|
||||
"docs/rust-api.md",
|
||||
"LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"/pdf_inspector.pyi",
|
||||
"pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
@@ -29,11 +29,18 @@ crate-type = ["lib", "cdylib"]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
# Parallel processing
|
||||
rayon = "1.10"
|
||||
|
||||
# Logging
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Text processing
|
||||
regex = "1.10"
|
||||
@@ -43,19 +50,6 @@ unicode-normalization = "0.1"
|
||||
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
||||
ttf-parser = "0.25"
|
||||
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.41.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.3"
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -19,28 +19,24 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only local engines without model-based PDF parsing are shown; OCR was disabled. Scores are 0-1, higher is better.
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| pdf-inspector | 0.83 | 0.89 | 0.66 | 0.74 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
|
||||
Results were refreshed on July 16, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Speed is the median of three complete corpus runs.
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the low end of that range without any OCR, in 4 seconds.
|
||||
|
||||
For context, engines that use OCR or model-based document parsing (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the top of that range without either, in 2.8 seconds.
|
||||
**Where we do well:** Speed (fastest of all engines), the best table detection of any engine shown, and heading detection now on par with opendataloader. Overall lands within 0.01 of opendataloader at roughly 2.5× the speed.
|
||||
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. pdf-inspector delivered the highest overall, reading-order, and table scores, along with the fastest complete run in this benchmark. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
|
||||
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
|
||||
**Where we lag:** Reading order still trails opendataloader slightly, and table structure trails OCR-based engines that can see visual layout.
|
||||
|
||||
## Quick start
|
||||
|
||||
@@ -78,26 +74,6 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Browser WebAssembly
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
```javascript
|
||||
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
|
||||
|
||||
await init();
|
||||
const response = await fetch('/document.pdf');
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
> Full API reference: [wasm/README.md](wasm/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
@@ -209,7 +185,6 @@ src/
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
wasm/ — Browser bindings (wasm-bindgen)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
|
||||
@@ -1,59 +0,0 @@
|
||||
# Benchmarking against OpenDataLoader
|
||||
|
||||
The paired harness runs two `pdf2md` binaries through the same local
|
||||
OpenDataLoader corpus, evaluates both outputs, and reports aggregate and
|
||||
per-document deltas. This avoids comparing results produced from different
|
||||
corpus revisions or evaluator versions.
|
||||
|
||||
Build a candidate and provide a released or worktree build as the baseline:
|
||||
|
||||
```bash
|
||||
cargo build --release
|
||||
python3 scripts/bench_opendataloader.py \
|
||||
--bench-dir ../opendataloader-bench \
|
||||
--baseline ../pdf-inspector-main/target/release/pdf2md \
|
||||
--candidate target/release/pdf2md \
|
||||
--max-document-regression 0.02 \
|
||||
--json-output /tmp/pdf-inspector-benchmark.json
|
||||
```
|
||||
|
||||
Pass `--reference-evaluation path/to/evaluation.json` to report the candidate
|
||||
delta against another evaluation, and add `--require-reference-lead` to make a
|
||||
negative reference delta fail the run. By default, the candidate must not
|
||||
regress the baseline overall score or introduce missing predictions. Use
|
||||
`--min-overall-delta` to require a specific aggregate gain.
|
||||
|
||||
The OpenDataLoader repository is external and keeps its normal
|
||||
`prediction/pdf-inspector` output. Paired evaluation copies each run into a
|
||||
temporary directory before evaluating it, so the baseline and candidate cannot
|
||||
overwrite one another.
|
||||
|
||||
## Published comparison protocol
|
||||
|
||||
The public benchmark table was refreshed on July 16, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Every engine processed the same 200
|
||||
PDFs with OCR disabled. Reported speed is the median of three complete corpus
|
||||
runs; quality scores come from the benchmark evaluator over all 200 outputs.
|
||||
|
||||
## Optional backend evidence probe
|
||||
|
||||
The evidence probe compares positioned `pdf2md` items with MuPDF structured
|
||||
text on the same pages. It is intended to find deterministic extraction or
|
||||
layout evidence that could justify a future native implementation; it does not
|
||||
merge MuPDF output into Markdown, invoke OCR, or add a runtime dependency.
|
||||
|
||||
Install MuPDF's `mutool`, build `pdf2md`, then run:
|
||||
|
||||
```bash
|
||||
python3 scripts/probe_backend_evidence.py document.pdf \
|
||||
--pdf2md target/release/pdf2md \
|
||||
--json-output /tmp/backend-evidence.json
|
||||
```
|
||||
|
||||
The report flags pages when MuPDF exposes a material net token gain, repeated
|
||||
alignment anchors absent from local evidence, or additional image blocks. The
|
||||
JSON includes bounded token samples and page-level counts so promising cases
|
||||
can be inspected without treating backend disagreement as automatically
|
||||
correct. Thresholds are configurable with `--min-token-gain`,
|
||||
`--min-alternate-only-ratio`, and `--min-anchor-gain`.
|
||||
@@ -19,20 +19,3 @@ The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
|
||||
## Browser WebAssembly package
|
||||
|
||||
The browser package is published as `@firecrawl/pdf-inspector-wasm`. Its version lives in `wasm/Cargo.toml`, and `.github/workflows/publish-wasm.yml` builds the `web` target with `wasm-pack` before publishing the generated package.
|
||||
|
||||
The npm package must exist before a trusted publisher can be configured. For the first release only:
|
||||
|
||||
1. Build with `wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release`.
|
||||
2. Inspect with `npm pack --dry-run ./wasm/pkg`.
|
||||
3. Publish with `npm publish ./wasm/pkg --access public` from an authorized maintainer session.
|
||||
4. In the package settings on npm, configure the GitHub Actions trusted publisher:
|
||||
- Organization: `firecrawl`
|
||||
- Repository: `pdf-inspector`
|
||||
- Workflow: `publish-wasm.yml`
|
||||
- Allowed action: `npm publish`
|
||||
|
||||
After that one-time bootstrap, bumping the version in `wasm/Cargo.toml` and merging it to `main` publishes through OIDC. Until the package exists, the workflow exits cleanly without attempting an unauthenticated first publish. See npm's [trusted publishing documentation](https://docs.npmjs.com/trusted-publishers/) for the registry-side setup.
|
||||
|
||||
+5
-7
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
|
||||
+5
-7
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
|
||||
+5
-7
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
|
||||
@@ -1,351 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run a paired pdf-inspector OpenDataLoader benchmark and report deltas."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SCORE_KEYS = (
|
||||
"overall_mean",
|
||||
"nid_mean",
|
||||
"nid_s_mean",
|
||||
"teds_mean",
|
||||
"teds_s_mean",
|
||||
"mhs_mean",
|
||||
"mhs_s_mean",
|
||||
)
|
||||
|
||||
|
||||
def _non_negative_int(value: str) -> int:
|
||||
parsed = int(value)
|
||||
if parsed < 0:
|
||||
raise argparse.ArgumentTypeError("must be non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _non_negative_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed) or parsed < 0.0:
|
||||
raise argparse.ArgumentTypeError("must be finite and non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _finite_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed):
|
||||
raise argparse.ArgumentTypeError("must be finite")
|
||||
return parsed
|
||||
|
||||
|
||||
def _scores(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
score = evaluation.get("metrics", {}).get("score", {})
|
||||
return {key: float(score[key]) for key in SCORE_KEYS if score.get(key) is not None}
|
||||
|
||||
|
||||
def _documents(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
documents: dict[str, float] = {}
|
||||
for document in evaluation.get("documents", []):
|
||||
overall = document.get("scores", {}).get("overall")
|
||||
if overall is not None:
|
||||
documents[str(document["document_id"])] = float(overall)
|
||||
return documents
|
||||
|
||||
|
||||
def compare_evaluations(
|
||||
baseline: dict[str, Any],
|
||||
candidate: dict[str, Any],
|
||||
reference: dict[str, Any] | None = None,
|
||||
*,
|
||||
top: int = 10,
|
||||
) -> dict[str, Any]:
|
||||
"""Build aggregate and per-document deltas from evaluator JSON payloads."""
|
||||
baseline_scores = _scores(baseline)
|
||||
candidate_scores = _scores(candidate)
|
||||
metric_deltas = {
|
||||
key: candidate_scores[key] - baseline_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in baseline_scores and key in candidate_scores
|
||||
}
|
||||
|
||||
baseline_documents = _documents(baseline)
|
||||
candidate_documents = _documents(candidate)
|
||||
shared = sorted(baseline_documents.keys() & candidate_documents.keys())
|
||||
document_deltas = [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"baseline": baseline_documents[document_id],
|
||||
"candidate": candidate_documents[document_id],
|
||||
"delta": candidate_documents[document_id] - baseline_documents[document_id],
|
||||
}
|
||||
for document_id in shared
|
||||
]
|
||||
epsilon = 1e-12
|
||||
improvements = sorted(document_deltas, key=lambda item: item["delta"], reverse=True)
|
||||
regressions = sorted(document_deltas, key=lambda item: item["delta"])
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"baseline": baseline_scores,
|
||||
"candidate": candidate_scores,
|
||||
"deltas": metric_deltas,
|
||||
"missing_predictions": {
|
||||
"baseline": int(baseline.get("metrics", {}).get("missing_predictions", 0)),
|
||||
"candidate": int(candidate.get("metrics", {}).get("missing_predictions", 0)),
|
||||
},
|
||||
"documents": {
|
||||
"shared": len(shared),
|
||||
"improved": sum(item["delta"] > epsilon for item in document_deltas),
|
||||
"regressed": sum(item["delta"] < -epsilon for item in document_deltas),
|
||||
"unchanged": sum(abs(item["delta"]) <= epsilon for item in document_deltas),
|
||||
"largest_improvements": [
|
||||
item for item in improvements if item["delta"] > epsilon
|
||||
][:top],
|
||||
"largest_regressions": [
|
||||
item for item in regressions if item["delta"] < -epsilon
|
||||
][:top],
|
||||
"worst_regression": next(
|
||||
(item for item in regressions if item["delta"] < -epsilon), None
|
||||
),
|
||||
},
|
||||
}
|
||||
if reference is not None:
|
||||
reference_scores = _scores(reference)
|
||||
result["reference"] = reference_scores
|
||||
result["candidate_vs_reference"] = {
|
||||
key: candidate_scores[key] - reference_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in candidate_scores and key in reference_scores
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def evaluate_gates(
|
||||
comparison: dict[str, Any],
|
||||
*,
|
||||
min_overall_delta: float,
|
||||
max_document_regression: float | None,
|
||||
max_missing: int,
|
||||
require_reference_lead: bool,
|
||||
) -> list[str]:
|
||||
"""Return human-readable gate failures; an empty list means pass."""
|
||||
failures: list[str] = []
|
||||
overall_delta = comparison["deltas"].get("overall_mean")
|
||||
if overall_delta is None or overall_delta < min_overall_delta:
|
||||
failures.append(
|
||||
f"overall delta {overall_delta!r} is below {min_overall_delta:+.6f}"
|
||||
)
|
||||
candidate_missing = comparison["missing_predictions"]["candidate"]
|
||||
if candidate_missing > max_missing:
|
||||
failures.append(
|
||||
f"candidate has {candidate_missing} missing predictions (maximum {max_missing})"
|
||||
)
|
||||
if max_document_regression is not None:
|
||||
regression = comparison["documents"].get("worst_regression")
|
||||
if regression is not None and regression["delta"] < -max_document_regression:
|
||||
failures.append(
|
||||
"largest document regression "
|
||||
f"{regression['document_id']}={regression['delta']:+.6f} "
|
||||
f"exceeds {-max_document_regression:+.6f}"
|
||||
)
|
||||
if require_reference_lead:
|
||||
reference_delta = comparison.get("candidate_vs_reference", {}).get("overall_mean")
|
||||
if reference_delta is None:
|
||||
failures.append("reference overall score is unavailable")
|
||||
elif reference_delta < 0.0:
|
||||
failures.append(
|
||||
f"candidate trails reference overall by {reference_delta!r}"
|
||||
)
|
||||
return failures
|
||||
|
||||
|
||||
def _run(command: list[str], *, cwd: Path, env: dict[str, str] | None = None) -> None:
|
||||
print("+", " ".join(command), flush=True)
|
||||
subprocess.run(command, cwd=cwd, env=env, check=True)
|
||||
|
||||
|
||||
def _run_engine(
|
||||
*,
|
||||
bench_dir: Path,
|
||||
python: Path,
|
||||
binary: Path,
|
||||
label: str,
|
||||
scratch_root: Path,
|
||||
) -> dict[str, Any]:
|
||||
env = os.environ.copy()
|
||||
env["PDF_INSPECTOR_BINARY"] = str(binary)
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
if source.exists():
|
||||
if source.is_dir():
|
||||
shutil.rmtree(source)
|
||||
else:
|
||||
source.unlink()
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/pdf_parser.py",
|
||||
"--engine",
|
||||
"pdf-inspector",
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
env=env,
|
||||
)
|
||||
|
||||
if not source.is_dir():
|
||||
raise RuntimeError(f"parser did not produce predictions: {source}")
|
||||
destination = scratch_root / label
|
||||
shutil.copytree(source, destination)
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/evaluator.py",
|
||||
"--prediction-root",
|
||||
str(scratch_root),
|
||||
"--engine",
|
||||
label,
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
)
|
||||
with (destination / "evaluation.json").open(encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def _print_report(comparison: dict[str, Any]) -> None:
|
||||
print("\nMetric baseline candidate delta")
|
||||
print("-------------------- ---------- ---------- ----------")
|
||||
for key in SCORE_KEYS:
|
||||
if key not in comparison["deltas"]:
|
||||
continue
|
||||
print(
|
||||
f"{key:<20} {comparison['baseline'][key]:>10.6f} "
|
||||
f"{comparison['candidate'][key]:>10.6f} "
|
||||
f"{comparison['deltas'][key]:>+10.6f}"
|
||||
)
|
||||
if "reference" in comparison:
|
||||
delta = comparison["candidate_vs_reference"].get("overall_mean")
|
||||
reference = comparison["reference"].get("overall_mean")
|
||||
reference_display = f"{reference:.6f}" if reference is not None else "n/a"
|
||||
delta_display = f"{delta:+.6f}" if delta is not None else "n/a"
|
||||
print(f"\nReference overall: {reference_display}; candidate delta: {delta_display}")
|
||||
|
||||
documents = comparison["documents"]
|
||||
print(
|
||||
"\nDocuments: "
|
||||
f"{documents['improved']} improved, {documents['regressed']} regressed, "
|
||||
f"{documents['unchanged']} unchanged ({documents['shared']} shared)"
|
||||
)
|
||||
for heading, key in (
|
||||
("Largest improvements", "largest_improvements"),
|
||||
("Largest regressions", "largest_regressions"),
|
||||
):
|
||||
print(f"\n{heading}:")
|
||||
rows = documents[key]
|
||||
if not rows:
|
||||
print(" none")
|
||||
for row in rows:
|
||||
print(
|
||||
f" {row['document_id']}: {row['delta']:+.6f} "
|
||||
f"({row['baseline']:.6f} -> {row['candidate']:.6f})"
|
||||
)
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--bench-dir", type=Path, required=True)
|
||||
parser.add_argument("--baseline", type=Path, required=True)
|
||||
parser.add_argument("--candidate", type=Path, required=True)
|
||||
parser.add_argument("--python", type=Path)
|
||||
parser.add_argument("--reference-evaluation", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--top", type=_non_negative_int, default=10)
|
||||
parser.add_argument("--min-overall-delta", type=_finite_float, default=0.0)
|
||||
parser.add_argument("--max-document-regression", type=_non_negative_float)
|
||||
parser.add_argument("--max-missing", type=_non_negative_int, default=0)
|
||||
parser.add_argument("--require-reference-lead", action="store_true")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
bench_dir = args.bench_dir.resolve()
|
||||
baseline = args.baseline.resolve()
|
||||
candidate = args.candidate.resolve()
|
||||
# Keep the virtualenv launcher path intact. Resolving its symlink would
|
||||
# invoke the underlying system interpreter without the benchmark's site
|
||||
# packages.
|
||||
python = (args.python or bench_dir / ".venv" / "bin" / "python").absolute()
|
||||
for path, description in (
|
||||
(bench_dir / "src" / "pdf_parser.py", "OpenDataLoader parser"),
|
||||
(bench_dir / "src" / "evaluator.py", "OpenDataLoader evaluator"),
|
||||
(baseline, "baseline binary"),
|
||||
(candidate, "candidate binary"),
|
||||
(python, "Python interpreter"),
|
||||
):
|
||||
if not path.exists():
|
||||
raise SystemExit(f"{description} not found: {path}")
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="pdf-inspector-opendataloader-") as temporary:
|
||||
scratch_root = Path(temporary)
|
||||
baseline_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=baseline,
|
||||
label="baseline",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
candidate_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=candidate,
|
||||
label="candidate",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
|
||||
reference = None
|
||||
if args.reference_evaluation is not None:
|
||||
with args.reference_evaluation.resolve().open(encoding="utf-8") as handle:
|
||||
reference = json.load(handle)
|
||||
|
||||
comparison = compare_evaluations(
|
||||
baseline_evaluation,
|
||||
candidate_evaluation,
|
||||
reference,
|
||||
top=args.top,
|
||||
)
|
||||
|
||||
_print_report(comparison)
|
||||
if args.json_output is not None:
|
||||
args.json_output.resolve().write_text(
|
||||
json.dumps(comparison, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=args.min_overall_delta,
|
||||
max_document_regression=args.max_document_regression,
|
||||
max_missing=args.max_missing,
|
||||
require_reference_lead=args.require_reference_lead,
|
||||
)
|
||||
if failures:
|
||||
print("\nBenchmark gate failed:", file=sys.stderr)
|
||||
for failure in failures:
|
||||
print(f" - {failure}", file=sys.stderr)
|
||||
return 1
|
||||
print("\nBenchmark gate passed.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,351 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare pdf-inspector evidence with optional MuPDF structured text.
|
||||
|
||||
This is an experiment and diagnostic tool, not an extraction fallback. It runs
|
||||
MuPDF's deterministic ``stext.json`` backend without OCR and highlights pages
|
||||
where that backend exposes materially different text or layout evidence.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from collections import Counter
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any, Iterable
|
||||
|
||||
|
||||
TOKEN_PATTERN = re.compile(r"[^\W_]+(?:[\u2019'][^\W_]+)*", re.UNICODE)
|
||||
|
||||
|
||||
def _tokens(texts: Iterable[str]) -> Counter[str]:
|
||||
tokens: Counter[str] = Counter()
|
||||
for text in texts:
|
||||
for token in TOKEN_PATTERN.findall(text.casefold()):
|
||||
# Lone letters are frequently bullets, chart labels, or fragmented
|
||||
# glyphs. Digits remain useful even when they are one character.
|
||||
if len(token) > 1 or token.isdigit():
|
||||
tokens[token] += 1
|
||||
return tokens
|
||||
|
||||
|
||||
def _repeated_x_anchors(xs: Iterable[float], *, tolerance: float = 4.0) -> int:
|
||||
buckets = Counter(round(float(x) / tolerance) for x in xs)
|
||||
return sum(count >= 3 for count in buckets.values())
|
||||
|
||||
|
||||
def local_pages(payload: dict[str, Any]) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize positioned ``pdf2md --items-json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
for item in payload.get("items", []):
|
||||
page_number = int(item["page"])
|
||||
page = pages.setdefault(
|
||||
page_number,
|
||||
{"texts": [], "xs": [], "text_items": 0, "image_items": 0},
|
||||
)
|
||||
if item.get("item_type") == "image":
|
||||
page["image_items"] += 1
|
||||
continue
|
||||
text = str(item.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
page["xs"].append(float(item.get("x", 0.0)))
|
||||
page["text_items"] += 1
|
||||
return pages
|
||||
|
||||
|
||||
def alternate_pages(
|
||||
payload: dict[str, Any] | list[dict[str, Any]],
|
||||
) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize MuPDF ``stext.json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
raw_pages = payload if isinstance(payload, list) else payload.get("pages", [])
|
||||
for index, raw_page in enumerate(raw_pages, start=1):
|
||||
page_number = int(raw_page.get("number", index))
|
||||
page = {
|
||||
"texts": [],
|
||||
"xs": [],
|
||||
"text_blocks": 0,
|
||||
"text_lines": 0,
|
||||
"image_blocks": 0,
|
||||
}
|
||||
for block in raw_page.get("blocks", []):
|
||||
if block.get("type") == "image":
|
||||
page["image_blocks"] += 1
|
||||
continue
|
||||
if block.get("type") != "text":
|
||||
continue
|
||||
page["text_blocks"] += 1
|
||||
for line in block.get("lines", []):
|
||||
text = str(line.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
bbox = line.get("bbox", {})
|
||||
page["xs"].append(float(bbox.get("x", line.get("x", 0.0))))
|
||||
page["text_lines"] += 1
|
||||
pages[page_number] = page
|
||||
return pages
|
||||
|
||||
|
||||
def compare_page(
|
||||
local: dict[str, Any],
|
||||
alternate: dict[str, Any],
|
||||
*,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
"""Compare semantic and coarse layout evidence for one page."""
|
||||
local_tokens = _tokens(local.get("texts", []))
|
||||
alternate_tokens = _tokens(alternate.get("texts", []))
|
||||
shared = local_tokens & alternate_tokens
|
||||
alternate_only = alternate_tokens - local_tokens
|
||||
local_only = local_tokens - alternate_tokens
|
||||
local_total = sum(local_tokens.values())
|
||||
alternate_total = sum(alternate_tokens.values())
|
||||
shared_total = sum(shared.values())
|
||||
alternate_only_total = sum(alternate_only.values())
|
||||
local_only_total = sum(local_only.values())
|
||||
net_token_gain = alternate_total - local_total
|
||||
alternate_only_ratio = alternate_only_total / max(alternate_total, 1)
|
||||
|
||||
local_anchors = _repeated_x_anchors(local.get("xs", []))
|
||||
alternate_anchors = _repeated_x_anchors(alternate.get("xs", []))
|
||||
anchor_gain = alternate_anchors - local_anchors
|
||||
image_gain = int(alternate.get("image_blocks", 0)) - int(
|
||||
local.get("image_items", 0)
|
||||
)
|
||||
|
||||
reasons: list[str] = []
|
||||
if local_total == 0 and alternate_total >= max(5, min_token_gain // 2):
|
||||
reasons.append("local_text_empty")
|
||||
elif (
|
||||
net_token_gain >= min_token_gain
|
||||
and alternate_only_ratio >= min_alternate_only_ratio
|
||||
):
|
||||
reasons.append("alternate_has_more_text")
|
||||
if anchor_gain >= min_anchor_gain:
|
||||
reasons.append("alternate_has_more_alignment_anchors")
|
||||
if image_gain > 0:
|
||||
reasons.append("alternate_has_more_image_blocks")
|
||||
|
||||
if reasons:
|
||||
classification = "investigate_alternate_evidence"
|
||||
elif local_total - alternate_total >= min_token_gain:
|
||||
classification = "local_has_more_text"
|
||||
elif alternate_only_total + local_only_total:
|
||||
classification = "different_segmentation_or_decoding"
|
||||
else:
|
||||
classification = "equivalent_text_evidence"
|
||||
|
||||
return {
|
||||
"classification": classification,
|
||||
"reasons": reasons,
|
||||
"tokens": {
|
||||
"local": local_total,
|
||||
"alternate": alternate_total,
|
||||
"shared": shared_total,
|
||||
"net_alternate_gain": net_token_gain,
|
||||
"alternate_only": alternate_only_total,
|
||||
"local_only": local_only_total,
|
||||
"alternate_only_ratio": alternate_only_ratio,
|
||||
"alternate_only_sample": sorted(alternate_only)[:12],
|
||||
"local_only_sample": sorted(local_only)[:12],
|
||||
},
|
||||
"layout": {
|
||||
"local_text_items": int(local.get("text_items", 0)),
|
||||
"local_image_items": int(local.get("image_items", 0)),
|
||||
"local_repeated_x_anchors": local_anchors,
|
||||
"alternate_text_blocks": int(alternate.get("text_blocks", 0)),
|
||||
"alternate_text_lines": int(alternate.get("text_lines", 0)),
|
||||
"alternate_image_blocks": int(alternate.get("image_blocks", 0)),
|
||||
"alternate_repeated_x_anchors": alternate_anchors,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def compare_documents(
|
||||
local_payload: dict[str, Any],
|
||||
alternate_payload: dict[str, Any] | list[dict[str, Any]],
|
||||
*,
|
||||
min_token_gain: int = 20,
|
||||
min_alternate_only_ratio: float = 0.15,
|
||||
min_anchor_gain: int = 2,
|
||||
) -> dict[str, Any]:
|
||||
"""Return a page-level evidence report for already extracted payloads."""
|
||||
local = local_pages(local_payload)
|
||||
alternate = alternate_pages(alternate_payload)
|
||||
page_numbers = sorted(local.keys() | alternate.keys())
|
||||
pages = []
|
||||
for page_number in page_numbers:
|
||||
result = compare_page(
|
||||
local.get(page_number, {}),
|
||||
alternate.get(page_number, {}),
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
result["page"] = page_number
|
||||
pages.append(result)
|
||||
|
||||
flagged = [
|
||||
page
|
||||
for page in pages
|
||||
if page["classification"] == "investigate_alternate_evidence"
|
||||
]
|
||||
return {
|
||||
"summary": {
|
||||
"pages": len(pages),
|
||||
"flagged_pages": len(flagged),
|
||||
"flagged_page_numbers": [page["page"] for page in flagged],
|
||||
"local_tokens": sum(page["tokens"]["local"] for page in pages),
|
||||
"alternate_tokens": sum(page["tokens"]["alternate"] for page in pages),
|
||||
"alternate_only_tokens": sum(
|
||||
page["tokens"]["alternate_only"] for page in pages
|
||||
),
|
||||
},
|
||||
"pages": pages,
|
||||
}
|
||||
|
||||
|
||||
def _json_command(command: list[str]) -> Any:
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
check=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as error:
|
||||
detail = error.stderr.strip() or error.stdout.strip() or "no diagnostic output"
|
||||
raise RuntimeError(f"command failed: {' '.join(command)}\n{detail}") from error
|
||||
try:
|
||||
return json.loads(completed.stdout)
|
||||
except json.JSONDecodeError as error:
|
||||
raise RuntimeError(
|
||||
f"command did not return JSON: {' '.join(command)}: {error}"
|
||||
) from error
|
||||
|
||||
|
||||
def probe_pdf(
|
||||
pdf: Path,
|
||||
*,
|
||||
pdf2md: Path,
|
||||
mutool: Path,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
local_payload = _json_command([str(pdf2md), str(pdf), "--items-json"])
|
||||
# `stext.json` is MuPDF's native structured text output. The OCR formats
|
||||
# are intentionally not used so this remains a deterministic no-model
|
||||
# comparison.
|
||||
alternate_payload = _json_command(
|
||||
[str(mutool), "draw", "-q", "-F", "stext.json", "-o", "-", str(pdf)]
|
||||
)
|
||||
report = compare_documents(
|
||||
local_payload,
|
||||
alternate_payload,
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
report["pdf"] = str(pdf)
|
||||
return report
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("pdf", type=Path, nargs="+")
|
||||
parser.add_argument("--pdf2md", type=Path, default=Path("target/release/pdf2md"))
|
||||
parser.add_argument("--mutool", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--min-token-gain", type=int, default=20)
|
||||
parser.add_argument("--min-alternate-only-ratio", type=float, default=0.15)
|
||||
parser.add_argument("--min-anchor-gain", type=int, default=2)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def _print_report(result: dict[str, Any]) -> None:
|
||||
summary = result["summary"]
|
||||
print(f"\n{result['pdf']}")
|
||||
print(
|
||||
f" {summary['flagged_pages']}/{summary['pages']} pages flagged; "
|
||||
f"tokens local={summary['local_tokens']} alternate={summary['alternate_tokens']} "
|
||||
f"alternate-only={summary['alternate_only_tokens']}"
|
||||
)
|
||||
for page in result["pages"]:
|
||||
if page["classification"] != "investigate_alternate_evidence":
|
||||
continue
|
||||
reasons = ", ".join(page["reasons"])
|
||||
tokens = page["tokens"]
|
||||
print(
|
||||
f" page {page['page']}: {reasons}; "
|
||||
f"net tokens={tokens['net_alternate_gain']:+d}, "
|
||||
f"alternate-only={tokens['alternate_only']}"
|
||||
)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
pdf2md = args.pdf2md.absolute()
|
||||
mutool = args.mutool or (Path(found) if (found := shutil.which("mutool")) else None)
|
||||
if not pdf2md.is_file():
|
||||
print(f"error: pdf2md binary not found: {pdf2md}", file=sys.stderr)
|
||||
return 2
|
||||
if mutool is None or not mutool.is_file():
|
||||
print("error: mutool not found; install MuPDF or pass --mutool", file=sys.stderr)
|
||||
return 2
|
||||
if (
|
||||
args.min_token_gain < 0
|
||||
or args.min_anchor_gain < 0
|
||||
or not 0.0 <= args.min_alternate_only_ratio <= 1.0
|
||||
):
|
||||
print("error: thresholds must be non-negative and ratio must be in [0, 1]", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
results = []
|
||||
for pdf in args.pdf:
|
||||
path = pdf.absolute()
|
||||
if not path.is_file():
|
||||
print(f"error: PDF not found: {path}", file=sys.stderr)
|
||||
return 2
|
||||
try:
|
||||
result = probe_pdf(
|
||||
path,
|
||||
pdf2md=pdf2md,
|
||||
mutool=mutool,
|
||||
min_token_gain=args.min_token_gain,
|
||||
min_alternate_only_ratio=args.min_alternate_only_ratio,
|
||||
min_anchor_gain=args.min_anchor_gain,
|
||||
)
|
||||
except RuntimeError as error:
|
||||
print(f"error: {error}", file=sys.stderr)
|
||||
return 1
|
||||
results.append(result)
|
||||
_print_report(result)
|
||||
|
||||
payload = {
|
||||
"schema_version": 1,
|
||||
"experiment": "optional_mupdf_stext_evidence",
|
||||
"ocr": False,
|
||||
"thresholds": {
|
||||
"min_token_gain": args.min_token_gain,
|
||||
"min_alternate_only_ratio": args.min_alternate_only_ratio,
|
||||
"min_anchor_gain": args.min_anchor_gain,
|
||||
},
|
||||
"documents": results,
|
||||
}
|
||||
if args.json_output:
|
||||
args.json_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.json_output.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,203 +0,0 @@
|
||||
import io
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from contextlib import redirect_stderr, redirect_stdout
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from bench_opendataloader import (
|
||||
_arguments,
|
||||
_print_report,
|
||||
_run_engine,
|
||||
compare_evaluations,
|
||||
evaluate_gates,
|
||||
)
|
||||
|
||||
|
||||
def evaluation(overall, documents, *, missing=0):
|
||||
return {
|
||||
"metrics": {
|
||||
"score": {
|
||||
"overall_mean": overall,
|
||||
"nid_mean": overall + 0.01,
|
||||
},
|
||||
"missing_predictions": missing,
|
||||
},
|
||||
"documents": [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"scores": {"overall": score},
|
||||
}
|
||||
for document_id, score in documents.items()
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class ComparisonTests(unittest.TestCase):
|
||||
def test_reports_metric_and_document_deltas(self):
|
||||
baseline = evaluation(0.80, {"a": 0.8, "b": 0.6, "c": 0.7})
|
||||
candidate = evaluation(0.82, {"a": 0.9, "b": 0.5, "c": 0.7})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, top=1)
|
||||
|
||||
self.assertAlmostEqual(result["deltas"]["overall_mean"], 0.02)
|
||||
self.assertEqual(result["documents"]["improved"], 1)
|
||||
self.assertEqual(result["documents"]["regressed"], 1)
|
||||
self.assertEqual(result["documents"]["unchanged"], 1)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_improvements"][0]["document_id"], "a"
|
||||
)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_regressions"][0]["document_id"], "b"
|
||||
)
|
||||
|
||||
def test_reference_delta_is_reported(self):
|
||||
baseline = evaluation(0.80, {})
|
||||
candidate = evaluation(0.82, {})
|
||||
reference = evaluation(0.81, {})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, reference)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
result["candidate_vs_reference"]["overall_mean"], 0.01
|
||||
)
|
||||
|
||||
def test_gates_cover_aggregate_document_missing_and_reference(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.79, {"a": 0.7}, missing=1),
|
||||
evaluation(0.81, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 4)
|
||||
|
||||
def test_regression_gate_is_independent_of_report_limit(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.80, {"a": 0.7}),
|
||||
top=0,
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=False,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 1)
|
||||
self.assertIn("largest document regression", failures[0])
|
||||
|
||||
def test_report_handles_reference_without_overall_score(self):
|
||||
result = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
{"metrics": {"score": {"nid_mean": 0.81}}},
|
||||
)
|
||||
|
||||
output = io.StringIO()
|
||||
with redirect_stdout(output):
|
||||
_print_report(result)
|
||||
|
||||
self.assertIn("Reference overall: n/a; candidate delta: n/a", output.getvalue())
|
||||
|
||||
def test_reference_gate_reports_missing_score_as_unavailable(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=None,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(failures, ["reference overall score is unavailable"])
|
||||
|
||||
def test_arguments_reject_negative_counts_and_allow_zero_top(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
self.assertEqual(_arguments(required + ["--top", "0"]).top, 0)
|
||||
for option in ("--top", "--max-document-regression", "--max-missing"):
|
||||
with self.subTest(option=option), redirect_stderr(io.StringIO()):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, "-1"])
|
||||
|
||||
def test_arguments_reject_nonfinite_float_thresholds(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
for option in ("--min-overall-delta", "--max-document-regression"):
|
||||
for value in ("nan", "inf", "-inf"):
|
||||
with self.subTest(option=option, value=value), redirect_stderr(
|
||||
io.StringIO()
|
||||
):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, value])
|
||||
|
||||
def test_run_engine_clears_stale_predictions_before_parser(self):
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
bench_dir = root / "bench"
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
source.mkdir(parents=True)
|
||||
(source / "stale.md").write_text("stale", encoding="utf-8")
|
||||
scratch = root / "scratch"
|
||||
scratch.mkdir()
|
||||
|
||||
def fake_run(command, *, cwd, env=None):
|
||||
if any(part.endswith("pdf_parser.py") for part in command):
|
||||
self.assertFalse(source.exists())
|
||||
(source / "markdown").mkdir(parents=True)
|
||||
(source / "markdown" / "new.md").write_text(
|
||||
"new", encoding="utf-8"
|
||||
)
|
||||
else:
|
||||
destination = scratch / "candidate"
|
||||
(destination / "evaluation.json").write_text(
|
||||
json.dumps(evaluation(0.82, {})), encoding="utf-8"
|
||||
)
|
||||
|
||||
with patch("bench_opendataloader._run", side_effect=fake_run):
|
||||
result = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=Path("python"),
|
||||
binary=Path("pdf2md"),
|
||||
label="candidate",
|
||||
scratch_root=scratch,
|
||||
)
|
||||
|
||||
self.assertEqual(result["metrics"]["score"]["overall_mean"], 0.82)
|
||||
self.assertFalse((source / "stale.md").exists())
|
||||
self.assertFalse((scratch / "candidate" / "stale.md").exists())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,103 +0,0 @@
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from probe_backend_evidence import compare_documents
|
||||
|
||||
|
||||
def local_payload(items):
|
||||
return {"items": items}
|
||||
|
||||
|
||||
def item(page, text, x=10, item_type="text"):
|
||||
return {"page": page, "text": text, "x": x, "item_type": item_type}
|
||||
|
||||
|
||||
def alternate_payload(pages):
|
||||
return {"pages": pages}
|
||||
|
||||
|
||||
def page(lines, *, images=0):
|
||||
blocks = [
|
||||
{
|
||||
"type": "text",
|
||||
"lines": [
|
||||
{"text": text, "bbox": {"x": x, "y": index * 10, "w": 80, "h": 8}}
|
||||
for index, (text, x) in enumerate(lines)
|
||||
],
|
||||
}
|
||||
]
|
||||
blocks.extend({"type": "image"} for _ in range(images))
|
||||
return {"blocks": blocks}
|
||||
|
||||
|
||||
class EvidenceComparisonTests(unittest.TestCase):
|
||||
def test_accepts_real_top_level_page_array(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = [page([("alpha beta gamma", 10)])]
|
||||
|
||||
result = compare_documents(local, alternate)["pages"][0]
|
||||
|
||||
self.assertEqual(result["tokens"]["alternate"], 3)
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 1)
|
||||
|
||||
def test_flags_material_alternate_text_gain(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = alternate_payload(
|
||||
[page([("alpha beta gamma delta epsilon zeta", 10)])]
|
||||
)
|
||||
|
||||
report = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=3,
|
||||
min_alternate_only_ratio=0.2,
|
||||
)
|
||||
|
||||
result = report["pages"][0]
|
||||
self.assertEqual(result["classification"], "investigate_alternate_evidence")
|
||||
self.assertIn("alternate_has_more_text", result["reasons"])
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 4)
|
||||
|
||||
def test_repeated_alignment_and_image_evidence_are_reported(self):
|
||||
local = local_payload([item(1, "one two", 10)])
|
||||
alternate = alternate_payload(
|
||||
[
|
||||
page(
|
||||
[
|
||||
("one two", 10),
|
||||
("row three", 100),
|
||||
("row four", 100),
|
||||
("row five", 100),
|
||||
],
|
||||
images=1,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
result = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=99,
|
||||
min_anchor_gain=1,
|
||||
)["pages"][0]
|
||||
|
||||
self.assertIn("alternate_has_more_alignment_anchors", result["reasons"])
|
||||
self.assertIn("alternate_has_more_image_blocks", result["reasons"])
|
||||
self.assertEqual(result["layout"]["alternate_repeated_x_anchors"], 1)
|
||||
|
||||
def test_token_segmentation_difference_does_not_imply_more_evidence(self):
|
||||
local = local_payload([item(1, "Revenue 2025")])
|
||||
alternate = alternate_payload([page([("Revenue 2024", 10)])])
|
||||
|
||||
result = compare_documents(local, alternate, min_token_gain=2)["pages"][0]
|
||||
|
||||
self.assertEqual(result["classification"], "different_segmentation_or_decoding")
|
||||
self.assertEqual(result["reasons"], [])
|
||||
self.assertEqual(result["tokens"]["alternate_only_sample"], ["2024"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+353
-1097
File diff suppressed because it is too large
Load Diff
@@ -63,7 +63,6 @@ fn json_escape(s: &str) -> String {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
|
||||
@@ -190,7 +190,6 @@ fn print_layout_info(layout: &LayoutComplexity) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
|
||||
@@ -162,7 +162,7 @@ pub(crate) fn extract_page_text_items(
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
|
||||
// Build font encoding maps from Differences arrays
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
|
||||
|
||||
// Build font width info for accurate text positioning
|
||||
let font_widths = build_font_widths(doc, &fonts);
|
||||
|
||||
+12
-154
@@ -497,13 +497,9 @@ pub(crate) fn get_operand_bytes(obj: &Object) -> Option<&[u8]> {
|
||||
/// Build encoding maps for all fonts on a page.
|
||||
/// Returns `(encodings, has_gid_fonts)` where `has_gid_fonts` is true when
|
||||
/// any font uses raw glyph ID names (gidNNNNN) that can't be decoded.
|
||||
/// Gid names whose codes the font's own ToUnicode CMap maps are decodable
|
||||
/// and do not set the flag (LibreOffice subsets write /gidNNNN Differences
|
||||
/// names alongside a complete ToUnicode CMap).
|
||||
pub(crate) fn build_font_encodings(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
cmaps: &FontCMaps,
|
||||
) -> (PageFontEncodings, bool) {
|
||||
let mut encodings = PageFontEncodings::new();
|
||||
let mut has_gid_fonts = false;
|
||||
@@ -512,9 +508,7 @@ pub(crate) fn build_font_encodings(
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
|
||||
if let Some(result) = parse_font_encoding(doc, font_dict) {
|
||||
if !result.gid_codes.is_empty()
|
||||
&& !tounicode_maps_codes(font_dict, cmaps, &result.gid_codes)
|
||||
{
|
||||
if result.gid_glyph_count > 0 {
|
||||
has_gid_fonts = true;
|
||||
}
|
||||
if !result.map.is_empty() {
|
||||
@@ -526,34 +520,6 @@ pub(crate) fn build_font_encodings(
|
||||
(encodings, has_gid_fonts)
|
||||
}
|
||||
|
||||
/// True when the font's ToUnicode CMap maps the gid-named character codes,
|
||||
/// so the Differences entries still decode through the CMap.
|
||||
fn tounicode_maps_codes(font_dict: &lopdf::Dictionary, cmaps: &FontCMaps, codes: &[u8]) -> bool {
|
||||
let Some(obj_ref) = font_dict
|
||||
.get(b"ToUnicode")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
else {
|
||||
return false;
|
||||
};
|
||||
let Some(entry) = cmaps.get_by_obj(obj_ref.0) else {
|
||||
return false;
|
||||
};
|
||||
// At least one gid code usably mapped means the CMap addresses these
|
||||
// codes; remaining unmapped codes are subset leftovers (e.g. the
|
||||
// component glyphs of an emoji ZWJ sequence mapped whole on its first
|
||||
// code). A mapping is usable only when extraction would accept it —
|
||||
// empty or U+FFFD results are rejected there as invalid. Fonts whose
|
||||
// CMap ignores the gid codes entirely stay flagged, and the downstream
|
||||
// garbage/encoding checks still catch partial damage.
|
||||
codes.iter().any(|&code| {
|
||||
entry
|
||||
.primary
|
||||
.lookup(code as u16)
|
||||
.is_some_and(|s| !s.is_empty() && !s.contains('\u{FFFD}'))
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse font encoding from a font dictionary
|
||||
pub(crate) fn parse_font_encoding(
|
||||
doc: &Document,
|
||||
@@ -592,10 +558,11 @@ pub(crate) fn parse_font_encoding(
|
||||
/// Result of parsing an encoding dictionary's Differences array.
|
||||
pub(crate) struct EncodingResult {
|
||||
pub map: FontEncodingMap,
|
||||
/// Character codes whose glyph names match the `gidNNNNN` pattern (raw
|
||||
/// glyph IDs). These reference the original font's glyph table and are
|
||||
/// only decodable when the font's ToUnicode CMap maps the code.
|
||||
pub gid_codes: Vec<u8>,
|
||||
/// Number of glyph names matching the `gidNNNNN` pattern (raw glyph IDs).
|
||||
/// These indicate a font with unresolvable encoding — the glyph IDs
|
||||
/// reference the original font's glyph table, but without the original
|
||||
/// font's cmap there is no way to map them to Unicode.
|
||||
pub gid_glyph_count: u32,
|
||||
}
|
||||
|
||||
/// Parse an encoding dictionary with Differences array
|
||||
@@ -621,7 +588,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
let mut encoding_map = FontEncodingMap::new();
|
||||
let mut current_code: u8 = 0;
|
||||
let mut ligature_count = 0u32;
|
||||
let mut gid_codes: Vec<u8> = Vec::new();
|
||||
let mut gid_glyph_count = 0u32;
|
||||
|
||||
for item in diff_array {
|
||||
match item {
|
||||
@@ -647,7 +614,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
&& glyph_name.len() >= 4
|
||||
&& glyph_name[3..].chars().all(|c| c.is_ascii_digit())
|
||||
{
|
||||
gid_codes.push(current_code);
|
||||
gid_glyph_count += 1;
|
||||
}
|
||||
if let Some(ch) = mapped_char {
|
||||
encoding_map.insert(current_code, ch);
|
||||
@@ -671,16 +638,16 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
);
|
||||
}
|
||||
|
||||
if !gid_codes.is_empty() {
|
||||
if gid_glyph_count > 0 {
|
||||
debug!(
|
||||
" Differences: {} gid-encoded glyphs (decodable only via ToUnicode)",
|
||||
gid_codes.len()
|
||||
" Differences: {} gid-encoded glyphs (unresolvable without original font)",
|
||||
gid_glyph_count
|
||||
);
|
||||
}
|
||||
|
||||
Some(EncodingResult {
|
||||
map: encoding_map,
|
||||
gid_codes,
|
||||
gid_glyph_count,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1971,113 +1938,4 @@ mod tests {
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
fn gid_font_doc(bfchar: Option<&str>) -> (Document, lopdf::ObjectId) {
|
||||
use lopdf::Stream;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let cmap = format!(
|
||||
"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
{}
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end",
|
||||
bfchar.unwrap_or_default()
|
||||
);
|
||||
let tounicode_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
cmap.into_bytes(),
|
||||
)));
|
||||
let enc_id = doc.add_object(dictionary! {
|
||||
"Type" => "Encoding",
|
||||
"Differences" => vec![
|
||||
1.into(),
|
||||
Object::Name(b"gid1283".to_vec()),
|
||||
Object::Name(b"gid1464".to_vec()),
|
||||
],
|
||||
});
|
||||
let mut font = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "ABCDEF+OpenSymbol",
|
||||
"Encoding" => Object::Reference(enc_id),
|
||||
};
|
||||
if bfchar.is_some() {
|
||||
font.set("ToUnicode", Object::Reference(tounicode_id));
|
||||
}
|
||||
let font_id = doc.add_object(font);
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn gid_flagged(bfchar: Option<&str>) -> bool {
|
||||
let (doc, page_id) = gid_font_doc(bfchar);
|
||||
let cmaps = FontCMaps::from_doc(&doc);
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap();
|
||||
let (_, has_gid_fonts) = build_font_encodings(&doc, &fonts, &cmaps);
|
||||
has_gid_fonts
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_covering_tounicode_are_not_flagged() {
|
||||
// LibreOffice subsets write /gidNNNN Differences names alongside a
|
||||
// ToUnicode CMap that decodes those codes; the page must not be
|
||||
// flagged as unresolvable (which would suppress the whole document's
|
||||
// markdown when every page carries such a font).
|
||||
assert!(!gid_flagged(Some("<01> <2022>\n<02> <25E6>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_partial_tounicode_are_not_flagged() {
|
||||
// An emoji ZWJ sequence maps whole on its first code; the remaining
|
||||
// component-glyph codes are subset leftovers, not damage.
|
||||
assert!(!gid_flagged(Some(
|
||||
"<01> <D83DDC68200DD83DDC69200DD83DDC67>"
|
||||
)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_without_tounicode_are_flagged() {
|
||||
assert!(
|
||||
gid_flagged(None),
|
||||
"gid glyphs without ToUnicode are unresolvable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_disjoint_tounicode_are_flagged() {
|
||||
// A ToUnicode that never addresses the gid codes leaves them
|
||||
// unresolvable.
|
||||
assert!(gid_flagged(Some("<10> <0041>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_replacement_char_tounicode_are_flagged() {
|
||||
// A mapping to U+FFFD is not usable — extraction rejects it as an
|
||||
// invalid CMap result — so it must not clear the gid flag.
|
||||
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
|
||||
}
|
||||
}
|
||||
|
||||
+5
-92
@@ -1153,22 +1153,6 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds(items, &HashMap::new(), &HashSet::new())
|
||||
}
|
||||
|
||||
/// Group text items into lines without removing numeric page headers or footers.
|
||||
///
|
||||
/// Plain-text extraction uses this path because every extracted item is part of
|
||||
/// the API result. Markdown conversion keeps using [`group_into_lines`], where
|
||||
/// page-number suppression is an intentional presentation cleanup.
|
||||
pub fn group_into_lines_preserving_all_text(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
/// Group text items into lines, using pre-computed per-page adaptive thresholds
|
||||
/// from Canva-style letter-spacing detection. Falls back to computing the
|
||||
/// threshold from item gaps when no pre-computed value is available.
|
||||
@@ -1195,55 +1179,16 @@ pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
&HashMap::new(),
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
true,
|
||||
)
|
||||
}
|
||||
|
||||
fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
filter_page_numbers: bool,
|
||||
) -> Vec<TextLine> {
|
||||
if items.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Markdown output omits standalone numeric headers/footers. Plain-text
|
||||
// callers opt out because dropping extracted text violates that API.
|
||||
let items = if filter_page_numbers {
|
||||
items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.collect()
|
||||
} else {
|
||||
items
|
||||
};
|
||||
// Filter out page numbers (standalone numbers at top/bottom of page)
|
||||
let items: Vec<TextItem> = items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.collect();
|
||||
|
||||
// Get unique pages
|
||||
let mut pages: Vec<u32> = items.iter().map(|i| i.page).collect();
|
||||
@@ -1260,38 +1205,6 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
// Non-Canva pages use the default 0.10 threshold.
|
||||
let adaptive_threshold = page_thresholds.get(&page).copied().unwrap_or(0.10);
|
||||
|
||||
// Image-backed region graphs recover local/asymmetric column flows
|
||||
// that a whole-page projection cannot represent. Charts already have
|
||||
// their own positioned-region ordering and therefore stay on that path.
|
||||
if !chart_regions.contains_key(&page) {
|
||||
let preliminary_columns =
|
||||
detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
let detected_split =
|
||||
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
|
||||
if let Some(band) = image_regions.get(&page).and_then(|regions| {
|
||||
super::reading_order::infer_image_anchored_flow(
|
||||
&page_items,
|
||||
regions,
|
||||
detected_split,
|
||||
)
|
||||
}) {
|
||||
debug!(
|
||||
"page {}: image-anchored region graph split={:.1} y=[{:.1}..{:.1}]",
|
||||
page, band.split_x, band.y_bottom, band.y_top
|
||||
);
|
||||
for node in super::reading_order::build_region_graph(page_items, band) {
|
||||
debug!(
|
||||
"page {}: region node {:?} items={}",
|
||||
page,
|
||||
node.kind,
|
||||
node.items.len()
|
||||
);
|
||||
all_lines.extend(group_single_column(node.items, adaptive_threshold));
|
||||
}
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Detect columns for this page, blind to chart text.
|
||||
debug!(
|
||||
"page {}: grouping chart-aware={} regions={:?}",
|
||||
|
||||
+1
-15
@@ -6,7 +6,6 @@ pub(crate) mod content_stream;
|
||||
mod fonts;
|
||||
mod layout;
|
||||
mod links;
|
||||
mod reading_order;
|
||||
pub(crate) mod underline;
|
||||
mod xobjects;
|
||||
|
||||
@@ -27,12 +26,11 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
pub use layout::group_into_lines;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
@@ -1519,18 +1517,6 @@ mod tests {
|
||||
assert_eq!(lines[1].text(), "Next line");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserving_all_text_keeps_numeric_page_footer() {
|
||||
let mut page_number = make_merge_item("42", 100.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
assert!(group_into_lines(vec![page_number.clone()]).is_empty());
|
||||
|
||||
let lines = group_into_lines_preserving_all_text(vec![page_number]);
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bold_italic_detection() {
|
||||
// Test bold detection
|
||||
|
||||
@@ -1,591 +0,0 @@
|
||||
//! Region-graph evidence for page reading order.
|
||||
//!
|
||||
//! Whole-page column histograms fail when images or spanning captions occupy
|
||||
//! only part of a page. This module turns image geometry and repeated row
|
||||
//! gutters into a small directed acyclic graph: content above a local column
|
||||
//! band, the left flow, the right flow, and content below it. The graph is
|
||||
//! deliberately evidence-gated; ordinary pages keep the established layout
|
||||
//! path.
|
||||
|
||||
use crate::text_utils::{effective_width, is_cjk_char, is_rtl_text};
|
||||
use crate::types::TextItem;
|
||||
|
||||
const MIN_IMAGE_WIDTH: f32 = 60.0;
|
||||
const MIN_IMAGE_HEIGHT: f32 = 40.0;
|
||||
const MIN_ROW_GUTTER: f32 = 8.0;
|
||||
const SPLIT_CLUSTER_TOLERANCE: f32 = 20.0;
|
||||
const MIN_ALIGNED_ROWS: usize = 4;
|
||||
|
||||
pub(crate) type ImageRegion = (f32, f32, f32, f32);
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub(crate) struct ColumnFlowBand {
|
||||
pub(crate) split_x: f32,
|
||||
pub(crate) y_bottom: f32,
|
||||
pub(crate) y_top: f32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum RegionKind {
|
||||
FullWidth,
|
||||
Column,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RegionNode {
|
||||
pub(crate) kind: RegionKind,
|
||||
pub(crate) items: Vec<TextItem>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Row<'a> {
|
||||
y: f32,
|
||||
items: Vec<&'a TextItem>,
|
||||
}
|
||||
|
||||
fn page_x_bounds(items: &[TextItem], images: &[ImageRegion]) -> Option<(f32, f32)> {
|
||||
let text_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let text_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + effective_width(item))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let image_min = images
|
||||
.iter()
|
||||
.map(|region| region.0.min(region.2))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_max = images
|
||||
.iter()
|
||||
.map(|region| region.0.max(region.2))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let x_min = text_min.min(image_min);
|
||||
let x_max = text_max.max(image_max);
|
||||
(x_min.is_finite() && x_max.is_finite() && x_max > x_min).then_some((x_min, x_max))
|
||||
}
|
||||
|
||||
fn group_rows(items: &[TextItem]) -> Vec<Row<'_>> {
|
||||
const Y_TOLERANCE: f32 = 3.0;
|
||||
let mut sorted: Vec<&TextItem> = items.iter().collect();
|
||||
sorted.sort_by(|left, right| right.y.total_cmp(&left.y));
|
||||
let mut rows: Vec<Row<'_>> = Vec::new();
|
||||
for item in sorted {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| (row.y - item.y).abs() <= Y_TOLERANCE)
|
||||
{
|
||||
row.items.push(item);
|
||||
row.y = row.items.iter().map(|member| member.y).sum::<f32>() / row.items.len() as f32;
|
||||
} else {
|
||||
rows.push(Row {
|
||||
y: item.y,
|
||||
items: vec![item],
|
||||
});
|
||||
}
|
||||
}
|
||||
for row in &mut rows {
|
||||
row.items.sort_by(|left, right| left.x.total_cmp(&right.x));
|
||||
}
|
||||
rows
|
||||
}
|
||||
|
||||
fn side_is_prose(items: &[&TextItem]) -> bool {
|
||||
let text = items
|
||||
.iter()
|
||||
.map(|item| item.text.trim())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let alphabetic_count = text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphabetic())
|
||||
.count();
|
||||
let cjk_count = text
|
||||
.chars()
|
||||
.filter(|character| is_cjk_char(*character))
|
||||
.count();
|
||||
(text.split_whitespace().count() >= 3 || cjk_count >= 10) && alphabetic_count >= 10
|
||||
}
|
||||
|
||||
fn aligned_row_split(row: &Row<'_>, x_min: f32, x_max: f32) -> Option<f32> {
|
||||
if row.items.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let page_width = x_max - x_min;
|
||||
let center_low = x_min + page_width * 0.25;
|
||||
let center_high = x_min + page_width * 0.75;
|
||||
row.items
|
||||
.windows(2)
|
||||
.filter_map(|pair| {
|
||||
let left_end = pair[0].x + effective_width(pair[0]);
|
||||
let right_start = pair[1].x;
|
||||
let gap = right_start - left_end;
|
||||
let split_x = (left_end + right_start) / 2.0;
|
||||
if gap < MIN_ROW_GUTTER || split_x < center_low || split_x > center_high {
|
||||
return None;
|
||||
}
|
||||
let left: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 < split_x)
|
||||
.collect();
|
||||
let right: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 >= split_x)
|
||||
.collect();
|
||||
(side_is_prose(&left) && side_is_prose(&right)).then_some((split_x, gap))
|
||||
})
|
||||
.max_by(|left, right| left.1.total_cmp(&right.1))
|
||||
.map(|candidate| candidate.0)
|
||||
}
|
||||
|
||||
fn local_flow_below_full_width_image(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
let full_width_images: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let width = (x1 - x0).abs();
|
||||
let height = (y1 - y0).abs();
|
||||
width >= page_width * 0.65 && height >= 60.0
|
||||
})
|
||||
.collect();
|
||||
// A local column flow below an image is only unambiguous for a single,
|
||||
// nearly square hero/figure. Wide report banners and full-page artwork
|
||||
// frequently sit above unrelated page furniture whose aligned labels can
|
||||
// mimic prose columns.
|
||||
if full_width_images.len() != 1 {
|
||||
return None;
|
||||
}
|
||||
let (image_x0, _, image_x1, _) = full_width_images[0];
|
||||
let anchor_width = (image_x1 - image_x0).abs();
|
||||
let anchor_height = (full_width_images[0].3 - full_width_images[0].1).abs();
|
||||
if anchor_width < page_width * 0.85
|
||||
|| anchor_height < anchor_width * 0.85
|
||||
|| anchor_height > anchor_width * 1.2
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let image_bottom = full_width_images
|
||||
.iter()
|
||||
.map(|&(_, y0, _, y1)| y0.min(y1))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if !image_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let below: Vec<TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| item.y < image_bottom && item.y >= image_bottom - 220.0)
|
||||
.cloned()
|
||||
.collect();
|
||||
let candidates: Vec<(f32, f32)> = group_rows(&below)
|
||||
.into_iter()
|
||||
.filter_map(|row| aligned_row_split(&row, x_min, x_max).map(|split| (split, row.y)))
|
||||
.collect();
|
||||
if candidates.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut clusters: Vec<Vec<(f32, f32)>> = Vec::new();
|
||||
for candidate in candidates {
|
||||
if let Some(cluster) = clusters.iter_mut().find(|cluster| {
|
||||
let mean = cluster.iter().map(|entry| entry.0).sum::<f32>() / cluster.len() as f32;
|
||||
(mean - candidate.0).abs() <= SPLIT_CLUSTER_TOLERANCE
|
||||
}) {
|
||||
cluster.push(candidate);
|
||||
} else {
|
||||
clusters.push(vec![candidate]);
|
||||
}
|
||||
}
|
||||
let dominant = clusters.into_iter().max_by_key(Vec::len)?;
|
||||
if dominant.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
let split_x = dominant.iter().map(|entry| entry.0).sum::<f32>() / dominant.len() as f32;
|
||||
let y_top = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
let image_gap = image_bottom - y_top;
|
||||
if !(60.0..=120.0).contains(&image_gap) {
|
||||
return None;
|
||||
}
|
||||
let y_bottom = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if y_top - y_bottom > 130.0 {
|
||||
return None;
|
||||
}
|
||||
log::debug!(
|
||||
"page {}: full-width image flow images={} aligned_rows={} split={:.1} page=[{:.1}..{:.1}] image_bottom={:.1} y=[{:.1}..{:.1}] full_width={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
images.len(),
|
||||
dominant.len(),
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
image_bottom,
|
||||
y_bottom,
|
||||
y_top,
|
||||
full_width_images
|
||||
);
|
||||
Some(ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
})
|
||||
}
|
||||
|
||||
fn paired_column_images(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
split_x: f32,
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
if split_x < x_min + page_width * 0.4 || split_x > x_min + page_width * 0.6 {
|
||||
return None;
|
||||
}
|
||||
let qualifying: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let image_left = x0.min(x1);
|
||||
let image_right = x0.max(x1);
|
||||
let confined_to_one_column = image_right <= split_x || image_left >= split_x;
|
||||
confined_to_one_column
|
||||
&& (x1 - x0).abs() >= MIN_IMAGE_WIDTH
|
||||
&& (y1 - y0).abs() >= MIN_IMAGE_HEIGHT
|
||||
})
|
||||
.collect();
|
||||
let wide_images: Vec<ImageRegion> = qualifying
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(x0, _, x1, _)| (x1 - x0).abs() >= page_width * 0.35)
|
||||
.collect();
|
||||
if qualifying.len() < 3 || wide_images.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
let has_left = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 < split_x);
|
||||
let has_right = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 >= split_x);
|
||||
if !has_left || !has_right {
|
||||
return None;
|
||||
}
|
||||
// A meaningful image-backed column flow spans multiple vertical panels.
|
||||
// Three same-row header/logo images can otherwise satisfy the image count
|
||||
// and send an ordinary asymmetric page through sequential column order.
|
||||
let image_y_min = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.min(region.3))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_y_max = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let has_vertical_stack = wide_images.iter().enumerate().any(|(index, left)| {
|
||||
wide_images.iter().skip(index + 1).any(|right| {
|
||||
let same_side =
|
||||
((left.0 + left.2) / 2.0 < split_x) == ((right.0 + right.2) / 2.0 < split_x);
|
||||
let left_center = (left.1 + left.3) / 2.0;
|
||||
let right_center = (right.1 + right.3) / 2.0;
|
||||
let left_height = (left.3 - left.1).abs();
|
||||
let right_height = (right.3 - right.1).abs();
|
||||
let vertical_gap = if left.1.max(left.3) < right.1.min(right.3) {
|
||||
right.1.min(right.3) - left.1.max(left.3)
|
||||
} else if right.1.max(right.3) < left.1.min(left.3) {
|
||||
left.1.min(left.3) - right.1.max(right.3)
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
same_side
|
||||
&& (left_center - right_center).abs() >= left_height.min(right_height) * 0.5
|
||||
&& vertical_gap <= left_height.max(right_height) * 0.5
|
||||
})
|
||||
});
|
||||
if image_y_max - image_y_min < page_width * 0.45 || !has_vertical_stack {
|
||||
return None;
|
||||
}
|
||||
let y_top = qualifying
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
// Only column-confined text proves the lower extent of the flow. A
|
||||
// spanning heading or caption below the columns must become the trailing
|
||||
// full-width node rather than stretching the column band to the page foot.
|
||||
let y_bottom = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
let item_right = item.x + effective_width(item);
|
||||
item.y <= y_top && (item_right <= split_x || item.x >= split_x)
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if !y_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
let distinct_rows = |right: bool| {
|
||||
let mut ys: Vec<f32> = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
item.y <= y_top && (item.x + effective_width(item) / 2.0 >= split_x) == right
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.collect();
|
||||
ys.sort_by(|left, right| left.total_cmp(right));
|
||||
ys.dedup_by(|left, right| (*left - *right).abs() <= 3.0);
|
||||
ys.len()
|
||||
};
|
||||
let left_rows = distinct_rows(false);
|
||||
let right_rows = distinct_rows(true);
|
||||
let line_balance = left_rows.min(right_rows) as f32 / left_rows.max(right_rows).max(1) as f32;
|
||||
(left_rows >= 5 && right_rows >= 5 && line_balance < 0.55).then(|| {
|
||||
log::debug!(
|
||||
"page {}: paired-image flow qualifying_images={} rows={}/{} split={:.1} page=[{:.1}..{:.1}] y=[{:.1}..{:.1}] images={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
qualifying.len(),
|
||||
left_rows,
|
||||
right_rows,
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
y_bottom,
|
||||
y_top,
|
||||
qualifying
|
||||
);
|
||||
ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn infer_image_anchored_flow(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
detected_split: Option<f32>,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
if items.is_empty() || images.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let (x_min, x_max) = page_x_bounds(items, images)?;
|
||||
detected_split
|
||||
.and_then(|split_x| paired_column_images(items, images, split_x, x_min, x_max))
|
||||
.or_else(|| local_flow_below_full_width_image(items, images, x_min, x_max))
|
||||
}
|
||||
|
||||
/// Partition a page into the topological order `above -> left -> right -> below`.
|
||||
/// These edges encode the reading-order DAG; empty nodes are omitted.
|
||||
pub(crate) fn build_region_graph(items: Vec<TextItem>, band: ColumnFlowBand) -> Vec<RegionNode> {
|
||||
let mut above = Vec::new();
|
||||
let mut left = Vec::new();
|
||||
let mut right = Vec::new();
|
||||
let mut below = Vec::new();
|
||||
for item in items {
|
||||
if item.y > band.y_top {
|
||||
above.push(item);
|
||||
} else if item.y < band.y_bottom {
|
||||
below.push(item);
|
||||
} else if item.x + effective_width(&item) / 2.0 < band.split_x {
|
||||
left.push(item);
|
||||
} else {
|
||||
right.push(item);
|
||||
}
|
||||
}
|
||||
let rtl = is_rtl_text(left.iter().chain(right.iter()).map(|item| &item.text));
|
||||
let mut ordered = vec![(RegionKind::FullWidth, above)];
|
||||
if rtl {
|
||||
ordered.push((RegionKind::Column, right));
|
||||
ordered.push((RegionKind::Column, left));
|
||||
} else {
|
||||
ordered.push((RegionKind::Column, left));
|
||||
ordered.push((RegionKind::Column, right));
|
||||
}
|
||||
ordered.push((RegionKind::FullWidth, below));
|
||||
ordered
|
||||
.into_iter()
|
||||
.filter_map(|(kind, items)| (!items.is_empty()).then_some(RegionNode { kind, items }))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.into(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 11.0,
|
||||
font: "F1".into(),
|
||||
font_size: 11.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_local_two_column_flow() {
|
||||
let mut items = vec![
|
||||
item("A full width caption", 55.0, 230.0, 430.0),
|
||||
item("A trailing full width heading", 55.0, 80.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
let band = infer_image_anchored_flow(&items, &images, None).unwrap();
|
||||
assert!((band.split_x - 272.5).abs() < 2.0);
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph.len(), 4);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "A trailing full width heading");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_cjk_column_flow() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("左栏这是没有空格的正文内容", 55.0, y, 210.0));
|
||||
items.push(item("右栏这是没有空格的正文内容", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_images_anchor_unbalanced_column_flows() {
|
||||
let mut items = vec![
|
||||
item("running header", 55.0, 700.0, 430.0),
|
||||
item("trailing full width caption", 55.0, 300.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
500.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 5..12 {
|
||||
items.push(item(
|
||||
"right continuation prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 530.0, 255.0, 680.0),
|
||||
(55.0, 380.0, 255.0, 530.0),
|
||||
(280.0, 560.0, 490.0, 680.0),
|
||||
];
|
||||
let band = infer_image_anchored_flow(&items, &images, Some(270.0)).unwrap();
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "trailing full width caption");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rtl_region_graph_reads_right_column_first() {
|
||||
let items = vec![
|
||||
item("A long English report header", 55.0, 250.0, 430.0),
|
||||
item("نص العمود الأيسر", 55.0, 150.0, 180.0),
|
||||
item("نص العمود الأيمن", 300.0, 150.0, 180.0),
|
||||
];
|
||||
let graph = build_region_graph(
|
||||
items,
|
||||
ColumnFlowBand {
|
||||
split_x: 270.0,
|
||||
y_bottom: 100.0,
|
||||
y_top: 200.0,
|
||||
},
|
||||
);
|
||||
|
||||
assert_eq!(graph.len(), 3);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert!(graph[1].items[0].x > graph[2].items[0].x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_header_logos_do_not_anchor_page_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 0..30 {
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 720.0, 205.0, 770.0),
|
||||
(60.0, 718.0, 210.0, 768.0),
|
||||
(280.0, 720.0, 450.0, 770.0),
|
||||
];
|
||||
assert!(infer_image_anchored_flow(&items, &images, Some(270.0)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_banner_does_not_anchor_local_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
let y = 270.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 310.0, 490.0, 550.0)];
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_none());
|
||||
}
|
||||
}
|
||||
@@ -162,7 +162,7 @@ fn extract_form_xobject_text_inner(
|
||||
|
||||
// Get fonts from the Form's Resources
|
||||
let form_fonts = get_form_fonts(doc, &stream.dict);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
|
||||
|
||||
// Build font width info for the form
|
||||
let font_widths = build_font_widths(doc, &form_fonts);
|
||||
|
||||
+6
-40
@@ -68,40 +68,6 @@ use text_quality::{
|
||||
};
|
||||
use tounicode::FontCMaps;
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
struct ProcessingTimer(std::time::Instant);
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
struct ProcessingTimer;
|
||||
|
||||
impl ProcessingTimer {
|
||||
fn start() -> Self {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
Self(std::time::Instant::now())
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
Self
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(&self) -> u64 {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
self.0.elapsed().as_millis() as u64
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
// The wasm32-unknown-unknown standard library has no clock.
|
||||
// Browser bindings measure with JavaScript's host clock.
|
||||
0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reason emitted when the extracted text layer appears garbled due to
|
||||
/// broken font decoding or mojibake.
|
||||
pub const OCR_REASON_SUSPECTED_GARBLED_TEXT: &str = "suspected_garbled_text";
|
||||
@@ -284,7 +250,7 @@ pub fn process_pdf_with_options<P: AsRef<Path>>(
|
||||
path: P,
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = ProcessingTimer::start();
|
||||
let start = std::time::Instant::now();
|
||||
validate_pdf_file(&path)?;
|
||||
|
||||
// Load the document once — shared by detection AND extraction.
|
||||
@@ -311,7 +277,7 @@ pub fn process_pdf_mem_with_options(
|
||||
buffer: &[u8],
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = ProcessingTimer::start();
|
||||
let start = std::time::Instant::now();
|
||||
validate_pdf_bytes(buffer)?;
|
||||
|
||||
let (doc, page_count) =
|
||||
@@ -3560,7 +3526,7 @@ fn process_document(
|
||||
doc: Document,
|
||||
page_count: u32,
|
||||
options: PdfOptions,
|
||||
start: ProcessingTimer,
|
||||
start: std::time::Instant,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
// Step 1 — Detection (cheap: scans content streams for text operators)
|
||||
let detection = detector::detect_from_document(&doc, page_count, &options.detection)?;
|
||||
@@ -3576,7 +3542,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3592,7 +3558,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3858,7 +3824,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: {
|
||||
// Detector reasons (scanned / no_text / vector_text / garbled) merged
|
||||
|
||||
+2
-11
@@ -1012,19 +1012,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Separate images and links from text items
|
||||
let mut images: Vec<TextItem> = Vec::new();
|
||||
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
let mut links: Vec<TextItem> = Vec::new();
|
||||
let mut text_items: Vec<TextItem> = Vec::new();
|
||||
|
||||
for item in items {
|
||||
match &item.item_type {
|
||||
ItemType::Image => {
|
||||
page_image_regions.entry(item.page).or_default().push((
|
||||
item.x,
|
||||
item.y,
|
||||
item.x + item.width,
|
||||
item.y + item.height,
|
||||
));
|
||||
if options.include_images {
|
||||
images.push(item);
|
||||
}
|
||||
@@ -1662,12 +1655,11 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// items from different side-by-side zones (e.g. left/right month columns
|
||||
// in a calendar) don't merge into the same line.
|
||||
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
|
||||
crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
non_table_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
)
|
||||
} else {
|
||||
// Separate items into physical-band pages, chart/prose pages, and
|
||||
@@ -1690,12 +1682,11 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
// Process unsplit pages normally
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
// Process each split page's bands independently, then interleave
|
||||
// by Y position so paired zones (e.g. left/right months) appear together.
|
||||
|
||||
+10
-23
@@ -4,17 +4,11 @@
|
||||
|
||||
use log::{debug, warn};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::borrow::Cow;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::glyph_names::glyph_to_char;
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
static BUILTIN_CMAPS: include_dir::Dir<'_> =
|
||||
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
|
||||
|
||||
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ToUnicodeCMap {
|
||||
@@ -1155,7 +1149,9 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
|
||||
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
|
||||
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
|
||||
let data = read_builtin_cmap_file(&name)?;
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(name);
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -1163,14 +1159,13 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
cmap.code_byte_length = 2;
|
||||
debug!(
|
||||
"Built-in CMap {}: char_map={} ranges={}",
|
||||
name,
|
||||
path.display(),
|
||||
cmap.char_map.len(),
|
||||
cmap.ranges.len()
|
||||
);
|
||||
Some(cmap)
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
|
||||
let p = PathBuf::from(dir);
|
||||
@@ -1187,18 +1182,6 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let path = find_bcmaps_dir()?.join(name);
|
||||
std::fs::read(path).ok().map(Cow::Owned)
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let file = BUILTIN_CMAPS.get_file(name)?;
|
||||
Some(Cow::Borrowed(file.contents()))
|
||||
}
|
||||
|
||||
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
|
||||
let mut stream = BinaryCMapStream::new(data);
|
||||
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
|
||||
@@ -1509,7 +1492,9 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
|
||||
}
|
||||
|
||||
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
parse_binary_cmap_encoding(&data).ok()
|
||||
}
|
||||
|
||||
@@ -1794,7 +1779,9 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
|
||||
if !name.ends_with("UCS2") {
|
||||
return None;
|
||||
}
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
|
||||
Generated
-1304
File diff suppressed because it is too large
Load Diff
@@ -1,37 +0,0 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.2"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "README.md"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[dependencies]
|
||||
console_error_panic_hook = "0.1"
|
||||
js-sys = "0.3"
|
||||
pdf-inspector = { path = ".." }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde-wasm-bindgen = "0.6"
|
||||
wasm-bindgen = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
wasm-bindgen-test = "0.3"
|
||||
|
||||
[profile.release]
|
||||
codegen-units = 1
|
||||
lto = true
|
||||
opt-level = "s"
|
||||
strip = true
|
||||
|
||||
[package.metadata.wasm-pack.profile.release]
|
||||
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
|
||||
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
|
||||
# size, and LTO optimizations above.
|
||||
wasm-opt = false
|
||||
@@ -1,58 +0,0 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
Third-party notices
|
||||
===================
|
||||
|
||||
Adobe CMaps
|
||||
-----------
|
||||
|
||||
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
|
||||
|
||||
Copyright 1990-2009 Adobe Systems Incorporated.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
Neither the name of Adobe Systems Incorporated nor the names of its
|
||||
contributors may be used to endorse or promote products derived from this
|
||||
software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -1,60 +0,0 @@
|
||||
# @firecrawl/pdf-inspector-wasm
|
||||
|
||||
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```ts
|
||||
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
|
||||
|
||||
await init();
|
||||
|
||||
const response = await fetch("/annual-report.pdf");
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
Pass options when you need selected pages or compact Markdown:
|
||||
|
||||
```ts
|
||||
const result = processPdf(pdf, {
|
||||
pages: [1, 3, 5],
|
||||
profile: "compact",
|
||||
includePageMarkers: true,
|
||||
});
|
||||
```
|
||||
|
||||
The package also exports:
|
||||
|
||||
- `detectPdf(pdf, options?)` for detection without extraction.
|
||||
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
|
||||
- `extractText(pdf)` for plain text.
|
||||
- `version()` for the WASM package version.
|
||||
|
||||
## Browser behavior
|
||||
|
||||
- Parsing runs locally. PDF bytes are not uploaded anywhere.
|
||||
- The build is single-threaded and does not require cross-origin isolation.
|
||||
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
|
||||
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
|
||||
- Image-only documents still require a separate OCR step.
|
||||
|
||||
## Build from source
|
||||
|
||||
```bash
|
||||
cargo install wasm-pack --version 0.15.0 --locked
|
||||
wasm-pack build wasm --target web --scope firecrawl --release
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
-440
@@ -1,440 +0,0 @@
|
||||
use pdf_inspector::{
|
||||
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
|
||||
ProcessMode,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wasm_bindgen::prelude::*;
|
||||
|
||||
#[wasm_bindgen(typescript_custom_section)]
|
||||
const TYPESCRIPT_TYPES: &str = r#"
|
||||
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
|
||||
export type MarkdownProfile = "fidelity" | "compact";
|
||||
|
||||
export interface ProcessOptions {
|
||||
/** Restrict extraction to these 1-indexed page numbers. */
|
||||
pages?: number[];
|
||||
/** Password for an encrypted PDF. */
|
||||
password?: string;
|
||||
/** Source-faithful output by default, or compact output for fewer tokens. */
|
||||
profile?: MarkdownProfile;
|
||||
/** Insert `<!-- Page N -->` markers between pages. */
|
||||
includePageMarkers?: boolean;
|
||||
/** Include image placeholders in Markdown output. */
|
||||
includeImages?: boolean;
|
||||
}
|
||||
|
||||
export interface PageOcrReasons {
|
||||
/** 1-indexed page number. */
|
||||
page: number;
|
||||
reasons: string[];
|
||||
}
|
||||
|
||||
export interface LayoutComplexity {
|
||||
isComplex: boolean;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithTables: number[];
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithColumns: number[];
|
||||
}
|
||||
|
||||
export interface PdfProcessResult {
|
||||
pdfType: PdfType;
|
||||
markdown?: string;
|
||||
pageCount: number;
|
||||
processingTimeMs: number;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesNeedingOcr: number[];
|
||||
ocrReasonsByPage: PageOcrReasons[];
|
||||
title?: string;
|
||||
confidence: number;
|
||||
layout: LayoutComplexity;
|
||||
hasEncodingIssues: boolean;
|
||||
}
|
||||
|
||||
export interface PdfClassification {
|
||||
pdfType: PdfType;
|
||||
pageCount: number;
|
||||
/** 0-indexed page numbers, matching the native Node.js API. */
|
||||
pagesNeedingOcr: number[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
|
||||
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
|
||||
export function classifyPdf(data: Uint8Array): PdfClassification;
|
||||
export function extractText(data: Uint8Array): string;
|
||||
export function version(): string;
|
||||
"#;
|
||||
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
|
||||
struct WasmProcessOptions {
|
||||
pages: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
profile: Option<WasmMarkdownProfile>,
|
||||
include_page_markers: Option<bool>,
|
||||
include_images: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
enum WasmMarkdownProfile {
|
||||
Fidelity,
|
||||
Compact,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPageOcrReasons {
|
||||
page: u32,
|
||||
reasons: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<PageOcrReasons> for WasmPageOcrReasons {
|
||||
fn from(value: PageOcrReasons) -> Self {
|
||||
Self {
|
||||
page: value.page,
|
||||
reasons: value.reasons,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmLayoutComplexity {
|
||||
is_complex: bool,
|
||||
pages_with_tables: Vec<u32>,
|
||||
pages_with_columns: Vec<u32>,
|
||||
}
|
||||
|
||||
impl From<LayoutComplexity> for WasmLayoutComplexity {
|
||||
fn from(value: LayoutComplexity) -> Self {
|
||||
Self {
|
||||
is_complex: value.is_complex,
|
||||
pages_with_tables: value.pages_with_tables,
|
||||
pages_with_columns: value.pages_with_columns,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfProcessResult {
|
||||
pdf_type: &'static str,
|
||||
markdown: Option<String>,
|
||||
page_count: u32,
|
||||
processing_time_ms: f64,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
|
||||
title: Option<String>,
|
||||
confidence: f64,
|
||||
layout: WasmLayoutComplexity,
|
||||
has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
impl From<PdfProcessResult> for WasmPdfProcessResult {
|
||||
fn from(value: PdfProcessResult) -> Self {
|
||||
Self {
|
||||
pdf_type: pdf_type_name(value.pdf_type),
|
||||
markdown: value.markdown,
|
||||
page_count: value.page_count,
|
||||
processing_time_ms: value.processing_time_ms as f64,
|
||||
pages_needing_ocr: value.pages_needing_ocr,
|
||||
ocr_reasons_by_page: value
|
||||
.ocr_reasons_by_page
|
||||
.into_iter()
|
||||
.map(Into::into)
|
||||
.collect(),
|
||||
title: value.title,
|
||||
confidence: value.confidence as f64,
|
||||
layout: value.layout.into(),
|
||||
has_encoding_issues: value.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfClassification {
|
||||
pdf_type: &'static str,
|
||||
page_count: u32,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
confidence: f64,
|
||||
}
|
||||
|
||||
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
|
||||
match pdf_type {
|
||||
PdfType::TextBased => "TextBased",
|
||||
PdfType::Scanned => "Scanned",
|
||||
PdfType::ImageBased => "ImageBased",
|
||||
PdfType::Mixed => "Mixed",
|
||||
}
|
||||
}
|
||||
|
||||
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
|
||||
js_sys::Error::new(&format!("{context}: {error}")).into()
|
||||
}
|
||||
|
||||
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
|
||||
if value.is_undefined() || value.is_null() {
|
||||
return Ok(WasmProcessOptions::default());
|
||||
}
|
||||
|
||||
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
|
||||
}
|
||||
|
||||
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
|
||||
let options = deserialize_options(value)?;
|
||||
if options
|
||||
.pages
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(js_error(
|
||||
"invalid options",
|
||||
"pages are 1-indexed; page 0 is invalid",
|
||||
));
|
||||
}
|
||||
|
||||
let mut result = PdfOptions::new().mode(mode);
|
||||
if let Some(pages) = options.pages {
|
||||
result = result.pages(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(profile) = options.profile {
|
||||
result.markdown.profile = match profile {
|
||||
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
|
||||
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
|
||||
};
|
||||
}
|
||||
if let Some(include_page_markers) = options.include_page_markers {
|
||||
result.markdown.include_page_numbers = include_page_markers;
|
||||
}
|
||||
if let Some(include_images) = options.include_images {
|
||||
result.markdown.include_images = include_images;
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
|
||||
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
|
||||
}
|
||||
|
||||
fn initialize() {
|
||||
console_error_panic_hook::set_once();
|
||||
}
|
||||
|
||||
/// Process PDF bytes entirely inside WebAssembly.
|
||||
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
|
||||
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::Full)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("process PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Classify PDF bytes without extracting text or producing Markdown.
|
||||
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
|
||||
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::DetectOnly)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("detect PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Return the lightweight classification shape used by the native Node API.
|
||||
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
|
||||
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
|
||||
serialize(&WasmPdfClassification {
|
||||
pdf_type: pdf_type_name(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes without Markdown conversion.
|
||||
#[wasm_bindgen(js_name = extractText, skip_typescript)]
|
||||
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
|
||||
initialize();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
|
||||
.map_err(|error| js_error("extract text", error))?;
|
||||
Ok(
|
||||
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
|
||||
.into_iter()
|
||||
.map(|line| line.text())
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n"),
|
||||
)
|
||||
}
|
||||
|
||||
/// Return the WebAssembly package version.
|
||||
#[wasm_bindgen(skip_typescript)]
|
||||
pub fn version() -> String {
|
||||
env!("CARGO_PKG_VERSION").to_string()
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_arch = "wasm32"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use js_sys::Reflect;
|
||||
use wasm_bindgen_test::*;
|
||||
|
||||
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
|
||||
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
|
||||
|
||||
fn synthetic_korea1_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
|
||||
// There is deliberately no ToUnicode stream: decoding must use the
|
||||
// embedded predefined CMap rather than lopdf's plain-text fallback.
|
||||
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
|
||||
// the bottom edge so they look exactly like a numeric page footer.
|
||||
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn processes_pdf_to_markdown() {
|
||||
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
|
||||
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn rejects_non_pdf_bytes() {
|
||||
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn classifies_and_extracts_plain_text() {
|
||||
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
|
||||
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let text = extract_text(TEXT_PDF).expect("extract text");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!text.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn extracts_cjk_and_preserves_numeric_page_footer() {
|
||||
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
|
||||
|
||||
assert_eq!(text, "가\n42");
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn opens_encrypted_pdf_with_password() {
|
||||
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
|
||||
|
||||
let options = js_sys::Object::new();
|
||||
Reflect::set(
|
||||
&options,
|
||||
&JsValue::from_str("password"),
|
||||
&JsValue::from_str("secret123"),
|
||||
)
|
||||
.expect("set password");
|
||||
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user