Compare commits

..
31 changed files with 428 additions and 5237 deletions
-40
View File
@@ -25,9 +25,6 @@ jobs:
- name: Run tests
run: cargo test --verbose
- name: Test developer scripts
run: python3 -m unittest discover -s scripts/tests
fmt:
name: Format
runs-on: ubuntu-latest
@@ -42,9 +39,6 @@ jobs:
- name: Check formatting
run: cargo fmt --all -- --check
- name: Check WASM formatting
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
clippy:
name: Clippy
runs-on: ubuntu-latest
@@ -83,37 +77,3 @@ jobs:
- name: Build
run: cargo build --release --verbose
wasm:
name: WebAssembly
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
with:
targets: wasm32-unknown-unknown
components: clippy
- name: Cache cargo
uses: Swatinem/rust-cache@v2
with:
workspaces: |
wasm -> target
key: wasm
- name: Check WebAssembly bindings
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
- name: Check root package for WebAssembly
run: cargo check --target wasm32-unknown-unknown
- name: Lint WebAssembly bindings
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
- name: Install wasm-pack
run: cargo install wasm-pack --version 0.15.0 --locked
- name: Test WebAssembly package
run: wasm-pack test --node --release wasm
-111
View File
@@ -1,111 +0,0 @@
name: Publish WebAssembly package
on:
push:
branches: [main]
paths: ['wasm/Cargo.toml']
workflow_dispatch:
permissions:
contents: read
id-token: write
env:
CARGO_TERM_COLOR: always
jobs:
check-version:
name: Check version change
if: github.ref == 'refs/heads/main'
runs-on: ubuntu-latest
outputs:
changed: ${{ steps.check.outputs.changed }}
package_exists: ${{ steps.check.outputs.package_exists }}
published: ${{ steps.check.outputs.published }}
version: ${{ steps.check.outputs.version }}
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 2
- name: Check package version
id: check
run: |
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
echo "changed=true" >> "$GITHUB_OUTPUT"
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "published=false" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "changed=true" >> "$GITHUB_OUTPUT"
else
echo "changed=true" >> "$GITHUB_OUTPUT"
fi
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
echo "package_exists=false" >> "$GITHUB_OUTPUT"
echo "published=false" >> "$GITHUB_OUTPUT"
echo "The initial package must be published once before trusted publishing can be configured."
exit 0
fi
echo "package_exists=true" >> "$GITHUB_OUTPUT"
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
echo "published=true" >> "$GITHUB_OUTPUT"
else
echo "published=false" >> "$GITHUB_OUTPUT"
fi
publish:
name: Build and publish
needs: check-version
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
runs-on: ubuntu-latest
permissions:
contents: read
id-token: write
steps:
- uses: actions/checkout@v4
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
with:
targets: wasm32-unknown-unknown
- uses: actions/setup-node@v6
with:
node-version: '24'
registry-url: 'https://registry.npmjs.org'
- name: Install wasm-pack
run: cargo install wasm-pack --version 0.15.0 --locked
- name: Build browser package
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
- name: Prepare package metadata
run: |
node -e '
const fs = require("fs")
const path = "wasm/pkg/package.json"
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
pkg.name = "@firecrawl/pdf-inspector-wasm"
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
pkg.publishConfig = { access: "public" }
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
'
- name: Inspect package contents
run: npm pack --dry-run ./wasm/pkg
- name: Publish package
run: npm publish ./wasm/pkg --provenance --access public
+1 -3
View File
@@ -1,13 +1,10 @@
# Rust build artifacts
/target/
/wasm/target/
/wasm/pkg/
debug/
*.pdb
# Cargo lock (optional for libraries)
Cargo.lock
!/wasm/Cargo.lock
# IDE
.idea/
@@ -42,3 +39,4 @@ test_output/
__pycache__/
*.pyc
.pytest_cache/
+12 -18
View File
@@ -12,13 +12,13 @@ readme = "docs/rust-api.md"
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
# loads it at runtime relative to CARGO_MANIFEST_DIR.
include = [
"/src/**",
"/external/bcmaps/**",
"/docs/rust-api.md",
"/LICENSE",
"src/**",
"external/bcmaps/**",
"docs/rust-api.md",
"LICENSE",
# maturin derives the sdist file list from this allowlist; the stub must
# ship so wheels built from the sdist keep their type hints.
"/pdf_inspector.pyi",
"pdf_inspector.pyi",
]
[lib]
@@ -29,11 +29,18 @@ crate-type = ["lib", "cdylib"]
# Python bindings
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
# PDF parsing
lopdf = { version = "0.41.0", features = ["rayon"] }
# Error handling
thiserror = "2.0"
# Parallel processing
rayon = "1.10"
# Logging
log = "0.4"
env_logger = "0.11"
# Text processing
regex = "1.10"
@@ -43,19 +50,6 @@ unicode-normalization = "0.1"
# TrueType font parsing (for Identity-H CID font cmap extraction)
ttf-parser = "0.25"
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
# deliberately single-threaded so it works without cross-origin isolation.
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
lopdf = { version = "0.41.0", features = ["rayon"] }
rayon = "1.10"
env_logger = "0.11"
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
# bundled CMaps because there is no filesystem at runtime.
[target.'cfg(target_arch = "wasm32")'.dependencies]
lopdf = { version = "0.41.0", default-features = false, features = ["wasm_js"] }
include_dir = "0.7"
[dev-dependencies]
tempfile = "3.3"
+9 -34
View File
@@ -5,7 +5,7 @@
[![PyPI](https://img.shields.io/pypi/v/pdf-inspector.svg)](https://pypi.org/project/pdf-inspector/)
[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
@@ -19,28 +19,24 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
## Benchmark
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only local engines without model-based PDF parsing are shown; OCR was disabled. Scores are 0-1, higher is better.
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|---|---|---|---|---|---|
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
| pdf-inspector | 0.83 | 0.89 | 0.66 | 0.74 | 4s |
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
Results were refreshed on July 16, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Speed is the median of three complete corpus runs.
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the low end of that range without any OCR, in 4 seconds.
For context, engines that use OCR or model-based document parsing (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the top of that range without either, in 2.8 seconds.
**Where we do well:** Speed (fastest of all engines), the best table detection of any engine shown, and heading detection now on par with opendataloader. Overall lands within 0.01 of opendataloader at roughly 2.5× the speed.
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. pdf-inspector delivered the highest overall, reading-order, and table scores, along with the fastest complete run in this benchmark. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
**Where we lag:** Reading order still trails opendataloader slightly, and table structure trails OCR-based engines that can see visual layout.
## Quick start
@@ -78,26 +74,6 @@ console.log(result.markdown); // Markdown string or null
> Full API reference: [napi/README.md](napi/README.md)
### Browser WebAssembly
```bash
npm install @firecrawl/pdf-inspector-wasm
```
```javascript
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
await init();
const response = await fetch('/document.pdf');
const pdf = new Uint8Array(await response.arrayBuffer());
const result = processPdf(pdf);
console.log(result.pdfType);
console.log(result.markdown);
```
> Full API reference: [wasm/README.md](wasm/README.md)
### Rust
Install from [crates.io](https://crates.io/crates/pdf-inspector):
@@ -209,7 +185,6 @@ src/
markdown/ — Markdown conversion and structure detection
bin/ — CLI tools (pdf2md, detect_pdf)
napi/ — Node.js/Bun bindings (napi-rs)
wasm/ — Browser bindings (wasm-bindgen)
```
## How classification works
-59
View File
@@ -1,59 +0,0 @@
# Benchmarking against OpenDataLoader
The paired harness runs two `pdf2md` binaries through the same local
OpenDataLoader corpus, evaluates both outputs, and reports aggregate and
per-document deltas. This avoids comparing results produced from different
corpus revisions or evaluator versions.
Build a candidate and provide a released or worktree build as the baseline:
```bash
cargo build --release
python3 scripts/bench_opendataloader.py \
--bench-dir ../opendataloader-bench \
--baseline ../pdf-inspector-main/target/release/pdf2md \
--candidate target/release/pdf2md \
--max-document-regression 0.02 \
--json-output /tmp/pdf-inspector-benchmark.json
```
Pass `--reference-evaluation path/to/evaluation.json` to report the candidate
delta against another evaluation, and add `--require-reference-lead` to make a
negative reference delta fail the run. By default, the candidate must not
regress the baseline overall score or introduce missing predictions. Use
`--min-overall-delta` to require a specific aggregate gain.
The OpenDataLoader repository is external and keeps its normal
`prediction/pdf-inspector` output. Paired evaluation copies each run into a
temporary directory before evaluating it, so the baseline and candidate cannot
overwrite one another.
## Published comparison protocol
The public benchmark table was refreshed on July 16, 2026, on an Apple M4 Pro
using pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1,
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Every engine processed the same 200
PDFs with OCR disabled. Reported speed is the median of three complete corpus
runs; quality scores come from the benchmark evaluator over all 200 outputs.
## Optional backend evidence probe
The evidence probe compares positioned `pdf2md` items with MuPDF structured
text on the same pages. It is intended to find deterministic extraction or
layout evidence that could justify a future native implementation; it does not
merge MuPDF output into Markdown, invoke OCR, or add a runtime dependency.
Install MuPDF's `mutool`, build `pdf2md`, then run:
```bash
python3 scripts/probe_backend_evidence.py document.pdf \
--pdf2md target/release/pdf2md \
--json-output /tmp/backend-evidence.json
```
The report flags pages when MuPDF exposes a material net token gain, repeated
alignment anchors absent from local evidence, or additional image blocks. The
JSON includes bounded token samples and page-level counts so promising cases
can be inspected without treating backend disagreement as automatically
correct. Thresholds are configurable with `--min-token-gain`,
`--min-alternate-only-ratio`, and `--min-anchor-gain`.
-17
View File
@@ -19,20 +19,3 @@ The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
## Browser WebAssembly package
The browser package is published as `@firecrawl/pdf-inspector-wasm`. Its version lives in `wasm/Cargo.toml`, and `.github/workflows/publish-wasm.yml` builds the `web` target with `wasm-pack` before publishing the generated package.
The npm package must exist before a trusted publisher can be configured. For the first release only:
1. Build with `wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release`.
2. Inspect with `npm pack --dry-run ./wasm/pkg`.
3. Publish with `npm publish ./wasm/pkg --access public` from an authorized maintainer session.
4. In the package settings on npm, configure the GitHub Actions trusted publisher:
- Organization: `firecrawl`
- Repository: `pdf-inspector`
- Workflow: `publish-wasm.yml`
- Allowed action: `npm publish`
After that one-time bootstrap, bumping the version in `wasm/Cargo.toml` and merging it to `main` publishes through OIDC. Until the package exists, the workflow exits cleanly without attempting an unauthenticated first publish. See npm's [trusted publishing documentation](https://docs.npmjs.com/trusted-publishers/) for the registry-side setup.
+5 -7
View File
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
## Benchmark
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 01, higher is better:
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 01, higher is better:
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|---|---|---|---|---|---|
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
OCR/ML engines (docling, marker, mineru) score 0.830.88 overall but take 2180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
## Install
+5 -7
View File
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
## Benchmark
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 01, higher is better:
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 01, higher is better:
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|---|---|---|---|---|---|
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
OCR/ML engines (docling, marker, mineru) score 0.830.88 overall but take 2180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
## Install
+5 -7
View File
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
## Benchmark
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 01, higher is better:
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 01, higher is better:
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|---|---|---|---|---|---|
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
OCR/ML engines (docling, marker, mineru) score 0.830.88 overall but take 2180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
## Install
-351
View File
@@ -1,351 +0,0 @@
#!/usr/bin/env python3
"""Run a paired pdf-inspector OpenDataLoader benchmark and report deltas."""
from __future__ import annotations
import argparse
import json
import math
import os
import shutil
import subprocess
import sys
import tempfile
from pathlib import Path
from typing import Any
SCORE_KEYS = (
"overall_mean",
"nid_mean",
"nid_s_mean",
"teds_mean",
"teds_s_mean",
"mhs_mean",
"mhs_s_mean",
)
def _non_negative_int(value: str) -> int:
parsed = int(value)
if parsed < 0:
raise argparse.ArgumentTypeError("must be non-negative")
return parsed
def _non_negative_float(value: str) -> float:
parsed = float(value)
if not math.isfinite(parsed) or parsed < 0.0:
raise argparse.ArgumentTypeError("must be finite and non-negative")
return parsed
def _finite_float(value: str) -> float:
parsed = float(value)
if not math.isfinite(parsed):
raise argparse.ArgumentTypeError("must be finite")
return parsed
def _scores(evaluation: dict[str, Any]) -> dict[str, float]:
score = evaluation.get("metrics", {}).get("score", {})
return {key: float(score[key]) for key in SCORE_KEYS if score.get(key) is not None}
def _documents(evaluation: dict[str, Any]) -> dict[str, float]:
documents: dict[str, float] = {}
for document in evaluation.get("documents", []):
overall = document.get("scores", {}).get("overall")
if overall is not None:
documents[str(document["document_id"])] = float(overall)
return documents
def compare_evaluations(
baseline: dict[str, Any],
candidate: dict[str, Any],
reference: dict[str, Any] | None = None,
*,
top: int = 10,
) -> dict[str, Any]:
"""Build aggregate and per-document deltas from evaluator JSON payloads."""
baseline_scores = _scores(baseline)
candidate_scores = _scores(candidate)
metric_deltas = {
key: candidate_scores[key] - baseline_scores[key]
for key in SCORE_KEYS
if key in baseline_scores and key in candidate_scores
}
baseline_documents = _documents(baseline)
candidate_documents = _documents(candidate)
shared = sorted(baseline_documents.keys() & candidate_documents.keys())
document_deltas = [
{
"document_id": document_id,
"baseline": baseline_documents[document_id],
"candidate": candidate_documents[document_id],
"delta": candidate_documents[document_id] - baseline_documents[document_id],
}
for document_id in shared
]
epsilon = 1e-12
improvements = sorted(document_deltas, key=lambda item: item["delta"], reverse=True)
regressions = sorted(document_deltas, key=lambda item: item["delta"])
result: dict[str, Any] = {
"baseline": baseline_scores,
"candidate": candidate_scores,
"deltas": metric_deltas,
"missing_predictions": {
"baseline": int(baseline.get("metrics", {}).get("missing_predictions", 0)),
"candidate": int(candidate.get("metrics", {}).get("missing_predictions", 0)),
},
"documents": {
"shared": len(shared),
"improved": sum(item["delta"] > epsilon for item in document_deltas),
"regressed": sum(item["delta"] < -epsilon for item in document_deltas),
"unchanged": sum(abs(item["delta"]) <= epsilon for item in document_deltas),
"largest_improvements": [
item for item in improvements if item["delta"] > epsilon
][:top],
"largest_regressions": [
item for item in regressions if item["delta"] < -epsilon
][:top],
"worst_regression": next(
(item for item in regressions if item["delta"] < -epsilon), None
),
},
}
if reference is not None:
reference_scores = _scores(reference)
result["reference"] = reference_scores
result["candidate_vs_reference"] = {
key: candidate_scores[key] - reference_scores[key]
for key in SCORE_KEYS
if key in candidate_scores and key in reference_scores
}
return result
def evaluate_gates(
comparison: dict[str, Any],
*,
min_overall_delta: float,
max_document_regression: float | None,
max_missing: int,
require_reference_lead: bool,
) -> list[str]:
"""Return human-readable gate failures; an empty list means pass."""
failures: list[str] = []
overall_delta = comparison["deltas"].get("overall_mean")
if overall_delta is None or overall_delta < min_overall_delta:
failures.append(
f"overall delta {overall_delta!r} is below {min_overall_delta:+.6f}"
)
candidate_missing = comparison["missing_predictions"]["candidate"]
if candidate_missing > max_missing:
failures.append(
f"candidate has {candidate_missing} missing predictions (maximum {max_missing})"
)
if max_document_regression is not None:
regression = comparison["documents"].get("worst_regression")
if regression is not None and regression["delta"] < -max_document_regression:
failures.append(
"largest document regression "
f"{regression['document_id']}={regression['delta']:+.6f} "
f"exceeds {-max_document_regression:+.6f}"
)
if require_reference_lead:
reference_delta = comparison.get("candidate_vs_reference", {}).get("overall_mean")
if reference_delta is None:
failures.append("reference overall score is unavailable")
elif reference_delta < 0.0:
failures.append(
f"candidate trails reference overall by {reference_delta!r}"
)
return failures
def _run(command: list[str], *, cwd: Path, env: dict[str, str] | None = None) -> None:
print("+", " ".join(command), flush=True)
subprocess.run(command, cwd=cwd, env=env, check=True)
def _run_engine(
*,
bench_dir: Path,
python: Path,
binary: Path,
label: str,
scratch_root: Path,
) -> dict[str, Any]:
env = os.environ.copy()
env["PDF_INSPECTOR_BINARY"] = str(binary)
source = bench_dir / "prediction" / "pdf-inspector"
if source.exists():
if source.is_dir():
shutil.rmtree(source)
else:
source.unlink()
_run(
[
str(python),
"src/pdf_parser.py",
"--engine",
"pdf-inspector",
"--log-level",
"WARNING",
],
cwd=bench_dir,
env=env,
)
if not source.is_dir():
raise RuntimeError(f"parser did not produce predictions: {source}")
destination = scratch_root / label
shutil.copytree(source, destination)
_run(
[
str(python),
"src/evaluator.py",
"--prediction-root",
str(scratch_root),
"--engine",
label,
"--log-level",
"WARNING",
],
cwd=bench_dir,
)
with (destination / "evaluation.json").open(encoding="utf-8") as handle:
return json.load(handle)
def _print_report(comparison: dict[str, Any]) -> None:
print("\nMetric baseline candidate delta")
print("-------------------- ---------- ---------- ----------")
for key in SCORE_KEYS:
if key not in comparison["deltas"]:
continue
print(
f"{key:<20} {comparison['baseline'][key]:>10.6f} "
f"{comparison['candidate'][key]:>10.6f} "
f"{comparison['deltas'][key]:>+10.6f}"
)
if "reference" in comparison:
delta = comparison["candidate_vs_reference"].get("overall_mean")
reference = comparison["reference"].get("overall_mean")
reference_display = f"{reference:.6f}" if reference is not None else "n/a"
delta_display = f"{delta:+.6f}" if delta is not None else "n/a"
print(f"\nReference overall: {reference_display}; candidate delta: {delta_display}")
documents = comparison["documents"]
print(
"\nDocuments: "
f"{documents['improved']} improved, {documents['regressed']} regressed, "
f"{documents['unchanged']} unchanged ({documents['shared']} shared)"
)
for heading, key in (
("Largest improvements", "largest_improvements"),
("Largest regressions", "largest_regressions"),
):
print(f"\n{heading}:")
rows = documents[key]
if not rows:
print(" none")
for row in rows:
print(
f" {row['document_id']}: {row['delta']:+.6f} "
f"({row['baseline']:.6f} -> {row['candidate']:.6f})"
)
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--bench-dir", type=Path, required=True)
parser.add_argument("--baseline", type=Path, required=True)
parser.add_argument("--candidate", type=Path, required=True)
parser.add_argument("--python", type=Path)
parser.add_argument("--reference-evaluation", type=Path)
parser.add_argument("--json-output", type=Path)
parser.add_argument("--top", type=_non_negative_int, default=10)
parser.add_argument("--min-overall-delta", type=_finite_float, default=0.0)
parser.add_argument("--max-document-regression", type=_non_negative_float)
parser.add_argument("--max-missing", type=_non_negative_int, default=0)
parser.add_argument("--require-reference-lead", action="store_true")
return parser.parse_args(argv)
def main(argv: list[str] | None = None) -> int:
args = _arguments(argv)
bench_dir = args.bench_dir.resolve()
baseline = args.baseline.resolve()
candidate = args.candidate.resolve()
# Keep the virtualenv launcher path intact. Resolving its symlink would
# invoke the underlying system interpreter without the benchmark's site
# packages.
python = (args.python or bench_dir / ".venv" / "bin" / "python").absolute()
for path, description in (
(bench_dir / "src" / "pdf_parser.py", "OpenDataLoader parser"),
(bench_dir / "src" / "evaluator.py", "OpenDataLoader evaluator"),
(baseline, "baseline binary"),
(candidate, "candidate binary"),
(python, "Python interpreter"),
):
if not path.exists():
raise SystemExit(f"{description} not found: {path}")
with tempfile.TemporaryDirectory(prefix="pdf-inspector-opendataloader-") as temporary:
scratch_root = Path(temporary)
baseline_evaluation = _run_engine(
bench_dir=bench_dir,
python=python,
binary=baseline,
label="baseline",
scratch_root=scratch_root,
)
candidate_evaluation = _run_engine(
bench_dir=bench_dir,
python=python,
binary=candidate,
label="candidate",
scratch_root=scratch_root,
)
reference = None
if args.reference_evaluation is not None:
with args.reference_evaluation.resolve().open(encoding="utf-8") as handle:
reference = json.load(handle)
comparison = compare_evaluations(
baseline_evaluation,
candidate_evaluation,
reference,
top=args.top,
)
_print_report(comparison)
if args.json_output is not None:
args.json_output.resolve().write_text(
json.dumps(comparison, indent=2) + "\n", encoding="utf-8"
)
failures = evaluate_gates(
comparison,
min_overall_delta=args.min_overall_delta,
max_document_regression=args.max_document_regression,
max_missing=args.max_missing,
require_reference_lead=args.require_reference_lead,
)
if failures:
print("\nBenchmark gate failed:", file=sys.stderr)
for failure in failures:
print(f" - {failure}", file=sys.stderr)
return 1
print("\nBenchmark gate passed.")
return 0
if __name__ == "__main__":
raise SystemExit(main())
-351
View File
@@ -1,351 +0,0 @@
#!/usr/bin/env python3
"""Compare pdf-inspector evidence with optional MuPDF structured text.
This is an experiment and diagnostic tool, not an extraction fallback. It runs
MuPDF's deterministic ``stext.json`` backend without OCR and highlights pages
where that backend exposes materially different text or layout evidence.
"""
from __future__ import annotations
import argparse
from collections import Counter
import json
from pathlib import Path
import re
import shutil
import subprocess
import sys
from typing import Any, Iterable
TOKEN_PATTERN = re.compile(r"[^\W_]+(?:[\u2019'][^\W_]+)*", re.UNICODE)
def _tokens(texts: Iterable[str]) -> Counter[str]:
tokens: Counter[str] = Counter()
for text in texts:
for token in TOKEN_PATTERN.findall(text.casefold()):
# Lone letters are frequently bullets, chart labels, or fragmented
# glyphs. Digits remain useful even when they are one character.
if len(token) > 1 or token.isdigit():
tokens[token] += 1
return tokens
def _repeated_x_anchors(xs: Iterable[float], *, tolerance: float = 4.0) -> int:
buckets = Counter(round(float(x) / tolerance) for x in xs)
return sum(count >= 3 for count in buckets.values())
def local_pages(payload: dict[str, Any]) -> dict[int, dict[str, Any]]:
"""Summarize positioned ``pdf2md --items-json`` evidence by page."""
pages: dict[int, dict[str, Any]] = {}
for item in payload.get("items", []):
page_number = int(item["page"])
page = pages.setdefault(
page_number,
{"texts": [], "xs": [], "text_items": 0, "image_items": 0},
)
if item.get("item_type") == "image":
page["image_items"] += 1
continue
text = str(item.get("text", ""))
if text.strip():
page["texts"].append(text)
page["xs"].append(float(item.get("x", 0.0)))
page["text_items"] += 1
return pages
def alternate_pages(
payload: dict[str, Any] | list[dict[str, Any]],
) -> dict[int, dict[str, Any]]:
"""Summarize MuPDF ``stext.json`` evidence by page."""
pages: dict[int, dict[str, Any]] = {}
raw_pages = payload if isinstance(payload, list) else payload.get("pages", [])
for index, raw_page in enumerate(raw_pages, start=1):
page_number = int(raw_page.get("number", index))
page = {
"texts": [],
"xs": [],
"text_blocks": 0,
"text_lines": 0,
"image_blocks": 0,
}
for block in raw_page.get("blocks", []):
if block.get("type") == "image":
page["image_blocks"] += 1
continue
if block.get("type") != "text":
continue
page["text_blocks"] += 1
for line in block.get("lines", []):
text = str(line.get("text", ""))
if text.strip():
page["texts"].append(text)
bbox = line.get("bbox", {})
page["xs"].append(float(bbox.get("x", line.get("x", 0.0))))
page["text_lines"] += 1
pages[page_number] = page
return pages
def compare_page(
local: dict[str, Any],
alternate: dict[str, Any],
*,
min_token_gain: int,
min_alternate_only_ratio: float,
min_anchor_gain: int,
) -> dict[str, Any]:
"""Compare semantic and coarse layout evidence for one page."""
local_tokens = _tokens(local.get("texts", []))
alternate_tokens = _tokens(alternate.get("texts", []))
shared = local_tokens & alternate_tokens
alternate_only = alternate_tokens - local_tokens
local_only = local_tokens - alternate_tokens
local_total = sum(local_tokens.values())
alternate_total = sum(alternate_tokens.values())
shared_total = sum(shared.values())
alternate_only_total = sum(alternate_only.values())
local_only_total = sum(local_only.values())
net_token_gain = alternate_total - local_total
alternate_only_ratio = alternate_only_total / max(alternate_total, 1)
local_anchors = _repeated_x_anchors(local.get("xs", []))
alternate_anchors = _repeated_x_anchors(alternate.get("xs", []))
anchor_gain = alternate_anchors - local_anchors
image_gain = int(alternate.get("image_blocks", 0)) - int(
local.get("image_items", 0)
)
reasons: list[str] = []
if local_total == 0 and alternate_total >= max(5, min_token_gain // 2):
reasons.append("local_text_empty")
elif (
net_token_gain >= min_token_gain
and alternate_only_ratio >= min_alternate_only_ratio
):
reasons.append("alternate_has_more_text")
if anchor_gain >= min_anchor_gain:
reasons.append("alternate_has_more_alignment_anchors")
if image_gain > 0:
reasons.append("alternate_has_more_image_blocks")
if reasons:
classification = "investigate_alternate_evidence"
elif local_total - alternate_total >= min_token_gain:
classification = "local_has_more_text"
elif alternate_only_total + local_only_total:
classification = "different_segmentation_or_decoding"
else:
classification = "equivalent_text_evidence"
return {
"classification": classification,
"reasons": reasons,
"tokens": {
"local": local_total,
"alternate": alternate_total,
"shared": shared_total,
"net_alternate_gain": net_token_gain,
"alternate_only": alternate_only_total,
"local_only": local_only_total,
"alternate_only_ratio": alternate_only_ratio,
"alternate_only_sample": sorted(alternate_only)[:12],
"local_only_sample": sorted(local_only)[:12],
},
"layout": {
"local_text_items": int(local.get("text_items", 0)),
"local_image_items": int(local.get("image_items", 0)),
"local_repeated_x_anchors": local_anchors,
"alternate_text_blocks": int(alternate.get("text_blocks", 0)),
"alternate_text_lines": int(alternate.get("text_lines", 0)),
"alternate_image_blocks": int(alternate.get("image_blocks", 0)),
"alternate_repeated_x_anchors": alternate_anchors,
},
}
def compare_documents(
local_payload: dict[str, Any],
alternate_payload: dict[str, Any] | list[dict[str, Any]],
*,
min_token_gain: int = 20,
min_alternate_only_ratio: float = 0.15,
min_anchor_gain: int = 2,
) -> dict[str, Any]:
"""Return a page-level evidence report for already extracted payloads."""
local = local_pages(local_payload)
alternate = alternate_pages(alternate_payload)
page_numbers = sorted(local.keys() | alternate.keys())
pages = []
for page_number in page_numbers:
result = compare_page(
local.get(page_number, {}),
alternate.get(page_number, {}),
min_token_gain=min_token_gain,
min_alternate_only_ratio=min_alternate_only_ratio,
min_anchor_gain=min_anchor_gain,
)
result["page"] = page_number
pages.append(result)
flagged = [
page
for page in pages
if page["classification"] == "investigate_alternate_evidence"
]
return {
"summary": {
"pages": len(pages),
"flagged_pages": len(flagged),
"flagged_page_numbers": [page["page"] for page in flagged],
"local_tokens": sum(page["tokens"]["local"] for page in pages),
"alternate_tokens": sum(page["tokens"]["alternate"] for page in pages),
"alternate_only_tokens": sum(
page["tokens"]["alternate_only"] for page in pages
),
},
"pages": pages,
}
def _json_command(command: list[str]) -> Any:
try:
completed = subprocess.run(
command,
check=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
)
except subprocess.CalledProcessError as error:
detail = error.stderr.strip() or error.stdout.strip() or "no diagnostic output"
raise RuntimeError(f"command failed: {' '.join(command)}\n{detail}") from error
try:
return json.loads(completed.stdout)
except json.JSONDecodeError as error:
raise RuntimeError(
f"command did not return JSON: {' '.join(command)}: {error}"
) from error
def probe_pdf(
pdf: Path,
*,
pdf2md: Path,
mutool: Path,
min_token_gain: int,
min_alternate_only_ratio: float,
min_anchor_gain: int,
) -> dict[str, Any]:
local_payload = _json_command([str(pdf2md), str(pdf), "--items-json"])
# `stext.json` is MuPDF's native structured text output. The OCR formats
# are intentionally not used so this remains a deterministic no-model
# comparison.
alternate_payload = _json_command(
[str(mutool), "draw", "-q", "-F", "stext.json", "-o", "-", str(pdf)]
)
report = compare_documents(
local_payload,
alternate_payload,
min_token_gain=min_token_gain,
min_alternate_only_ratio=min_alternate_only_ratio,
min_anchor_gain=min_anchor_gain,
)
report["pdf"] = str(pdf)
return report
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("pdf", type=Path, nargs="+")
parser.add_argument("--pdf2md", type=Path, default=Path("target/release/pdf2md"))
parser.add_argument("--mutool", type=Path)
parser.add_argument("--json-output", type=Path)
parser.add_argument("--min-token-gain", type=int, default=20)
parser.add_argument("--min-alternate-only-ratio", type=float, default=0.15)
parser.add_argument("--min-anchor-gain", type=int, default=2)
return parser.parse_args(argv)
def _print_report(result: dict[str, Any]) -> None:
summary = result["summary"]
print(f"\n{result['pdf']}")
print(
f" {summary['flagged_pages']}/{summary['pages']} pages flagged; "
f"tokens local={summary['local_tokens']} alternate={summary['alternate_tokens']} "
f"alternate-only={summary['alternate_only_tokens']}"
)
for page in result["pages"]:
if page["classification"] != "investigate_alternate_evidence":
continue
reasons = ", ".join(page["reasons"])
tokens = page["tokens"]
print(
f" page {page['page']}: {reasons}; "
f"net tokens={tokens['net_alternate_gain']:+d}, "
f"alternate-only={tokens['alternate_only']}"
)
def main(argv: list[str] | None = None) -> int:
args = _arguments(argv)
pdf2md = args.pdf2md.absolute()
mutool = args.mutool or (Path(found) if (found := shutil.which("mutool")) else None)
if not pdf2md.is_file():
print(f"error: pdf2md binary not found: {pdf2md}", file=sys.stderr)
return 2
if mutool is None or not mutool.is_file():
print("error: mutool not found; install MuPDF or pass --mutool", file=sys.stderr)
return 2
if (
args.min_token_gain < 0
or args.min_anchor_gain < 0
or not 0.0 <= args.min_alternate_only_ratio <= 1.0
):
print("error: thresholds must be non-negative and ratio must be in [0, 1]", file=sys.stderr)
return 2
results = []
for pdf in args.pdf:
path = pdf.absolute()
if not path.is_file():
print(f"error: PDF not found: {path}", file=sys.stderr)
return 2
try:
result = probe_pdf(
path,
pdf2md=pdf2md,
mutool=mutool,
min_token_gain=args.min_token_gain,
min_alternate_only_ratio=args.min_alternate_only_ratio,
min_anchor_gain=args.min_anchor_gain,
)
except RuntimeError as error:
print(f"error: {error}", file=sys.stderr)
return 1
results.append(result)
_print_report(result)
payload = {
"schema_version": 1,
"experiment": "optional_mupdf_stext_evidence",
"ocr": False,
"thresholds": {
"min_token_gain": args.min_token_gain,
"min_alternate_only_ratio": args.min_alternate_only_ratio,
"min_anchor_gain": args.min_anchor_gain,
},
"documents": results,
}
if args.json_output:
args.json_output.parent.mkdir(parents=True, exist_ok=True)
args.json_output.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8")
return 0
if __name__ == "__main__":
raise SystemExit(main())
-203
View File
@@ -1,203 +0,0 @@
import io
import json
import sys
import tempfile
import unittest
from contextlib import redirect_stderr, redirect_stdout
from pathlib import Path
from unittest.mock import patch
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from bench_opendataloader import (
_arguments,
_print_report,
_run_engine,
compare_evaluations,
evaluate_gates,
)
def evaluation(overall, documents, *, missing=0):
return {
"metrics": {
"score": {
"overall_mean": overall,
"nid_mean": overall + 0.01,
},
"missing_predictions": missing,
},
"documents": [
{
"document_id": document_id,
"scores": {"overall": score},
}
for document_id, score in documents.items()
],
}
class ComparisonTests(unittest.TestCase):
def test_reports_metric_and_document_deltas(self):
baseline = evaluation(0.80, {"a": 0.8, "b": 0.6, "c": 0.7})
candidate = evaluation(0.82, {"a": 0.9, "b": 0.5, "c": 0.7})
result = compare_evaluations(baseline, candidate, top=1)
self.assertAlmostEqual(result["deltas"]["overall_mean"], 0.02)
self.assertEqual(result["documents"]["improved"], 1)
self.assertEqual(result["documents"]["regressed"], 1)
self.assertEqual(result["documents"]["unchanged"], 1)
self.assertEqual(
result["documents"]["largest_improvements"][0]["document_id"], "a"
)
self.assertEqual(
result["documents"]["largest_regressions"][0]["document_id"], "b"
)
def test_reference_delta_is_reported(self):
baseline = evaluation(0.80, {})
candidate = evaluation(0.82, {})
reference = evaluation(0.81, {})
result = compare_evaluations(baseline, candidate, reference)
self.assertAlmostEqual(
result["candidate_vs_reference"]["overall_mean"], 0.01
)
def test_gates_cover_aggregate_document_missing_and_reference(self):
comparison = compare_evaluations(
evaluation(0.80, {"a": 0.8}),
evaluation(0.79, {"a": 0.7}, missing=1),
evaluation(0.81, {}),
)
failures = evaluate_gates(
comparison,
min_overall_delta=0.0,
max_document_regression=0.05,
max_missing=0,
require_reference_lead=True,
)
self.assertEqual(len(failures), 4)
def test_regression_gate_is_independent_of_report_limit(self):
comparison = compare_evaluations(
evaluation(0.80, {"a": 0.8}),
evaluation(0.80, {"a": 0.7}),
top=0,
)
failures = evaluate_gates(
comparison,
min_overall_delta=0.0,
max_document_regression=0.05,
max_missing=0,
require_reference_lead=False,
)
self.assertEqual(len(failures), 1)
self.assertIn("largest document regression", failures[0])
def test_report_handles_reference_without_overall_score(self):
result = compare_evaluations(
evaluation(0.80, {}),
evaluation(0.82, {}),
{"metrics": {"score": {"nid_mean": 0.81}}},
)
output = io.StringIO()
with redirect_stdout(output):
_print_report(result)
self.assertIn("Reference overall: n/a; candidate delta: n/a", output.getvalue())
def test_reference_gate_reports_missing_score_as_unavailable(self):
comparison = compare_evaluations(
evaluation(0.80, {}),
evaluation(0.82, {}),
)
failures = evaluate_gates(
comparison,
min_overall_delta=0.0,
max_document_regression=None,
max_missing=0,
require_reference_lead=True,
)
self.assertEqual(failures, ["reference overall score is unavailable"])
def test_arguments_reject_negative_counts_and_allow_zero_top(self):
required = [
"--bench-dir",
".",
"--baseline",
"baseline",
"--candidate",
"candidate",
]
self.assertEqual(_arguments(required + ["--top", "0"]).top, 0)
for option in ("--top", "--max-document-regression", "--max-missing"):
with self.subTest(option=option), redirect_stderr(io.StringIO()):
with self.assertRaises(SystemExit):
_arguments(required + [option, "-1"])
def test_arguments_reject_nonfinite_float_thresholds(self):
required = [
"--bench-dir",
".",
"--baseline",
"baseline",
"--candidate",
"candidate",
]
for option in ("--min-overall-delta", "--max-document-regression"):
for value in ("nan", "inf", "-inf"):
with self.subTest(option=option, value=value), redirect_stderr(
io.StringIO()
):
with self.assertRaises(SystemExit):
_arguments(required + [option, value])
def test_run_engine_clears_stale_predictions_before_parser(self):
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
bench_dir = root / "bench"
source = bench_dir / "prediction" / "pdf-inspector"
source.mkdir(parents=True)
(source / "stale.md").write_text("stale", encoding="utf-8")
scratch = root / "scratch"
scratch.mkdir()
def fake_run(command, *, cwd, env=None):
if any(part.endswith("pdf_parser.py") for part in command):
self.assertFalse(source.exists())
(source / "markdown").mkdir(parents=True)
(source / "markdown" / "new.md").write_text(
"new", encoding="utf-8"
)
else:
destination = scratch / "candidate"
(destination / "evaluation.json").write_text(
json.dumps(evaluation(0.82, {})), encoding="utf-8"
)
with patch("bench_opendataloader._run", side_effect=fake_run):
result = _run_engine(
bench_dir=bench_dir,
python=Path("python"),
binary=Path("pdf2md"),
label="candidate",
scratch_root=scratch,
)
self.assertEqual(result["metrics"]["score"]["overall_mean"], 0.82)
self.assertFalse((source / "stale.md").exists())
self.assertFalse((scratch / "candidate" / "stale.md").exists())
if __name__ == "__main__":
unittest.main()
@@ -1,103 +0,0 @@
import sys
import unittest
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from probe_backend_evidence import compare_documents
def local_payload(items):
return {"items": items}
def item(page, text, x=10, item_type="text"):
return {"page": page, "text": text, "x": x, "item_type": item_type}
def alternate_payload(pages):
return {"pages": pages}
def page(lines, *, images=0):
blocks = [
{
"type": "text",
"lines": [
{"text": text, "bbox": {"x": x, "y": index * 10, "w": 80, "h": 8}}
for index, (text, x) in enumerate(lines)
],
}
]
blocks.extend({"type": "image"} for _ in range(images))
return {"blocks": blocks}
class EvidenceComparisonTests(unittest.TestCase):
def test_accepts_real_top_level_page_array(self):
local = local_payload([item(1, "alpha beta")])
alternate = [page([("alpha beta gamma", 10)])]
result = compare_documents(local, alternate)["pages"][0]
self.assertEqual(result["tokens"]["alternate"], 3)
self.assertEqual(result["tokens"]["net_alternate_gain"], 1)
def test_flags_material_alternate_text_gain(self):
local = local_payload([item(1, "alpha beta")])
alternate = alternate_payload(
[page([("alpha beta gamma delta epsilon zeta", 10)])]
)
report = compare_documents(
local,
alternate,
min_token_gain=3,
min_alternate_only_ratio=0.2,
)
result = report["pages"][0]
self.assertEqual(result["classification"], "investigate_alternate_evidence")
self.assertIn("alternate_has_more_text", result["reasons"])
self.assertEqual(result["tokens"]["net_alternate_gain"], 4)
def test_repeated_alignment_and_image_evidence_are_reported(self):
local = local_payload([item(1, "one two", 10)])
alternate = alternate_payload(
[
page(
[
("one two", 10),
("row three", 100),
("row four", 100),
("row five", 100),
],
images=1,
)
]
)
result = compare_documents(
local,
alternate,
min_token_gain=99,
min_anchor_gain=1,
)["pages"][0]
self.assertIn("alternate_has_more_alignment_anchors", result["reasons"])
self.assertIn("alternate_has_more_image_blocks", result["reasons"])
self.assertEqual(result["layout"]["alternate_repeated_x_anchors"], 1)
def test_token_segmentation_difference_does_not_imply_more_evidence(self):
local = local_payload([item(1, "Revenue 2025")])
alternate = alternate_payload([page([("Revenue 2024", 10)])])
result = compare_documents(local, alternate, min_token_gain=2)["pages"][0]
self.assertEqual(result["classification"], "different_segmentation_or_decoding")
self.assertEqual(result["reasons"], [])
self.assertEqual(result["tokens"]["alternate_only_sample"], ["2024"])
if __name__ == "__main__":
unittest.main()
+353 -1097
View File
File diff suppressed because it is too large Load Diff
-1
View File
@@ -63,7 +63,6 @@ fn json_escape(s: &str) -> String {
}
fn main() {
#[cfg(not(target_arch = "wasm32"))]
env_logger::init();
let args: Vec<String> = env::args().collect();
-1
View File
@@ -190,7 +190,6 @@ fn print_layout_info(layout: &LayoutComplexity) {
}
fn main() {
#[cfg(not(target_arch = "wasm32"))]
env_logger::init();
let args: Vec<String> = env::args().collect();
+1 -1
View File
@@ -162,7 +162,7 @@ pub(crate) fn extract_page_text_items(
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
// Build font encoding maps from Differences arrays
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
// Build font width info for accurate text positioning
let font_widths = build_font_widths(doc, &fonts);
+12 -154
View File
@@ -497,13 +497,9 @@ pub(crate) fn get_operand_bytes(obj: &Object) -> Option<&[u8]> {
/// Build encoding maps for all fonts on a page.
/// Returns `(encodings, has_gid_fonts)` where `has_gid_fonts` is true when
/// any font uses raw glyph ID names (gidNNNNN) that can't be decoded.
/// Gid names whose codes the font's own ToUnicode CMap maps are decodable
/// and do not set the flag (LibreOffice subsets write /gidNNNN Differences
/// names alongside a complete ToUnicode CMap).
pub(crate) fn build_font_encodings(
doc: &Document,
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
cmaps: &FontCMaps,
) -> (PageFontEncodings, bool) {
let mut encodings = PageFontEncodings::new();
let mut has_gid_fonts = false;
@@ -512,9 +508,7 @@ pub(crate) fn build_font_encodings(
let resource_name = String::from_utf8_lossy(font_name).to_string();
if let Some(result) = parse_font_encoding(doc, font_dict) {
if !result.gid_codes.is_empty()
&& !tounicode_maps_codes(font_dict, cmaps, &result.gid_codes)
{
if result.gid_glyph_count > 0 {
has_gid_fonts = true;
}
if !result.map.is_empty() {
@@ -526,34 +520,6 @@ pub(crate) fn build_font_encodings(
(encodings, has_gid_fonts)
}
/// True when the font's ToUnicode CMap maps the gid-named character codes,
/// so the Differences entries still decode through the CMap.
fn tounicode_maps_codes(font_dict: &lopdf::Dictionary, cmaps: &FontCMaps, codes: &[u8]) -> bool {
let Some(obj_ref) = font_dict
.get(b"ToUnicode")
.ok()
.and_then(|o| o.as_reference().ok())
else {
return false;
};
let Some(entry) = cmaps.get_by_obj(obj_ref.0) else {
return false;
};
// At least one gid code usably mapped means the CMap addresses these
// codes; remaining unmapped codes are subset leftovers (e.g. the
// component glyphs of an emoji ZWJ sequence mapped whole on its first
// code). A mapping is usable only when extraction would accept it —
// empty or U+FFFD results are rejected there as invalid. Fonts whose
// CMap ignores the gid codes entirely stay flagged, and the downstream
// garbage/encoding checks still catch partial damage.
codes.iter().any(|&code| {
entry
.primary
.lookup(code as u16)
.is_some_and(|s| !s.is_empty() && !s.contains('\u{FFFD}'))
})
}
/// Parse font encoding from a font dictionary
pub(crate) fn parse_font_encoding(
doc: &Document,
@@ -592,10 +558,11 @@ pub(crate) fn parse_font_encoding(
/// Result of parsing an encoding dictionary's Differences array.
pub(crate) struct EncodingResult {
pub map: FontEncodingMap,
/// Character codes whose glyph names match the `gidNNNNN` pattern (raw
/// glyph IDs). These reference the original font's glyph table and are
/// only decodable when the font's ToUnicode CMap maps the code.
pub gid_codes: Vec<u8>,
/// Number of glyph names matching the `gidNNNNN` pattern (raw glyph IDs).
/// These indicate a font with unresolvable encoding — the glyph IDs
/// reference the original font's glyph table, but without the original
/// font's cmap there is no way to map them to Unicode.
pub gid_glyph_count: u32,
}
/// Parse an encoding dictionary with Differences array
@@ -621,7 +588,7 @@ pub(crate) fn parse_encoding_dictionary(
let mut encoding_map = FontEncodingMap::new();
let mut current_code: u8 = 0;
let mut ligature_count = 0u32;
let mut gid_codes: Vec<u8> = Vec::new();
let mut gid_glyph_count = 0u32;
for item in diff_array {
match item {
@@ -647,7 +614,7 @@ pub(crate) fn parse_encoding_dictionary(
&& glyph_name.len() >= 4
&& glyph_name[3..].chars().all(|c| c.is_ascii_digit())
{
gid_codes.push(current_code);
gid_glyph_count += 1;
}
if let Some(ch) = mapped_char {
encoding_map.insert(current_code, ch);
@@ -671,16 +638,16 @@ pub(crate) fn parse_encoding_dictionary(
);
}
if !gid_codes.is_empty() {
if gid_glyph_count > 0 {
debug!(
" Differences: {} gid-encoded glyphs (decodable only via ToUnicode)",
gid_codes.len()
" Differences: {} gid-encoded glyphs (unresolvable without original font)",
gid_glyph_count
);
}
Some(EncodingResult {
map: encoding_map,
gid_codes,
gid_glyph_count,
})
}
@@ -1971,113 +1938,4 @@ mod tests {
false
));
}
fn gid_font_doc(bfchar: Option<&str>) -> (Document, lopdf::ObjectId) {
use lopdf::Stream;
let mut doc = Document::with_version("1.4");
let cmap = format!(
"/CIDInit /ProcSet findresource begin
12 dict begin
begincmap
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfchar
{}
endbfchar
endcmap
CMapName currentdict /CMap defineresource pop
end
end",
bfchar.unwrap_or_default()
);
let tounicode_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
cmap.into_bytes(),
)));
let enc_id = doc.add_object(dictionary! {
"Type" => "Encoding",
"Differences" => vec![
1.into(),
Object::Name(b"gid1283".to_vec()),
Object::Name(b"gid1464".to_vec()),
],
});
let mut font = dictionary! {
"Type" => "Font",
"Subtype" => "TrueType",
"BaseFont" => "ABCDEF+OpenSymbol",
"Encoding" => Object::Reference(enc_id),
};
if bfchar.is_some() {
font.set("ToUnicode", Object::Reference(tounicode_id));
}
let font_id = doc.add_object(font);
let page_id = doc.add_object(dictionary! {
"Type" => "Page",
"Resources" => dictionary! {
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
},
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
});
let pages_id = doc.add_object(dictionary! {
"Type" => "Pages",
"Count" => Object::Integer(1),
"Kids" => vec![Object::Reference(page_id)],
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => Object::Reference(pages_id),
});
doc.trailer.set("Root", Object::Reference(catalog_id));
(doc, page_id)
}
fn gid_flagged(bfchar: Option<&str>) -> bool {
let (doc, page_id) = gid_font_doc(bfchar);
let cmaps = FontCMaps::from_doc(&doc);
let fonts = doc.get_page_fonts(page_id).unwrap();
let (_, has_gid_fonts) = build_font_encodings(&doc, &fonts, &cmaps);
has_gid_fonts
}
#[test]
fn gid_differences_with_covering_tounicode_are_not_flagged() {
// LibreOffice subsets write /gidNNNN Differences names alongside a
// ToUnicode CMap that decodes those codes; the page must not be
// flagged as unresolvable (which would suppress the whole document's
// markdown when every page carries such a font).
assert!(!gid_flagged(Some("<01> <2022>\n<02> <25E6>")));
}
#[test]
fn gid_differences_with_partial_tounicode_are_not_flagged() {
// An emoji ZWJ sequence maps whole on its first code; the remaining
// component-glyph codes are subset leftovers, not damage.
assert!(!gid_flagged(Some(
"<01> <D83DDC68200DD83DDC69200DD83DDC67>"
)));
}
#[test]
fn gid_differences_without_tounicode_are_flagged() {
assert!(
gid_flagged(None),
"gid glyphs without ToUnicode are unresolvable"
);
}
#[test]
fn gid_differences_with_disjoint_tounicode_are_flagged() {
// A ToUnicode that never addresses the gid codes leaves them
// unresolvable.
assert!(gid_flagged(Some("<10> <0041>")));
}
#[test]
fn gid_differences_with_replacement_char_tounicode_are_flagged() {
// A mapping to U+FFFD is not usable — extraction rejects it as an
// invalid CMap result — so it must not clear the gid flag.
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
}
}
+5 -92
View File
@@ -1153,22 +1153,6 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
group_into_lines_with_thresholds(items, &HashMap::new(), &HashSet::new())
}
/// Group text items into lines without removing numeric page headers or footers.
///
/// Plain-text extraction uses this path because every extracted item is part of
/// the API result. Markdown conversion keeps using [`group_into_lines`], where
/// page-number suppression is an intentional presentation cleanup.
pub fn group_into_lines_preserving_all_text(items: Vec<TextItem>) -> Vec<TextLine> {
group_into_lines_with_thresholds_and_regions_impl(
items,
&HashMap::new(),
&HashSet::new(),
&HashMap::new(),
&HashMap::new(),
false,
)
}
/// Group text items into lines, using pre-computed per-page adaptive thresholds
/// from Canva-style letter-spacing detection. Falls back to computing the
/// threshold from item gaps when no pre-computed value is available.
@@ -1195,55 +1179,16 @@ pub(crate) fn group_into_lines_with_thresholds_and_charts(
page_thresholds: &HashMap<u32, f32>,
table_pages: &HashSet<u32>,
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
) -> Vec<TextLine> {
group_into_lines_with_thresholds_and_regions(
items,
page_thresholds,
table_pages,
chart_regions,
&HashMap::new(),
)
}
pub(crate) fn group_into_lines_with_thresholds_and_regions(
items: Vec<TextItem>,
page_thresholds: &HashMap<u32, f32>,
table_pages: &HashSet<u32>,
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
) -> Vec<TextLine> {
group_into_lines_with_thresholds_and_regions_impl(
items,
page_thresholds,
table_pages,
chart_regions,
image_regions,
true,
)
}
fn group_into_lines_with_thresholds_and_regions_impl(
items: Vec<TextItem>,
page_thresholds: &HashMap<u32, f32>,
table_pages: &HashSet<u32>,
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
filter_page_numbers: bool,
) -> Vec<TextLine> {
if items.is_empty() {
return Vec::new();
}
// Markdown output omits standalone numeric headers/footers. Plain-text
// callers opt out because dropping extracted text violates that API.
let items = if filter_page_numbers {
items
.into_iter()
.filter(|item| !is_page_number(item))
.collect()
} else {
items
};
// Filter out page numbers (standalone numbers at top/bottom of page)
let items: Vec<TextItem> = items
.into_iter()
.filter(|item| !is_page_number(item))
.collect();
// Get unique pages
let mut pages: Vec<u32> = items.iter().map(|i| i.page).collect();
@@ -1260,38 +1205,6 @@ fn group_into_lines_with_thresholds_and_regions_impl(
// Non-Canva pages use the default 0.10 threshold.
let adaptive_threshold = page_thresholds.get(&page).copied().unwrap_or(0.10);
// Image-backed region graphs recover local/asymmetric column flows
// that a whole-page projection cannot represent. Charts already have
// their own positioned-region ordering and therefore stay on that path.
if !chart_regions.contains_key(&page) {
let preliminary_columns =
detect_columns(&page_items, page, table_pages.contains(&page));
let detected_split =
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
if let Some(band) = image_regions.get(&page).and_then(|regions| {
super::reading_order::infer_image_anchored_flow(
&page_items,
regions,
detected_split,
)
}) {
debug!(
"page {}: image-anchored region graph split={:.1} y=[{:.1}..{:.1}]",
page, band.split_x, band.y_bottom, band.y_top
);
for node in super::reading_order::build_region_graph(page_items, band) {
debug!(
"page {}: region node {:?} items={}",
page,
node.kind,
node.items.len()
);
all_lines.extend(group_single_column(node.items, adaptive_threshold));
}
continue;
}
}
// Detect columns for this page, blind to chart text.
debug!(
"page {}: grouping chart-aware={} regions={:?}",
+1 -15
View File
@@ -6,7 +6,6 @@ pub(crate) mod content_stream;
mod fonts;
mod layout;
mod links;
mod reading_order;
pub(crate) mod underline;
mod xobjects;
@@ -27,12 +26,11 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
pub use crate::types::{ItemType, TextLine};
pub(crate) use fonts::FontStyleCache;
pub(crate) use layout::detect_columns;
pub use layout::group_into_lines;
pub(crate) use layout::group_into_lines_with_thresholds;
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
pub(crate) use layout::group_into_lines_with_thresholds_and_regions;
pub(crate) use layout::is_newspaper_layout;
pub(crate) use layout::ColumnRegion;
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
// ---------------------------------------------------------------------------
// Public API
@@ -1519,18 +1517,6 @@ mod tests {
assert_eq!(lines[1].text(), "Next line");
}
#[test]
fn preserving_all_text_keeps_numeric_page_footer() {
let mut page_number = make_merge_item("42", 100.0, 12.0);
page_number.y = 50.0;
assert!(group_into_lines(vec![page_number.clone()]).is_empty());
let lines = group_into_lines_preserving_all_text(vec![page_number]);
assert_eq!(lines.len(), 1);
assert_eq!(lines[0].text(), "42");
}
#[test]
fn test_bold_italic_detection() {
// Test bold detection
-591
View File
@@ -1,591 +0,0 @@
//! Region-graph evidence for page reading order.
//!
//! Whole-page column histograms fail when images or spanning captions occupy
//! only part of a page. This module turns image geometry and repeated row
//! gutters into a small directed acyclic graph: content above a local column
//! band, the left flow, the right flow, and content below it. The graph is
//! deliberately evidence-gated; ordinary pages keep the established layout
//! path.
use crate::text_utils::{effective_width, is_cjk_char, is_rtl_text};
use crate::types::TextItem;
const MIN_IMAGE_WIDTH: f32 = 60.0;
const MIN_IMAGE_HEIGHT: f32 = 40.0;
const MIN_ROW_GUTTER: f32 = 8.0;
const SPLIT_CLUSTER_TOLERANCE: f32 = 20.0;
const MIN_ALIGNED_ROWS: usize = 4;
pub(crate) type ImageRegion = (f32, f32, f32, f32);
#[derive(Debug, Clone, Copy, PartialEq)]
pub(crate) struct ColumnFlowBand {
pub(crate) split_x: f32,
pub(crate) y_bottom: f32,
pub(crate) y_top: f32,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum RegionKind {
FullWidth,
Column,
}
#[derive(Debug)]
pub(crate) struct RegionNode {
pub(crate) kind: RegionKind,
pub(crate) items: Vec<TextItem>,
}
#[derive(Debug)]
struct Row<'a> {
y: f32,
items: Vec<&'a TextItem>,
}
fn page_x_bounds(items: &[TextItem], images: &[ImageRegion]) -> Option<(f32, f32)> {
let text_min = items
.iter()
.map(|item| item.x)
.fold(f32::INFINITY, f32::min);
let text_max = items
.iter()
.map(|item| item.x + effective_width(item))
.fold(f32::NEG_INFINITY, f32::max);
let image_min = images
.iter()
.map(|region| region.0.min(region.2))
.fold(f32::INFINITY, f32::min);
let image_max = images
.iter()
.map(|region| region.0.max(region.2))
.fold(f32::NEG_INFINITY, f32::max);
let x_min = text_min.min(image_min);
let x_max = text_max.max(image_max);
(x_min.is_finite() && x_max.is_finite() && x_max > x_min).then_some((x_min, x_max))
}
fn group_rows(items: &[TextItem]) -> Vec<Row<'_>> {
const Y_TOLERANCE: f32 = 3.0;
let mut sorted: Vec<&TextItem> = items.iter().collect();
sorted.sort_by(|left, right| right.y.total_cmp(&left.y));
let mut rows: Vec<Row<'_>> = Vec::new();
for item in sorted {
if let Some(row) = rows
.last_mut()
.filter(|row| (row.y - item.y).abs() <= Y_TOLERANCE)
{
row.items.push(item);
row.y = row.items.iter().map(|member| member.y).sum::<f32>() / row.items.len() as f32;
} else {
rows.push(Row {
y: item.y,
items: vec![item],
});
}
}
for row in &mut rows {
row.items.sort_by(|left, right| left.x.total_cmp(&right.x));
}
rows
}
fn side_is_prose(items: &[&TextItem]) -> bool {
let text = items
.iter()
.map(|item| item.text.trim())
.collect::<Vec<_>>()
.join(" ");
let alphabetic_count = text
.chars()
.filter(|character| character.is_alphabetic())
.count();
let cjk_count = text
.chars()
.filter(|character| is_cjk_char(*character))
.count();
(text.split_whitespace().count() >= 3 || cjk_count >= 10) && alphabetic_count >= 10
}
fn aligned_row_split(row: &Row<'_>, x_min: f32, x_max: f32) -> Option<f32> {
if row.items.len() < 2 {
return None;
}
let page_width = x_max - x_min;
let center_low = x_min + page_width * 0.25;
let center_high = x_min + page_width * 0.75;
row.items
.windows(2)
.filter_map(|pair| {
let left_end = pair[0].x + effective_width(pair[0]);
let right_start = pair[1].x;
let gap = right_start - left_end;
let split_x = (left_end + right_start) / 2.0;
if gap < MIN_ROW_GUTTER || split_x < center_low || split_x > center_high {
return None;
}
let left: Vec<&TextItem> = row
.items
.iter()
.copied()
.filter(|item| item.x + effective_width(item) / 2.0 < split_x)
.collect();
let right: Vec<&TextItem> = row
.items
.iter()
.copied()
.filter(|item| item.x + effective_width(item) / 2.0 >= split_x)
.collect();
(side_is_prose(&left) && side_is_prose(&right)).then_some((split_x, gap))
})
.max_by(|left, right| left.1.total_cmp(&right.1))
.map(|candidate| candidate.0)
}
fn local_flow_below_full_width_image(
items: &[TextItem],
images: &[ImageRegion],
x_min: f32,
x_max: f32,
) -> Option<ColumnFlowBand> {
let page_width = x_max - x_min;
let full_width_images: Vec<ImageRegion> = images
.iter()
.copied()
.filter(|&(x0, y0, x1, y1)| {
let width = (x1 - x0).abs();
let height = (y1 - y0).abs();
width >= page_width * 0.65 && height >= 60.0
})
.collect();
// A local column flow below an image is only unambiguous for a single,
// nearly square hero/figure. Wide report banners and full-page artwork
// frequently sit above unrelated page furniture whose aligned labels can
// mimic prose columns.
if full_width_images.len() != 1 {
return None;
}
let (image_x0, _, image_x1, _) = full_width_images[0];
let anchor_width = (image_x1 - image_x0).abs();
let anchor_height = (full_width_images[0].3 - full_width_images[0].1).abs();
if anchor_width < page_width * 0.85
|| anchor_height < anchor_width * 0.85
|| anchor_height > anchor_width * 1.2
{
return None;
}
let image_bottom = full_width_images
.iter()
.map(|&(_, y0, _, y1)| y0.min(y1))
.fold(f32::NEG_INFINITY, f32::max);
if !image_bottom.is_finite() {
return None;
}
let below: Vec<TextItem> = items
.iter()
.filter(|item| item.y < image_bottom && item.y >= image_bottom - 220.0)
.cloned()
.collect();
let candidates: Vec<(f32, f32)> = group_rows(&below)
.into_iter()
.filter_map(|row| aligned_row_split(&row, x_min, x_max).map(|split| (split, row.y)))
.collect();
if candidates.len() < MIN_ALIGNED_ROWS {
return None;
}
let mut clusters: Vec<Vec<(f32, f32)>> = Vec::new();
for candidate in candidates {
if let Some(cluster) = clusters.iter_mut().find(|cluster| {
let mean = cluster.iter().map(|entry| entry.0).sum::<f32>() / cluster.len() as f32;
(mean - candidate.0).abs() <= SPLIT_CLUSTER_TOLERANCE
}) {
cluster.push(candidate);
} else {
clusters.push(vec![candidate]);
}
}
let dominant = clusters.into_iter().max_by_key(Vec::len)?;
if dominant.len() < MIN_ALIGNED_ROWS {
return None;
}
let split_x = dominant.iter().map(|entry| entry.0).sum::<f32>() / dominant.len() as f32;
let y_top = dominant
.iter()
.map(|entry| entry.1)
.fold(f32::NEG_INFINITY, f32::max)
+ 3.0;
let image_gap = image_bottom - y_top;
if !(60.0..=120.0).contains(&image_gap) {
return None;
}
let y_bottom = dominant
.iter()
.map(|entry| entry.1)
.fold(f32::INFINITY, f32::min)
- 3.0;
if y_top - y_bottom > 130.0 {
return None;
}
log::debug!(
"page {}: full-width image flow images={} aligned_rows={} split={:.1} page=[{:.1}..{:.1}] image_bottom={:.1} y=[{:.1}..{:.1}] full_width={:?}",
items.first().map_or(0, |item| item.page),
images.len(),
dominant.len(),
split_x,
x_min,
x_max,
image_bottom,
y_bottom,
y_top,
full_width_images
);
Some(ColumnFlowBand {
split_x,
y_bottom,
y_top,
})
}
fn paired_column_images(
items: &[TextItem],
images: &[ImageRegion],
split_x: f32,
x_min: f32,
x_max: f32,
) -> Option<ColumnFlowBand> {
let page_width = x_max - x_min;
if split_x < x_min + page_width * 0.4 || split_x > x_min + page_width * 0.6 {
return None;
}
let qualifying: Vec<ImageRegion> = images
.iter()
.copied()
.filter(|&(x0, y0, x1, y1)| {
let image_left = x0.min(x1);
let image_right = x0.max(x1);
let confined_to_one_column = image_right <= split_x || image_left >= split_x;
confined_to_one_column
&& (x1 - x0).abs() >= MIN_IMAGE_WIDTH
&& (y1 - y0).abs() >= MIN_IMAGE_HEIGHT
})
.collect();
let wide_images: Vec<ImageRegion> = qualifying
.iter()
.copied()
.filter(|(x0, _, x1, _)| (x1 - x0).abs() >= page_width * 0.35)
.collect();
if qualifying.len() < 3 || wide_images.len() < 3 {
return None;
}
let has_left = qualifying
.iter()
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 < split_x);
let has_right = qualifying
.iter()
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 >= split_x);
if !has_left || !has_right {
return None;
}
// A meaningful image-backed column flow spans multiple vertical panels.
// Three same-row header/logo images can otherwise satisfy the image count
// and send an ordinary asymmetric page through sequential column order.
let image_y_min = wide_images
.iter()
.map(|region| region.1.min(region.3))
.fold(f32::INFINITY, f32::min);
let image_y_max = wide_images
.iter()
.map(|region| region.1.max(region.3))
.fold(f32::NEG_INFINITY, f32::max);
let has_vertical_stack = wide_images.iter().enumerate().any(|(index, left)| {
wide_images.iter().skip(index + 1).any(|right| {
let same_side =
((left.0 + left.2) / 2.0 < split_x) == ((right.0 + right.2) / 2.0 < split_x);
let left_center = (left.1 + left.3) / 2.0;
let right_center = (right.1 + right.3) / 2.0;
let left_height = (left.3 - left.1).abs();
let right_height = (right.3 - right.1).abs();
let vertical_gap = if left.1.max(left.3) < right.1.min(right.3) {
right.1.min(right.3) - left.1.max(left.3)
} else if right.1.max(right.3) < left.1.min(left.3) {
left.1.min(left.3) - right.1.max(right.3)
} else {
0.0
};
same_side
&& (left_center - right_center).abs() >= left_height.min(right_height) * 0.5
&& vertical_gap <= left_height.max(right_height) * 0.5
})
});
if image_y_max - image_y_min < page_width * 0.45 || !has_vertical_stack {
return None;
}
let y_top = qualifying
.iter()
.map(|region| region.1.max(region.3))
.fold(f32::NEG_INFINITY, f32::max)
+ 3.0;
// Only column-confined text proves the lower extent of the flow. A
// spanning heading or caption below the columns must become the trailing
// full-width node rather than stretching the column band to the page foot.
let y_bottom = items
.iter()
.filter(|item| {
let item_right = item.x + effective_width(item);
item.y <= y_top && (item_right <= split_x || item.x >= split_x)
})
.map(|item| item.y)
.fold(f32::INFINITY, f32::min)
- 3.0;
if !y_bottom.is_finite() {
return None;
}
let distinct_rows = |right: bool| {
let mut ys: Vec<f32> = items
.iter()
.filter(|item| {
item.y <= y_top && (item.x + effective_width(item) / 2.0 >= split_x) == right
})
.map(|item| item.y)
.collect();
ys.sort_by(|left, right| left.total_cmp(right));
ys.dedup_by(|left, right| (*left - *right).abs() <= 3.0);
ys.len()
};
let left_rows = distinct_rows(false);
let right_rows = distinct_rows(true);
let line_balance = left_rows.min(right_rows) as f32 / left_rows.max(right_rows).max(1) as f32;
(left_rows >= 5 && right_rows >= 5 && line_balance < 0.55).then(|| {
log::debug!(
"page {}: paired-image flow qualifying_images={} rows={}/{} split={:.1} page=[{:.1}..{:.1}] y=[{:.1}..{:.1}] images={:?}",
items.first().map_or(0, |item| item.page),
qualifying.len(),
left_rows,
right_rows,
split_x,
x_min,
x_max,
y_bottom,
y_top,
qualifying
);
ColumnFlowBand {
split_x,
y_bottom,
y_top,
}
})
}
pub(crate) fn infer_image_anchored_flow(
items: &[TextItem],
images: &[ImageRegion],
detected_split: Option<f32>,
) -> Option<ColumnFlowBand> {
if items.is_empty() || images.is_empty() {
return None;
}
let (x_min, x_max) = page_x_bounds(items, images)?;
detected_split
.and_then(|split_x| paired_column_images(items, images, split_x, x_min, x_max))
.or_else(|| local_flow_below_full_width_image(items, images, x_min, x_max))
}
/// Partition a page into the topological order `above -> left -> right -> below`.
/// These edges encode the reading-order DAG; empty nodes are omitted.
pub(crate) fn build_region_graph(items: Vec<TextItem>, band: ColumnFlowBand) -> Vec<RegionNode> {
let mut above = Vec::new();
let mut left = Vec::new();
let mut right = Vec::new();
let mut below = Vec::new();
for item in items {
if item.y > band.y_top {
above.push(item);
} else if item.y < band.y_bottom {
below.push(item);
} else if item.x + effective_width(&item) / 2.0 < band.split_x {
left.push(item);
} else {
right.push(item);
}
}
let rtl = is_rtl_text(left.iter().chain(right.iter()).map(|item| &item.text));
let mut ordered = vec![(RegionKind::FullWidth, above)];
if rtl {
ordered.push((RegionKind::Column, right));
ordered.push((RegionKind::Column, left));
} else {
ordered.push((RegionKind::Column, left));
ordered.push((RegionKind::Column, right));
}
ordered.push((RegionKind::FullWidth, below));
ordered
.into_iter()
.filter_map(|(kind, items)| (!items.is_empty()).then_some(RegionNode { kind, items }))
.collect()
}
#[cfg(test)]
mod tests {
use super::*;
use crate::types::ItemType;
fn item(text: &str, x: f32, y: f32, width: f32) -> TextItem {
TextItem {
text: text.into(),
x,
y,
width,
height: 11.0,
font: "F1".into(),
font_size: 11.0,
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
}
#[test]
fn full_width_image_anchors_local_two_column_flow() {
let mut items = vec![
item("A full width caption", 55.0, 230.0, 430.0),
item("A trailing full width heading", 55.0, 80.0, 430.0),
];
for index in 0..5 {
let y = 170.0 - index as f32 * 14.0;
items.push(item("left column prose words", 55.0, y, 210.0));
items.push(item("right column prose words", 280.0, y, 210.0));
}
let images = vec![(55.0, 250.0, 490.0, 680.0)];
let band = infer_image_anchored_flow(&items, &images, None).unwrap();
assert!((band.split_x - 272.5).abs() < 2.0);
let graph = build_region_graph(items, band);
assert_eq!(graph.len(), 4);
assert_eq!(graph[0].kind, RegionKind::FullWidth);
assert_eq!(graph[1].kind, RegionKind::Column);
assert_eq!(graph[2].kind, RegionKind::Column);
assert_eq!(graph[3].kind, RegionKind::FullWidth);
assert_eq!(graph[3].items[0].text, "A trailing full width heading");
}
#[test]
fn full_width_image_anchors_cjk_column_flow() {
let mut items = Vec::new();
for index in 0..5 {
let y = 170.0 - index as f32 * 14.0;
items.push(item("左栏这是没有空格的正文内容", 55.0, y, 210.0));
items.push(item("右栏这是没有空格的正文内容", 280.0, y, 210.0));
}
let images = vec![(55.0, 250.0, 490.0, 680.0)];
assert!(infer_image_anchored_flow(&items, &images, None).is_some());
}
#[test]
fn paired_images_anchor_unbalanced_column_flows() {
let mut items = vec![
item("running header", 55.0, 700.0, 430.0),
item("trailing full width caption", 55.0, 300.0, 430.0),
];
for index in 0..5 {
items.push(item(
"left prose words",
55.0,
500.0 - index as f32 * 14.0,
200.0,
));
items.push(item(
"right prose words",
280.0,
520.0 - index as f32 * 14.0,
200.0,
));
}
for index in 5..12 {
items.push(item(
"right continuation prose words",
280.0,
520.0 - index as f32 * 14.0,
200.0,
));
}
let images = vec![
(55.0, 530.0, 255.0, 680.0),
(55.0, 380.0, 255.0, 530.0),
(280.0, 560.0, 490.0, 680.0),
];
let band = infer_image_anchored_flow(&items, &images, Some(270.0)).unwrap();
let graph = build_region_graph(items, band);
assert_eq!(graph[0].kind, RegionKind::FullWidth);
assert_eq!(graph[1].kind, RegionKind::Column);
assert_eq!(graph[2].kind, RegionKind::Column);
assert_eq!(graph[3].kind, RegionKind::FullWidth);
assert_eq!(graph[3].items[0].text, "trailing full width caption");
}
#[test]
fn rtl_region_graph_reads_right_column_first() {
let items = vec![
item("A long English report header", 55.0, 250.0, 430.0),
item("نص العمود الأيسر", 55.0, 150.0, 180.0),
item("نص العمود الأيمن", 300.0, 150.0, 180.0),
];
let graph = build_region_graph(
items,
ColumnFlowBand {
split_x: 270.0,
y_bottom: 100.0,
y_top: 200.0,
},
);
assert_eq!(graph.len(), 3);
assert_eq!(graph[0].kind, RegionKind::FullWidth);
assert!(graph[1].items[0].x > graph[2].items[0].x);
}
#[test]
fn paired_header_logos_do_not_anchor_page_columns() {
let mut items = Vec::new();
for index in 0..7 {
items.push(item(
"left prose words",
55.0,
700.0 - index as f32 * 14.0,
200.0,
));
}
for index in 0..30 {
items.push(item(
"right prose words",
280.0,
700.0 - index as f32 * 14.0,
200.0,
));
}
let images = vec![
(55.0, 720.0, 205.0, 770.0),
(60.0, 718.0, 210.0, 768.0),
(280.0, 720.0, 450.0, 770.0),
];
assert!(infer_image_anchored_flow(&items, &images, Some(270.0)).is_none());
}
#[test]
fn wide_banner_does_not_anchor_local_columns() {
let mut items = Vec::new();
for index in 0..7 {
let y = 270.0 - index as f32 * 14.0;
items.push(item("left column prose words", 55.0, y, 210.0));
items.push(item("right column prose words", 280.0, y, 210.0));
}
let images = vec![(55.0, 310.0, 490.0, 550.0)];
assert!(infer_image_anchored_flow(&items, &images, None).is_none());
}
}
+1 -1
View File
@@ -162,7 +162,7 @@ fn extract_form_xobject_text_inner(
// Get fonts from the Form's Resources
let form_fonts = get_form_fonts(doc, &stream.dict);
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
// Build font width info for the form
let font_widths = build_font_widths(doc, &form_fonts);
+6 -40
View File
@@ -68,40 +68,6 @@ use text_quality::{
};
use tounicode::FontCMaps;
#[cfg(not(target_arch = "wasm32"))]
struct ProcessingTimer(std::time::Instant);
#[cfg(target_arch = "wasm32")]
struct ProcessingTimer;
impl ProcessingTimer {
fn start() -> Self {
#[cfg(not(target_arch = "wasm32"))]
{
Self(std::time::Instant::now())
}
#[cfg(target_arch = "wasm32")]
{
Self
}
}
fn elapsed_ms(&self) -> u64 {
#[cfg(not(target_arch = "wasm32"))]
{
self.0.elapsed().as_millis() as u64
}
#[cfg(target_arch = "wasm32")]
{
// The wasm32-unknown-unknown standard library has no clock.
// Browser bindings measure with JavaScript's host clock.
0
}
}
}
/// OCR reason emitted when the extracted text layer appears garbled due to
/// broken font decoding or mojibake.
pub const OCR_REASON_SUSPECTED_GARBLED_TEXT: &str = "suspected_garbled_text";
@@ -284,7 +250,7 @@ pub fn process_pdf_with_options<P: AsRef<Path>>(
path: P,
options: PdfOptions,
) -> Result<PdfProcessResult, PdfError> {
let start = ProcessingTimer::start();
let start = std::time::Instant::now();
validate_pdf_file(&path)?;
// Load the document once — shared by detection AND extraction.
@@ -311,7 +277,7 @@ pub fn process_pdf_mem_with_options(
buffer: &[u8],
options: PdfOptions,
) -> Result<PdfProcessResult, PdfError> {
let start = ProcessingTimer::start();
let start = std::time::Instant::now();
validate_pdf_bytes(buffer)?;
let (doc, page_count) =
@@ -3560,7 +3526,7 @@ fn process_document(
doc: Document,
page_count: u32,
options: PdfOptions,
start: ProcessingTimer,
start: std::time::Instant,
) -> Result<PdfProcessResult, PdfError> {
// Step 1 — Detection (cheap: scans content streams for text operators)
let detection = detector::detect_from_document(&doc, page_count, &options.detection)?;
@@ -3576,7 +3542,7 @@ fn process_document(
pdf_type,
markdown: None,
page_count,
processing_time_ms: start.elapsed_ms(),
processing_time_ms: start.elapsed().as_millis() as u64,
pages_needing_ocr,
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
title,
@@ -3592,7 +3558,7 @@ fn process_document(
pdf_type,
markdown: None,
page_count,
processing_time_ms: start.elapsed_ms(),
processing_time_ms: start.elapsed().as_millis() as u64,
pages_needing_ocr,
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
title,
@@ -3858,7 +3824,7 @@ fn process_document(
pdf_type,
markdown,
page_count,
processing_time_ms: start.elapsed_ms(),
processing_time_ms: start.elapsed().as_millis() as u64,
pages_needing_ocr,
ocr_reasons_by_page: {
// Detector reasons (scanned / no_text / vector_text / garbled) merged
+2 -11
View File
@@ -1012,19 +1012,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
// Separate images and links from text items
let mut images: Vec<TextItem> = Vec::new();
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
let mut links: Vec<TextItem> = Vec::new();
let mut text_items: Vec<TextItem> = Vec::new();
for item in items {
match &item.item_type {
ItemType::Image => {
page_image_regions.entry(item.page).or_default().push((
item.x,
item.y,
item.x + item.width,
item.y + item.height,
));
if options.include_images {
images.push(item);
}
@@ -1662,12 +1655,11 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
// items from different side-by-side zones (e.g. left/right month columns
// in a calendar) don't merge into the same line.
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
crate::extractor::group_into_lines_with_thresholds_and_regions(
crate::extractor::group_into_lines_with_thresholds_and_charts(
non_table_items,
page_thresholds,
&table_page_set,
&page_chart_map,
&page_image_regions,
)
} else {
// Separate items into physical-band pages, chart/prose pages, and
@@ -1690,12 +1682,11 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
}
}
// Process unsplit pages normally
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_regions(
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_charts(
unsplit_items,
page_thresholds,
&table_page_set,
&page_chart_map,
&page_image_regions,
);
// Process each split page's bands independently, then interleave
// by Y position so paired zones (e.g. left/right months) appear together.
+10 -23
View File
@@ -4,17 +4,11 @@
use log::{debug, warn};
use lopdf::{Document, Object, ObjectId};
use std::borrow::Cow;
use std::collections::{HashMap, HashSet};
#[cfg(not(target_arch = "wasm32"))]
use std::path::{Path, PathBuf};
use crate::glyph_names::glyph_to_char;
#[cfg(target_arch = "wasm32")]
static BUILTIN_CMAPS: include_dir::Dir<'_> =
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
#[derive(Debug, Default, Clone)]
pub struct ToUnicodeCMap {
@@ -1155,7 +1149,9 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
let data = read_builtin_cmap_file(&name)?;
let dir = find_bcmaps_dir()?;
let path = dir.join(name);
let data = std::fs::read(&path).ok()?;
let mut cmap = parse_binary_cmap(&data).ok()?;
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
return None;
@@ -1163,14 +1159,13 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
cmap.code_byte_length = 2;
debug!(
"Built-in CMap {}: char_map={} ranges={}",
name,
path.display(),
cmap.char_map.len(),
cmap.ranges.len()
);
Some(cmap)
}
#[cfg(not(target_arch = "wasm32"))]
fn find_bcmaps_dir() -> Option<PathBuf> {
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
let p = PathBuf::from(dir);
@@ -1187,18 +1182,6 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
None
}
#[cfg(not(target_arch = "wasm32"))]
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
let path = find_bcmaps_dir()?.join(name);
std::fs::read(path).ok().map(Cow::Owned)
}
#[cfg(target_arch = "wasm32")]
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
let file = BUILTIN_CMAPS.get_file(name)?;
Some(Cow::Borrowed(file.contents()))
}
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
let mut stream = BinaryCMapStream::new(data);
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
@@ -1509,7 +1492,9 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
}
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
let dir = find_bcmaps_dir()?;
let path = dir.join(format!("{}.bcmap", name));
let data = std::fs::read(&path).ok()?;
parse_binary_cmap_encoding(&data).ok()
}
@@ -1794,7 +1779,9 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
if !name.ends_with("UCS2") {
return None;
}
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
let dir = find_bcmaps_dir()?;
let path = dir.join(format!("{}.bcmap", name));
let data = std::fs::read(&path).ok()?;
let mut cmap = parse_binary_cmap(&data).ok()?;
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
return None;
-1304
View File
File diff suppressed because it is too large Load Diff
-37
View File
@@ -1,37 +0,0 @@
[package]
name = "pdf-inspector-wasm"
version = "0.1.2"
edition = "2021"
authors = ["Firecrawl Team"]
description = "Browser WebAssembly bindings for pdf-inspector"
license = "MIT"
repository = "https://github.com/firecrawl/pdf-inspector"
homepage = "https://github.com/firecrawl/pdf-inspector"
readme = "README.md"
publish = false
[lib]
crate-type = ["cdylib", "rlib"]
[dependencies]
console_error_panic_hook = "0.1"
js-sys = "0.3"
pdf-inspector = { path = ".." }
serde = { version = "1", features = ["derive"] }
serde-wasm-bindgen = "0.6"
wasm-bindgen = "0.2"
[dev-dependencies]
wasm-bindgen-test = "0.3"
[profile.release]
codegen-units = 1
lto = true
opt-level = "s"
strip = true
[package.metadata.wasm-pack.profile.release]
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
# size, and LTO optimizations above.
wasm-opt = false
-58
View File
@@ -1,58 +0,0 @@
MIT License
Copyright (c) 2026 Firecrawl
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
Third-party notices
===================
Adobe CMaps
-----------
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
Copyright 1990-2009 Adobe Systems Incorporated.
All rights reserved.
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the following conditions are met:
Redistributions of source code must retain the above copyright notice, this
list of conditions and the following disclaimer.
Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimer in the documentation
and/or other materials provided with the distribution.
Neither the name of Adobe Systems Incorporated nor the names of its
contributors may be used to endorse or promote products derived from this
software without specific prior written permission.
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
POSSIBILITY OF SUCH DAMAGE.
-60
View File
@@ -1,60 +0,0 @@
# @firecrawl/pdf-inspector-wasm
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
## Install
```bash
npm install @firecrawl/pdf-inspector-wasm
```
## Usage
```ts
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
await init();
const response = await fetch("/annual-report.pdf");
const pdf = new Uint8Array(await response.arrayBuffer());
const result = processPdf(pdf);
console.log(result.pdfType);
console.log(result.markdown);
```
Pass options when you need selected pages or compact Markdown:
```ts
const result = processPdf(pdf, {
pages: [1, 3, 5],
profile: "compact",
includePageMarkers: true,
});
```
The package also exports:
- `detectPdf(pdf, options?)` for detection without extraction.
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
- `extractText(pdf)` for plain text.
- `version()` for the WASM package version.
## Browser behavior
- Parsing runs locally. PDF bytes are not uploaded anywhere.
- The build is single-threaded and does not require cross-origin isolation.
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
- Image-only documents still require a separate OCR step.
## Build from source
```bash
cargo install wasm-pack --version 0.15.0 --locked
wasm-pack build wasm --target web --scope firecrawl --release
```
## License
MIT
-440
View File
@@ -1,440 +0,0 @@
use pdf_inspector::{
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
ProcessMode,
};
use serde::{Deserialize, Serialize};
use wasm_bindgen::prelude::*;
#[wasm_bindgen(typescript_custom_section)]
const TYPESCRIPT_TYPES: &str = r#"
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
export type MarkdownProfile = "fidelity" | "compact";
export interface ProcessOptions {
/** Restrict extraction to these 1-indexed page numbers. */
pages?: number[];
/** Password for an encrypted PDF. */
password?: string;
/** Source-faithful output by default, or compact output for fewer tokens. */
profile?: MarkdownProfile;
/** Insert `<!-- Page N -->` markers between pages. */
includePageMarkers?: boolean;
/** Include image placeholders in Markdown output. */
includeImages?: boolean;
}
export interface PageOcrReasons {
/** 1-indexed page number. */
page: number;
reasons: string[];
}
export interface LayoutComplexity {
isComplex: boolean;
/** 1-indexed page numbers. */
pagesWithTables: number[];
/** 1-indexed page numbers. */
pagesWithColumns: number[];
}
export interface PdfProcessResult {
pdfType: PdfType;
markdown?: string;
pageCount: number;
processingTimeMs: number;
/** 1-indexed page numbers. */
pagesNeedingOcr: number[];
ocrReasonsByPage: PageOcrReasons[];
title?: string;
confidence: number;
layout: LayoutComplexity;
hasEncodingIssues: boolean;
}
export interface PdfClassification {
pdfType: PdfType;
pageCount: number;
/** 0-indexed page numbers, matching the native Node.js API. */
pagesNeedingOcr: number[];
confidence: number;
}
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
export function classifyPdf(data: Uint8Array): PdfClassification;
export function extractText(data: Uint8Array): string;
export function version(): string;
"#;
#[derive(Debug, Default, Deserialize)]
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
struct WasmProcessOptions {
pages: Option<Vec<u32>>,
password: Option<String>,
profile: Option<WasmMarkdownProfile>,
include_page_markers: Option<bool>,
include_images: Option<bool>,
}
#[derive(Debug, Deserialize)]
#[serde(rename_all = "lowercase")]
enum WasmMarkdownProfile {
Fidelity,
Compact,
}
#[derive(Serialize)]
#[serde(rename_all = "camelCase")]
struct WasmPageOcrReasons {
page: u32,
reasons: Vec<String>,
}
impl From<PageOcrReasons> for WasmPageOcrReasons {
fn from(value: PageOcrReasons) -> Self {
Self {
page: value.page,
reasons: value.reasons,
}
}
}
#[derive(Serialize)]
#[serde(rename_all = "camelCase")]
struct WasmLayoutComplexity {
is_complex: bool,
pages_with_tables: Vec<u32>,
pages_with_columns: Vec<u32>,
}
impl From<LayoutComplexity> for WasmLayoutComplexity {
fn from(value: LayoutComplexity) -> Self {
Self {
is_complex: value.is_complex,
pages_with_tables: value.pages_with_tables,
pages_with_columns: value.pages_with_columns,
}
}
}
#[derive(Serialize)]
#[serde(rename_all = "camelCase")]
struct WasmPdfProcessResult {
pdf_type: &'static str,
markdown: Option<String>,
page_count: u32,
processing_time_ms: f64,
pages_needing_ocr: Vec<u32>,
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
title: Option<String>,
confidence: f64,
layout: WasmLayoutComplexity,
has_encoding_issues: bool,
}
impl From<PdfProcessResult> for WasmPdfProcessResult {
fn from(value: PdfProcessResult) -> Self {
Self {
pdf_type: pdf_type_name(value.pdf_type),
markdown: value.markdown,
page_count: value.page_count,
processing_time_ms: value.processing_time_ms as f64,
pages_needing_ocr: value.pages_needing_ocr,
ocr_reasons_by_page: value
.ocr_reasons_by_page
.into_iter()
.map(Into::into)
.collect(),
title: value.title,
confidence: value.confidence as f64,
layout: value.layout.into(),
has_encoding_issues: value.has_encoding_issues,
}
}
}
#[derive(Serialize)]
#[serde(rename_all = "camelCase")]
struct WasmPdfClassification {
pdf_type: &'static str,
page_count: u32,
pages_needing_ocr: Vec<u32>,
confidence: f64,
}
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
match pdf_type {
PdfType::TextBased => "TextBased",
PdfType::Scanned => "Scanned",
PdfType::ImageBased => "ImageBased",
PdfType::Mixed => "Mixed",
}
}
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
js_sys::Error::new(&format!("{context}: {error}")).into()
}
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
if value.is_undefined() || value.is_null() {
return Ok(WasmProcessOptions::default());
}
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
}
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
let options = deserialize_options(value)?;
if options
.pages
.as_ref()
.is_some_and(|pages| pages.contains(&0))
{
return Err(js_error(
"invalid options",
"pages are 1-indexed; page 0 is invalid",
));
}
let mut result = PdfOptions::new().mode(mode);
if let Some(pages) = options.pages {
result = result.pages(pages);
}
if let Some(password) = options.password {
result = result.password(password);
}
if let Some(profile) = options.profile {
result.markdown.profile = match profile {
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
};
}
if let Some(include_page_markers) = options.include_page_markers {
result.markdown.include_page_numbers = include_page_markers;
}
if let Some(include_images) = options.include_images {
result.markdown.include_images = include_images;
}
Ok(result)
}
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
}
fn initialize() {
console_error_panic_hook::set_once();
}
/// Process PDF bytes entirely inside WebAssembly.
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
initialize();
let options = build_options(options, ProcessMode::Full)?;
let started = js_sys::Date::now();
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
.map_err(|error| js_error("process PDF", error))?;
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
serialize(&WasmPdfProcessResult::from(result))
}
/// Classify PDF bytes without extracting text or producing Markdown.
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
initialize();
let options = build_options(options, ProcessMode::DetectOnly)?;
let started = js_sys::Date::now();
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
.map_err(|error| js_error("detect PDF", error))?;
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
serialize(&WasmPdfProcessResult::from(result))
}
/// Return the lightweight classification shape used by the native Node API.
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
initialize();
let result =
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
serialize(&WasmPdfClassification {
pdf_type: pdf_type_name(result.pdf_type),
page_count: result.page_count,
pages_needing_ocr: result.pages_needing_ocr,
confidence: result.confidence as f64,
})
}
/// Extract plain text from PDF bytes without Markdown conversion.
#[wasm_bindgen(js_name = extractText, skip_typescript)]
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
initialize();
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
.map_err(|error| js_error("extract text", error))?;
Ok(
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
.into_iter()
.map(|line| line.text())
.filter(|line| !line.trim().is_empty())
.collect::<Vec<_>>()
.join("\n"),
)
}
/// Return the WebAssembly package version.
#[wasm_bindgen(skip_typescript)]
pub fn version() -> String {
env!("CARGO_PKG_VERSION").to_string()
}
#[cfg(all(test, target_arch = "wasm32"))]
mod tests {
use super::*;
use js_sys::Reflect;
use wasm_bindgen_test::*;
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
fn synthetic_korea1_pdf() -> Vec<u8> {
let mut pdf = b"%PDF-1.4\n".to_vec();
let mut offsets = vec![0usize];
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
offsets.push(pdf.len());
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
pdf.extend_from_slice(body.as_bytes());
pdf.extend_from_slice(b"\nendobj\n");
}
add_object(
&mut pdf,
&mut offsets,
1,
"<< /Type /Catalog /Pages 2 0 R >>",
);
add_object(
&mut pdf,
&mut offsets,
2,
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
);
add_object(
&mut pdf,
&mut offsets,
3,
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
);
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
// There is deliberately no ToUnicode stream: decoding must use the
// embedded predefined CMap rather than lopdf's plain-text fallback.
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
// the bottom edge so they look exactly like a numeric page footer.
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
add_object(
&mut pdf,
&mut offsets,
4,
&format!(
"<< /Length {} >>\nstream\n{}\nendstream",
content.len(),
content
),
);
add_object(
&mut pdf,
&mut offsets,
5,
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
);
add_object(
&mut pdf,
&mut offsets,
6,
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
);
add_object(
&mut pdf,
&mut offsets,
7,
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
);
let xref_start = pdf.len();
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
pdf.extend_from_slice(b"0000000000 65535 f \n");
for offset in offsets.iter().skip(1) {
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
}
pdf.extend_from_slice(
format!(
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
offsets.len(),
xref_start
)
.as_bytes(),
);
pdf
}
#[wasm_bindgen_test]
fn processes_pdf_to_markdown() {
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
.expect("pdfType")
.as_string()
.expect("pdfType string");
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
.expect("markdown")
.as_string()
.expect("markdown string");
assert_eq!(pdf_type, "TextBased");
assert!(!markdown.is_empty());
}
#[wasm_bindgen_test]
fn rejects_non_pdf_bytes() {
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
}
#[wasm_bindgen_test]
fn classifies_and_extracts_plain_text() {
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
.expect("pdfType")
.as_string()
.expect("pdfType string");
let text = extract_text(TEXT_PDF).expect("extract text");
assert_eq!(pdf_type, "TextBased");
assert!(!text.is_empty());
}
#[wasm_bindgen_test]
fn extracts_cjk_and_preserves_numeric_page_footer() {
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
assert_eq!(text, "\n42");
}
#[wasm_bindgen_test]
fn opens_encrypted_pdf_with_password() {
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
let options = js_sys::Object::new();
Reflect::set(
&options,
&JsValue::from_str("password"),
&JsValue::from_str("secret123"),
)
.expect("set password");
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
.expect("markdown")
.as_string()
.expect("markdown string");
assert!(!markdown.is_empty());
}
}