Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
35e244fde9 | ||
|
|
3327ccddab | ||
|
|
228a3bfd33 | ||
|
|
37c3670bdd | ||
|
|
04324bfeb4 |
+11
-61
@@ -14,57 +14,44 @@ jobs:
|
||||
name: Test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
|
||||
- name: Check package version sync
|
||||
run: python3 scripts/version.py --check
|
||||
uses: Swatinem/rust-cache@v2
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
|
||||
- name: Test developer scripts
|
||||
run: python3 -m unittest discover -s scripts/tests
|
||||
|
||||
fmt:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: rustfmt
|
||||
|
||||
- name: Check formatting
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Check WASM formatting
|
||||
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
|
||||
|
||||
clippy:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
key: clippy
|
||||
|
||||
@@ -78,52 +65,15 @@ jobs:
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
key: build
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
key: wasm
|
||||
|
||||
- name: Check WebAssembly bindings
|
||||
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
|
||||
|
||||
- name: Check root package for WebAssembly
|
||||
run: cargo check --target wasm32-unknown-unknown
|
||||
|
||||
- name: Lint WebAssembly bindings
|
||||
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Test WebAssembly package
|
||||
run: wasm-pack test --node --release wasm
|
||||
|
||||
@@ -24,13 +24,13 @@ jobs:
|
||||
name: github-pages
|
||||
url: ${{ steps.deploy.outputs.page_url }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Upload site artifact
|
||||
uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deploy
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5.0.0
|
||||
uses: actions/deploy-pages@v4
|
||||
|
||||
@@ -27,13 +27,10 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version sync
|
||||
run: python3 scripts/version.py --check
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
@@ -88,19 +85,17 @@ jobs:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@c6f97d42243bad5fab37ca0427f495c86d5b1a18 # v1.0.5
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
|
||||
@@ -24,13 +24,10 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version sync
|
||||
run: python3 scripts/version.py --check
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
@@ -101,21 +98,21 @@ jobs:
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Build wheel
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
uses: PyO3/maturin-action@v1
|
||||
with:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: wheels-${{ matrix.target }}
|
||||
path: dist/*.whl
|
||||
@@ -127,16 +124,16 @@ jobs:
|
||||
name: Build sdist
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Build sdist
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
uses: PyO3/maturin-action@v1
|
||||
with:
|
||||
command: sdist
|
||||
args: --out dist
|
||||
|
||||
- name: Upload sdist
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: sdist
|
||||
path: dist/*.tar.gz
|
||||
@@ -152,7 +149,7 @@ jobs:
|
||||
id-token: write
|
||||
steps:
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
@@ -161,7 +158,7 @@ jobs:
|
||||
run: ls -la dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: dist
|
||||
# Tolerate already-uploaded files so a manual re-run can complete
|
||||
|
||||
@@ -1,115 +0,0 @@
|
||||
name: Publish WebAssembly package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['wasm/Cargo.toml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
package_exists: ${{ steps.check.outputs.package_exists }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version sync
|
||||
run: python3 scripts/version.py --check
|
||||
|
||||
- name: Check package version
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
|
||||
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
|
||||
echo "package_exists=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
echo "The initial package must be published once before trusted publishing can be configured."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "package_exists=true" >> "$GITHUB_OUTPUT"
|
||||
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
publish:
|
||||
name: Build and publish
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Build browser package
|
||||
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
|
||||
- name: Prepare package metadata
|
||||
run: |
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const path = "wasm/pkg/package.json"
|
||||
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
|
||||
pkg.name = "@firecrawl/pdf-inspector-wasm"
|
||||
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
|
||||
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
|
||||
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
|
||||
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
|
||||
pkg.publishConfig = { access: "public" }
|
||||
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
- name: Inspect package contents
|
||||
run: npm pack --dry-run ./wasm/pkg
|
||||
|
||||
- name: Publish package
|
||||
run: npm publish ./wasm/pkg --provenance --access public
|
||||
@@ -24,13 +24,10 @@ jobs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version sync
|
||||
run: python3 scripts/version.py --check
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
@@ -64,61 +61,31 @@ jobs:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
# napi-cross builds gnu targets against an old glibc sysroot for
|
||||
# broad distro compatibility; musl targets cross-compile with
|
||||
# zig via cargo-zigbuild (napi's -x flag).
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
build-flags: --use-napi-cross
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: ${{ matrix.target }}
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install zig
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: mlugg/setup-zig@d1434d08867e3ee9daa34448df10607b98908d29 # v2.2.1
|
||||
with:
|
||||
version: 0.14.1
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: taiki-e/install-action@67729d5c413db75907f0ad1e39bb04b9c868ff60 # v2.85.7
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
tool: cargo-zigbuild
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
~/.napi-rs
|
||||
napi/target/
|
||||
key: ${{ matrix.target }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
key: ${{ runner.os }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ matrix.target }}-cargo-napi-
|
||||
${{ runner.os }}-cargo-napi-
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: napi
|
||||
@@ -126,10 +93,10 @@ jobs:
|
||||
|
||||
- name: Build native addon
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
|
||||
run: bunx napi build --platform --release
|
||||
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
@@ -137,7 +104,7 @@ jobs:
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
@@ -145,68 +112,23 @@ jobs:
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
smoke-test:
|
||||
name: Smoke test ${{ matrix.target }}
|
||||
needs: [check-version, build]
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-gnu
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-musl
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Download native binary
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi
|
||||
|
||||
- name: Download generated JS bindings
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: napi
|
||||
|
||||
# musl binaries must load under a real musl libc, so run inside Alpine.
|
||||
- name: Run smoke test (Alpine)
|
||||
if: contains(matrix.target, 'musl')
|
||||
run: docker run --rm -v "$PWD:/repo" -w /repo/napi node:24-alpine node test.mjs
|
||||
|
||||
- name: Run smoke test
|
||||
if: ${{ !contains(matrix.target, 'musl') }}
|
||||
working-directory: napi
|
||||
run: node test.mjs
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: [check-version, build, smoke-test]
|
||||
needs: [check-version, build]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
- uses: actions/checkout@v6
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
@@ -232,12 +154,9 @@ jobs:
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"linux-x64-musl": { os: ["linux"], cpu: ["x64"], libc: ["musl"] },
|
||||
"linux-arm64-gnu": { os: ["linux"], cpu: ["arm64"], libc: ["glibc"] },
|
||||
"linux-arm64-musl": { os: ["linux"], cpu: ["arm64"], libc: ["musl"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
|
||||
+3
-5
@@ -1,13 +1,10 @@
|
||||
# Rust build artifacts
|
||||
/target/
|
||||
/wasm/target/
|
||||
/wasm/pkg/
|
||||
debug/
|
||||
*.pdb
|
||||
|
||||
# Cargo lock (optional for libraries)
|
||||
Cargo.lock
|
||||
!/wasm/Cargo.lock
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
@@ -31,14 +28,15 @@ Thumbs.db
|
||||
napi/index.js
|
||||
napi/index.d.ts
|
||||
|
||||
# Local samples
|
||||
# Local samples and scripts
|
||||
samples/
|
||||
scripts/
|
||||
|
||||
# Test output
|
||||
test_output/
|
||||
.firecrawl/
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
@@ -61,8 +61,7 @@ src/
|
||||
|
||||
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
|
||||
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with ~200 snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing. While iterating, prefer a subset run (`bench.py test -q` for the quick set, or `-s <name>` for a named test set) and save the full `bench.py test` for the final pre-commit check.
|
||||
- **Semantic quality**: run `bench.py score` in `pdf-evals` for the semantic verdict (TEDS + MHS + reading order + char/word + list preservation, composited). Character-level diff alone misclassifies structural improvements (e.g., column-detection rewrites) as regressions — `score` is the tie-breaker. See `pdf-evals/CLAUDE.md` "Semantic scoring".
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 179+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
|
||||
## Debugging
|
||||
|
||||
|
||||
@@ -61,7 +61,7 @@ src/
|
||||
|
||||
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
|
||||
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with ~200 snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing. While iterating, prefer a subset run (`bench.py test -q` for the quick set, or `-s <name>` for a named test set) and save the full `bench.py test` for the final pre-commit check.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 187+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
- **Semantic quality**: run `bench.py score` in `pdf-evals` for the semantic verdict (TEDS + MHS + reading order + char/word + list preservation, composited). Character-level diff alone misclassifies structural improvements (e.g., column-detection rewrites) as regressions — `score` is the tie-breaker. See `pdf-evals/CLAUDE.md` "Semantic scoring".
|
||||
|
||||
## Debugging
|
||||
|
||||
+13
-19
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "1.14.0"
|
||||
version = "0.1.6"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
@@ -12,13 +12,13 @@ readme = "docs/rust-api.md"
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"/src/**",
|
||||
"/external/bcmaps/**",
|
||||
"/docs/rust-api.md",
|
||||
"/LICENSE",
|
||||
"src/**",
|
||||
"external/bcmaps/**",
|
||||
"docs/rust-api.md",
|
||||
"LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"/pdf_inspector.pyi",
|
||||
"pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
@@ -29,11 +29,18 @@ crate-type = ["lib", "cdylib"]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
# Parallel processing
|
||||
rayon = "1.10"
|
||||
|
||||
# Logging
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Text processing
|
||||
regex = "1.10"
|
||||
@@ -43,19 +50,6 @@ unicode-normalization = "0.1"
|
||||
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
||||
ttf-parser = "0.25"
|
||||
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.42.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.42.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.3"
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -19,28 +19,24 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only local engines without model-based PDF parsing are shown; OCR was disabled. Scores are 0-1, higher is better.
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
| pdf-inspector | 0.83 | 0.89 | 0.66 | 0.74 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
|
||||
Results were refreshed on July 31, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Speed is the median of five alternating or rotating complete corpus runs after an excluded warm-up run, with each parser processing documents sequentially in a single process.
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the low end of that range without any OCR, in 4 seconds.
|
||||
|
||||
The complete parser configuration, per-document predictions, evaluator output, and generated charts are available in the [reproducible results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
**Where we do well:** Speed (fastest of all engines), the best table detection of any engine shown, and heading detection now on par with opendataloader. Overall lands within 0.01 of opendataloader at roughly 2.5× the speed.
|
||||
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. In this comparison, pdf-inspector delivered the higher overall, reading-order, and table scores, along with the fastest complete run. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
|
||||
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
|
||||
**Where we lag:** Reading order still trails opendataloader slightly, and table structure trails OCR-based engines that can see visual layout.
|
||||
|
||||
## Quick start
|
||||
|
||||
@@ -78,26 +74,6 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Browser WebAssembly
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
```javascript
|
||||
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
|
||||
|
||||
await init();
|
||||
const response = await fetch('/document.pdf');
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
> Full API reference: [wasm/README.md](wasm/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
@@ -110,7 +86,7 @@ Or add it manually:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = "1"
|
||||
pdf-inspector = "0.1"
|
||||
```
|
||||
|
||||
```rust
|
||||
@@ -209,7 +185,6 @@ src/
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
wasm/ — Browser bindings (wasm-bindgen)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
@@ -238,7 +213,7 @@ The converter handles:
|
||||
|---|---|
|
||||
| Headings (H1-H4) | Font size tiers relative to body text, with 0.5pt clustering |
|
||||
| Bold/italic | Font name patterns (Bold, Italic, Oblique) |
|
||||
| Bullet lists | `•`, `-`, `*`, `○`, `●`, `◦` prefixes |
|
||||
| Bullet lists | `*`, `-`, `*`, `○`, `●`, `◦` prefixes |
|
||||
| Numbered lists | `1.`, `1)`, `(1)` patterns |
|
||||
| Letter lists | `a.`, `a)`, `(a)` patterns |
|
||||
| Code blocks | Monospace fonts (Courier, Consolas, Monaco, Menlo, Fira Code, JetBrains Mono) and keyword detection |
|
||||
|
||||
+3
-4
@@ -5,15 +5,14 @@
|
||||
If you believe you've found a security vulnerability in pdf-inspector, please
|
||||
report it privately so we can fix it before public disclosure.
|
||||
|
||||
**Preferred:** Submit through Firecrawl's Bugcrowd vulnerability disclosure
|
||||
program at <https://bugcrowd.com/engagements/firecrawl-vdp-ess>. Please include:
|
||||
**Preferred:** Email **help@firecrawl.dev** with:
|
||||
|
||||
- A description of the issue and its impact
|
||||
- Steps to reproduce (a minimal PDF or input that triggers the bug is ideal)
|
||||
- The version or commit hash of pdf-inspector you tested against
|
||||
|
||||
**Alternative:** If you'd rather not use Bugcrowd, email
|
||||
**help@firecrawl.dev** with the same details.
|
||||
**Alternative:** Use GitHub's private vulnerability reporting under the
|
||||
[Security tab](https://github.com/firecrawl/pdf-inspector/security/advisories/new).
|
||||
|
||||
We'll acknowledge your report in a timely manner and keep you updated on
|
||||
remediation progress. Please do not open a public GitHub issue for security
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
# Benchmarking against OpenDataLoader
|
||||
|
||||
The paired harness runs two `pdf2md` binaries through the same local
|
||||
OpenDataLoader corpus, evaluates both outputs, and reports aggregate and
|
||||
per-document deltas. This avoids comparing results produced from different
|
||||
corpus revisions or evaluator versions.
|
||||
|
||||
Build a candidate and provide a released or worktree build as the baseline:
|
||||
|
||||
```bash
|
||||
cargo build --release
|
||||
python3 scripts/bench_opendataloader.py \
|
||||
--bench-dir ../opendataloader-bench \
|
||||
--baseline ../pdf-inspector-main/target/release/pdf2md \
|
||||
--candidate target/release/pdf2md \
|
||||
--max-document-regression 0.02 \
|
||||
--json-output /tmp/pdf-inspector-benchmark.json
|
||||
```
|
||||
|
||||
Pass `--reference-evaluation path/to/evaluation.json` to report the candidate
|
||||
delta against another evaluation, and add `--require-reference-lead` to make a
|
||||
negative reference delta fail the run. By default, the candidate must not
|
||||
regress the baseline overall score or introduce missing predictions. Use
|
||||
`--min-overall-delta` to require a specific aggregate gain.
|
||||
|
||||
The OpenDataLoader repository is external and keeps its normal
|
||||
`prediction/pdf-inspector` output. Paired evaluation copies each run into a
|
||||
temporary directory before evaluating it, so the baseline and candidate cannot
|
||||
overwrite one another.
|
||||
|
||||
## Published comparison protocol
|
||||
|
||||
The public benchmark table was refreshed on July 31, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Every engine processed the same 200
|
||||
PDFs sequentially in a single process with OCR disabled. Reported speed is the
|
||||
median of five alternating or rotating complete corpus runs after an excluded
|
||||
warm-up run; quality scores come from the benchmark evaluator over all 200
|
||||
outputs. Raw timings, predictions, evaluations, and charts are available in the
|
||||
[results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Optional backend evidence probe
|
||||
|
||||
The evidence probe compares positioned `pdf2md` items with MuPDF structured
|
||||
text on the same pages. It is intended to find deterministic extraction or
|
||||
layout evidence that could justify a future native implementation; it does not
|
||||
merge MuPDF output into Markdown, invoke OCR, or add a runtime dependency.
|
||||
|
||||
Install MuPDF's `mutool`, build `pdf2md`, then run:
|
||||
|
||||
```bash
|
||||
python3 scripts/probe_backend_evidence.py document.pdf \
|
||||
--pdf2md target/release/pdf2md \
|
||||
--json-output /tmp/backend-evidence.json
|
||||
```
|
||||
|
||||
The report flags pages when MuPDF exposes a material net token gain, repeated
|
||||
alignment anchors absent from local evidence, or additional image blocks. The
|
||||
JSON includes bounded token samples and page-level counts so promising cases
|
||||
can be inspected without treating backend disagreement as automatically
|
||||
correct. Thresholds are configurable with `--min-token-gain`,
|
||||
`--min-alternate-only-ratio`, and `--min-anchor-gain`.
|
||||
+12
-48
@@ -1,57 +1,21 @@
|
||||
# Publishing
|
||||
|
||||
Every pdf-inspector distribution uses one shared semantic version:
|
||||
The Rust crate is published to [crates.io](https://crates.io/crates/pdf-inspector) with trusted publishing from GitHub Actions. The first release was published manually; future releases publish from `.github/workflows/publish-crate.yml` when a `Cargo.toml` version change lands on `main`.
|
||||
|
||||
- Rust crate: `pdf-inspector`
|
||||
- Python package: `pdf-inspector`
|
||||
- Node package: `@firecrawl/pdf-inspector` and its platform packages
|
||||
- Browser package: `@firecrawl/pdf-inspector-wasm`
|
||||
- Internal NAPI and WASM Rust crates
|
||||
## crates.io Trusted Publisher
|
||||
|
||||
`Cargo.toml` is the canonical version source. Update every manifest and lockfile
|
||||
with:
|
||||
Configure the trusted publisher for the `pdf-inspector` crate with:
|
||||
|
||||
```bash
|
||||
python3 scripts/version.py <version>
|
||||
```
|
||||
- Repository: `firecrawl/pdf-inspector`
|
||||
- Workflow: `publish-crate.yml`
|
||||
- Environment: `crates-io`
|
||||
|
||||
Verify that nothing has diverged with:
|
||||
The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC token for a short-lived crates.io token, then passes it to `cargo publish`.
|
||||
|
||||
```bash
|
||||
python3 scripts/version.py --check
|
||||
```
|
||||
## Release Steps
|
||||
|
||||
CI and every publishing workflow run this check before building or publishing.
|
||||
1. Update `version` in `Cargo.toml`.
|
||||
2. Merge the version bump to `main`.
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
## Release steps
|
||||
|
||||
1. Choose the next shared semantic version and run `scripts/version.py`.
|
||||
2. Review the manifest and lockfile changes in the version-bump pull request.
|
||||
3. Merge the pull request to `main`.
|
||||
4. The crates.io, PyPI, Node, and WASM workflows independently build and
|
||||
publish that version from the same commit.
|
||||
5. After all registries succeed, create one `v<version>` GitHub release that
|
||||
links to each package and describes changes since the previous shared tag.
|
||||
|
||||
The independent workflows are intentionally idempotent. A manual dispatch from
|
||||
`main` can repair a partial release, and already-published artifacts are skipped.
|
||||
|
||||
## Trusted publishers
|
||||
|
||||
The repositories use GitHub Actions OIDC instead of long-lived registry tokens.
|
||||
Configure each registry's trusted publisher for `firecrawl/pdf-inspector` and
|
||||
its corresponding workflow:
|
||||
|
||||
- crates.io: `publish-crate.yml`, environment `crates-io`
|
||||
- PyPI: `publish-pypi.yml`, environment `pypi`
|
||||
- npm Node package: `publish.yml`
|
||||
- npm WASM package: `publish-wasm.yml`
|
||||
|
||||
The WASM package must exist before npm trusted publishing can be configured. If
|
||||
it ever needs to be bootstrapped again, build and inspect it before publishing:
|
||||
|
||||
```bash
|
||||
wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
npm pack --dry-run ./wasm/pkg
|
||||
npm publish ./wasm/pkg --access public
|
||||
```
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
|
||||
+8
-40
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
@@ -80,17 +78,6 @@ for page in result.pages:
|
||||
|
||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
|
||||
# Structure-tree elements from tagged PDFs (empty list when untagged).
|
||||
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
|
||||
# against extract_text_with_positions — e.g. to recover real heading levels:
|
||||
elements = pdf_inspector.extract_structure_elements("tagged.pdf")
|
||||
roles = {(e.page, e.mcid): e.role for e in elements}
|
||||
headings = [
|
||||
item.text
|
||||
for item in pdf_inspector.extract_text_with_positions("tagged.pdf")
|
||||
if item.mcid is not None and roles.get((item.page, item.mcid), "").startswith("H")
|
||||
]
|
||||
```
|
||||
|
||||
## API reference
|
||||
@@ -111,8 +98,6 @@ headings = [
|
||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
|
||||
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
|
||||
| `extract_structure_elements(path, pages=None)` | Structure-tree elements from tagged PDFs (page, mcid, role) |
|
||||
| `extract_structure_elements_bytes(data, pages=None)` | Structure-tree elements from bytes |
|
||||
|
||||
## Types
|
||||
|
||||
@@ -124,8 +109,7 @@ class PdfResult: # process_pdf / detect_pdf
|
||||
markdown: str | None # extracted Markdown (None for detect_pdf)
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
pages_needing_ocr: list[int]
|
||||
title: str | None
|
||||
confidence: float # 0.0 - 1.0
|
||||
is_complex_layout: bool
|
||||
@@ -133,10 +117,6 @@ class PdfResult: # process_pdf / detect_pdf
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool # broken font encodings — consider OCR fallback
|
||||
|
||||
class PageOcrReasons: # per-page OCR diagnostics
|
||||
page: int # 1-indexed
|
||||
reasons: list[str] # machine-readable reason identifiers
|
||||
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
@@ -157,27 +137,15 @@ class TextItem: # extract_text_with_positions
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
mcid: int | None # marked-content ID for tagged PDFs (None otherwise)
|
||||
|
||||
class StructureElement: # extract_structure_elements
|
||||
page: int # 1-indexed (matches TextItem.page)
|
||||
mcid: int
|
||||
role: str # "H1".."H6", "P", "Table", ... (resolved via /RoleMap)
|
||||
|
||||
class RegionText: # extract_text_in_regions
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
ocr_reason: str | None # machine-readable OCR reason
|
||||
|
||||
class PageRegionTexts: # extract_text_in_regions
|
||||
page: int # 0-indexed
|
||||
regions: list[RegionText]
|
||||
regions: list[RegionText] # RegionText: text: str, needs_ocr: bool
|
||||
|
||||
class PagesExtractionResult: # extract_pages_markdown
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr, ocr_reason
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr
|
||||
pages_with_tables: list[int] # 1-indexed
|
||||
pages_with_columns: list[int] # 1-indexed
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
is_complex: bool # any page has tables or multi-column layout
|
||||
```
|
||||
|
||||
+6
-39
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
@@ -138,34 +136,6 @@ for page in &result.pages {
|
||||
println!("Complex layout? {}", result.is_complex);
|
||||
```
|
||||
|
||||
Extract structure-tree elements from tagged PDFs, and join them against
|
||||
`extract_text_with_positions` to attach semantic roles (heading levels,
|
||||
paragraphs, table cells) to extracted text:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::{extract_structure_elements, extract_text_with_positions};
|
||||
use std::collections::HashMap;
|
||||
|
||||
// One entry per marked-content reference, sorted by (page, mcid); empty for
|
||||
// untagged PDFs. Pages are 1-indexed to match `TextItem::page`, so the
|
||||
// (page, mcid) pair is a direct join key.
|
||||
let elements = extract_structure_elements("tagged.pdf", None)?;
|
||||
let roles: HashMap<(u32, i64), &str> = elements
|
||||
.iter()
|
||||
.map(|e| ((e.page, e.mcid), e.role.as_str()))
|
||||
.collect();
|
||||
|
||||
for item in extract_text_with_positions("tagged.pdf")? {
|
||||
if let Some(mcid) = item.mcid {
|
||||
if let Some(role) = roles.get(&(item.page, mcid)) {
|
||||
if role.starts_with('H') {
|
||||
println!("{}: {}", role, item.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
@@ -191,8 +161,6 @@ for item in extract_text_with_positions("tagged.pdf")? {
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
|
||||
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
|
||||
| `extract_structure_elements(path, pages)` | Structure-tree elements from tagged PDFs (page, mcid, role) |
|
||||
| `extract_structure_elements_mem(bytes, pages)` | Structure-tree elements from bytes |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
@@ -208,8 +176,7 @@ Low-level detection functions are also available via the `detector` module (`det
|
||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, page number, and optional structure-tree `mcid` |
|
||||
| `StructureElement` | Tagged-PDF structure reference: page (1-indexed), mcid, role (`"H1"`..`"H6"`, `"P"`, …) |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
|
||||
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
|
||||
|
||||
Generated
+4
-26
@@ -499,13 +499,11 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"js-sys",
|
||||
"libc",
|
||||
"r-efi",
|
||||
"rand_core",
|
||||
"wasip2",
|
||||
"wasip3",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -559,25 +557,6 @@ version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
|
||||
|
||||
[[package]]
|
||||
name = "include_dir"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||
dependencies = [
|
||||
"include_dir_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir_macros"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.0"
|
||||
@@ -693,9 +672,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.42.0"
|
||||
version = "0.41.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -851,10 +830,9 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "1.14.0"
|
||||
version = "0.1.6"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
"log",
|
||||
"lopdf",
|
||||
"once_cell",
|
||||
@@ -867,7 +845,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "1.14.0"
|
||||
version = "0.2.2"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "1.14.0"
|
||||
version = "0.2.2"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
+9
-30
@@ -14,17 +14,15 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
@@ -34,7 +32,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64**, **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -83,22 +81,6 @@ for (const region of result[0].regions) {
|
||||
}
|
||||
```
|
||||
|
||||
### Async variants
|
||||
|
||||
`processPdf`, `classifyPdf`, and `extractPagesMarkdown` are synchronous and parse on the calling thread — in Node, that's the event loop. For a one-off call in a script that's fine, but in a server a large document can hold the loop for tens to hundreds of milliseconds.
|
||||
|
||||
`processPdfAsync`, `classifyPdfAsync`, and `extractPagesMarkdownAsync` take the same arguments and produce the same results, but run the parse on the libuv thread pool and return a promise, keeping the event loop free. The input buffer is copied before the call returns, so it's safe to reuse or mutate immediately:
|
||||
|
||||
```typescript
|
||||
import { classifyPdfAsync, extractPagesMarkdownAsync } from '@firecrawl/pdf-inspector'
|
||||
|
||||
const classification = await classifyPdfAsync(pdf)
|
||||
if (classification.pdfType === 'TextBased') {
|
||||
const { pages } = await extractPagesMarkdownAsync(pdf)
|
||||
// ...
|
||||
}
|
||||
```
|
||||
|
||||
## Types
|
||||
|
||||
```typescript
|
||||
@@ -132,12 +114,9 @@ Prebuilt binaries ship as platform-specific packages installed automatically via
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| Linux | x64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-x64-musl` |
|
||||
| Linux | ARM64 (glibc) | `@firecrawl/pdf-inspector-linux-arm64-gnu` |
|
||||
| Linux | ARM64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-arm64-musl` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
+3
-6
@@ -8,12 +8,9 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+4
-10
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.14.0",
|
||||
"version": "1.11.1",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -37,9 +37,6 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-unknown-linux-musl",
|
||||
"aarch64-unknown-linux-gnu",
|
||||
"aarch64-unknown-linux-musl",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
]
|
||||
@@ -52,11 +49,8 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.0"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.1"
|
||||
}
|
||||
}
|
||||
|
||||
+41
-236
@@ -89,11 +89,6 @@ pub struct TextItem {
|
||||
pub item_type: ItemType,
|
||||
/// URL for link items, `None` for other types.
|
||||
pub link_url: Option<String>,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator, `None`
|
||||
/// when the text is not part of marked content. Join with the
|
||||
/// `page`/`mcid` pairs from [`extractStructureElements`] to attach
|
||||
/// structure-tree roles (headings, paragraphs, …) in tagged PDFs.
|
||||
pub mcid: Option<i64>,
|
||||
}
|
||||
|
||||
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
||||
@@ -158,7 +153,9 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<PageOcrReasons> {
|
||||
fn to_napi_page_ocr_reasons(
|
||||
reasons: Vec<pdf_inspector::PageOcrReasons>,
|
||||
) -> Vec<PageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PageOcrReasons {
|
||||
@@ -205,31 +202,6 @@ where
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Shared implementations (single body behind sync and async entry points)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||
let mut opts = pdf_inspector::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = pdf_inspector::process_pdf_mem_with_options(bytes, opts)
|
||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
}
|
||||
|
||||
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: convert_pdf_type(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public NAPI API
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -238,7 +210,15 @@ fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||
#[napi]
|
||||
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("process_pdf", move || process_pdf_impl(&bytes, pages))
|
||||
catch_panic("process_pdf", move || {
|
||||
let mut opts = pdf_inspector::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = pdf_inspector::process_pdf_mem_with_options(&bytes, opts)
|
||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
})
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
@@ -258,7 +238,16 @@ pub fn detect_pdf(buffer: Buffer) -> Result<PdfResult> {
|
||||
#[napi]
|
||||
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("classify_pdf", move || classify_pdf_impl(&bytes))
|
||||
catch_panic("classify_pdf", move || {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: convert_pdf_type(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from a PDF Buffer.
|
||||
@@ -311,61 +300,12 @@ pub fn extract_text_with_positions(
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type,
|
||||
link_url,
|
||||
mcid: item.mcid,
|
||||
}
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// One structure-tree element reference from a tagged PDF.
|
||||
#[napi(object)]
|
||||
pub struct StructureElementJs {
|
||||
/// 1-indexed page number (matches `TextItem.page`).
|
||||
pub page: u32,
|
||||
/// Marked Content ID from the page's content stream (matches
|
||||
/// `TextItem.mcid`).
|
||||
pub mcid: i64,
|
||||
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
|
||||
/// Custom tags are resolved through the document's role map; tags with
|
||||
/// no standard mapping are returned verbatim.
|
||||
pub role: String,
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from a tagged PDF.
|
||||
///
|
||||
/// Parses the document's structure tree (when present) and returns one
|
||||
/// entry per marked-content reference, resolved to its 1-indexed page,
|
||||
/// MCID, and structure type name. Returns an empty array when the PDF is
|
||||
/// not tagged.
|
||||
///
|
||||
/// Join `(page, mcid)` against the `page`/`mcid` fields from
|
||||
/// [`extractTextWithPositions`] to attach heading levels (H1..H6) and other
|
||||
/// semantic roles to extracted text.
|
||||
///
|
||||
/// Pass 1-indexed page numbers (matching `TextItem.page`) to restrict
|
||||
/// output; omit `pages` for the whole document. Entries are sorted by
|
||||
/// `(page, mcid)`.
|
||||
#[napi]
|
||||
pub fn extract_structure_elements(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> Result<Vec<StructureElementJs>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_structure_elements", move || {
|
||||
let elements = pdf_inspector::extract_structure_elements_mem(&bytes, pages.as_deref())
|
||||
.map_err(|e| to_napi_err(e, "extract_structure_elements"))?;
|
||||
Ok(elements
|
||||
.into_iter()
|
||||
.map(|e| StructureElementJs {
|
||||
page: e.page,
|
||||
mcid: e.mcid,
|
||||
role: e.role,
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF.
|
||||
///
|
||||
/// For hybrid OCR: layout model detects regions in rendered images,
|
||||
@@ -693,32 +633,25 @@ pub fn extract_pages_markdown(
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_pages_markdown", move || {
|
||||
extract_pages_markdown_impl(&bytes, pages.as_deref())
|
||||
})
|
||||
}
|
||||
|
||||
fn extract_pages_markdown_impl(
|
||||
bytes: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(bytes, pages)
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|r| PageMarkdownResult {
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|r| PageMarkdownResult {
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
@@ -759,131 +692,3 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Async variants (libuv thread pool via AsyncTask)
|
||||
//
|
||||
// The synchronous exports above parse on the calling thread, which in Node is
|
||||
// the event loop. These `*Async` variants run the same shared implementations
|
||||
// on the libuv thread pool and hand JavaScript a promise, so servers under
|
||||
// concurrent load keep answering requests while a document parses. The sync
|
||||
// exports keep their names, signatures, and behaviour.
|
||||
//
|
||||
// Each factory copies the input Buffer to an owned `Vec<u8>` on the calling
|
||||
// (JS) thread — deliberately. JS execution is single-threaded, so no JS code
|
||||
// can mutate the buffer while the synchronous part of the call copies it.
|
||||
// Holding the napi `Buffer` and reading it from the worker instead would be
|
||||
// zero-copy, but a caller mutating the buffer before the promise settles
|
||||
// would then race the worker's reads — undefined behavior, not a recoverable
|
||||
// error (a known napi-rs soundness hazard with cross-thread Buffer access).
|
||||
// The copy is a one-time memcpy, negligible next to the parse it unblocks.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct ProcessPdfTask {
|
||||
bytes: Vec<u8>,
|
||||
pages: Option<Vec<u32>>,
|
||||
}
|
||||
|
||||
impl Task for ProcessPdfTask {
|
||||
type Output = PdfResult;
|
||||
type JsValue = PdfResult;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
let pages = self.pages.take();
|
||||
// AssertUnwindSafe: `bytes`/`pages` are moved into the closure and
|
||||
// dropped on unwind — no shared state can be observed broken.
|
||||
catch_panic(
|
||||
"process_pdf",
|
||||
panic::AssertUnwindSafe(move || process_pdf_impl(&bytes, pages)),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Async variant of [`processPdf`]: same result, but the parse runs on the
|
||||
/// libuv thread pool instead of the event loop and the call returns a
|
||||
/// promise. The buffer is copied before the call returns, so it may be
|
||||
/// reused or mutated immediately.
|
||||
// ts_return_type is required: napi-rs emits `Promise<unknown>` for
|
||||
// `AsyncTask<T>` returns without it.
|
||||
#[napi(ts_return_type = "Promise<PdfResult>")]
|
||||
pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<ProcessPdfTask> {
|
||||
AsyncTask::new(ProcessPdfTask {
|
||||
bytes: buffer.to_vec(),
|
||||
pages,
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ClassifyPdfTask {
|
||||
bytes: Vec<u8>,
|
||||
}
|
||||
|
||||
impl Task for ClassifyPdfTask {
|
||||
type Output = PdfClassification;
|
||||
type JsValue = PdfClassification;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
catch_panic(
|
||||
"classify_pdf",
|
||||
panic::AssertUnwindSafe(move || classify_pdf_impl(&bytes)),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Async variant of [`classifyPdf`]: same result, but the classification runs
|
||||
/// on the libuv thread pool instead of the event loop and the call returns a
|
||||
/// promise. The buffer is copied before the call returns, so it may be
|
||||
/// reused or mutated immediately.
|
||||
#[napi(ts_return_type = "Promise<PdfClassification>")]
|
||||
pub fn classify_pdf_async(buffer: Buffer) -> AsyncTask<ClassifyPdfTask> {
|
||||
AsyncTask::new(ClassifyPdfTask {
|
||||
bytes: buffer.to_vec(),
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ExtractPagesMarkdownTask {
|
||||
bytes: Vec<u8>,
|
||||
pages: Option<Vec<u32>>,
|
||||
}
|
||||
|
||||
impl Task for ExtractPagesMarkdownTask {
|
||||
type Output = PagesExtractionResult;
|
||||
type JsValue = PagesExtractionResult;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
let pages = self.pages.take();
|
||||
catch_panic(
|
||||
"extract_pages_markdown",
|
||||
panic::AssertUnwindSafe(move || extract_pages_markdown_impl(&bytes, pages.as_deref())),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Async variant of [`extractPagesMarkdown`]: same result, but the extraction
|
||||
/// runs on the libuv thread pool instead of the event loop and the call
|
||||
/// returns a promise. The buffer is copied before the call returns, so it
|
||||
/// may be reused or mutated immediately.
|
||||
#[napi(ts_return_type = "Promise<PagesExtractionResult>")]
|
||||
pub fn extract_pages_markdown_async(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> AsyncTask<ExtractPagesMarkdownTask> {
|
||||
AsyncTask::new(ExtractPagesMarkdownTask {
|
||||
bytes: buffer.to_vec(),
|
||||
pages,
|
||||
})
|
||||
}
|
||||
|
||||
-110
@@ -2,21 +2,16 @@ import { readFileSync } from 'fs';
|
||||
import { strict as assert } from 'assert';
|
||||
import {
|
||||
processPdf,
|
||||
processPdfAsync,
|
||||
detectPdf,
|
||||
classifyPdf,
|
||||
classifyPdfAsync,
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractStructureElements,
|
||||
extractTextInRegions,
|
||||
detectVectorGridInRegion,
|
||||
extractPagesMarkdown,
|
||||
extractPagesMarkdownAsync,
|
||||
} from './index.js';
|
||||
|
||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||
const taggedFixture = readFileSync('../tests/fixtures/firecrawl_docs_tagged.pdf');
|
||||
|
||||
// --- processPdf ---
|
||||
console.log('Testing processPdf...');
|
||||
@@ -84,46 +79,6 @@ assert.ok(page1Items.length > 0);
|
||||
assert.ok(page1Items.every(i => i.page === 1));
|
||||
console.log(' extractTextWithPositions with pages: OK');
|
||||
|
||||
// mcid: undefined on untagged PDFs, numeric on tagged marked content
|
||||
assert.ok(items.every(i => i.mcid === undefined || typeof i.mcid === 'number'));
|
||||
const taggedItems = extractTextWithPositions(taggedFixture);
|
||||
assert.ok(
|
||||
taggedItems.some(i => typeof i.mcid === 'number'),
|
||||
'tagged PDF text items should carry Marked Content IDs',
|
||||
);
|
||||
console.log(' extractTextWithPositions mcid: OK');
|
||||
|
||||
// --- extractStructureElements ---
|
||||
console.log('Testing extractStructureElements...');
|
||||
const structureElements = extractStructureElements(taggedFixture);
|
||||
assert.ok(structureElements.length > 0);
|
||||
assert.ok(structureElements.every(e => typeof e.page === 'number'));
|
||||
assert.ok(structureElements.every(e => typeof e.mcid === 'number'));
|
||||
assert.ok(structureElements.every(e => typeof e.role === 'string' && e.role.length > 0));
|
||||
assert.ok(
|
||||
structureElements.some(e => e.role === 'H1'),
|
||||
'tagged fixture should surface H1 heading roles',
|
||||
);
|
||||
|
||||
// (page, mcid) joins against extractTextWithPositions to recover heading text
|
||||
const h1Refs = new Set(
|
||||
structureElements.filter(e => e.role === 'H1').map(e => `${e.page}:${e.mcid}`),
|
||||
);
|
||||
const h1Text = taggedItems
|
||||
.filter(i => typeof i.mcid === 'number' && h1Refs.has(`${i.page}:${i.mcid}`))
|
||||
.map(i => i.text)
|
||||
.join('');
|
||||
assert.ok(h1Text.trim().length > 0, 'H1 join should recover heading text');
|
||||
|
||||
// pages filter is 1-indexed, matching TextItem.page
|
||||
const page1Elements = extractStructureElements(taggedFixture, [1]);
|
||||
assert.ok(page1Elements.length > 0);
|
||||
assert.ok(page1Elements.every(e => e.page === 1));
|
||||
|
||||
// untagged PDFs yield an empty array
|
||||
assert.deepEqual(extractStructureElements(fixture), []);
|
||||
console.log(' extractStructureElements: OK');
|
||||
|
||||
// --- extractTextInRegions ---
|
||||
console.log('Testing extractTextInRegions...');
|
||||
const regionResults = extractTextInRegions(fixture, [
|
||||
@@ -169,75 +124,10 @@ assert.equal(picked.pages[0].page, 2);
|
||||
assert.equal(picked.pages[1].page, 0);
|
||||
console.log(' extractPagesMarkdown with pages: OK');
|
||||
|
||||
// --- Async variants ---
|
||||
console.log('Testing async variants...');
|
||||
|
||||
// processPdfAsync returns a promise and matches the sync result
|
||||
const asyncResultPromise = processPdfAsync(fixture);
|
||||
assert.ok(asyncResultPromise instanceof Promise);
|
||||
const asyncResult = await asyncResultPromise;
|
||||
assert.equal(asyncResult.pdfType, result.pdfType);
|
||||
assert.equal(asyncResult.pageCount, result.pageCount);
|
||||
assert.equal(asyncResult.markdown, result.markdown);
|
||||
console.log(' processPdfAsync: OK');
|
||||
|
||||
// processPdfAsync with pages
|
||||
const asyncResult2 = await processPdfAsync(fixture, [1]);
|
||||
assert.equal(asyncResult2.markdown, result2.markdown);
|
||||
console.log(' processPdfAsync with pages: OK');
|
||||
|
||||
// classifyPdfAsync matches the sync result
|
||||
const asyncClassified = await classifyPdfAsync(fixture);
|
||||
assert.equal(asyncClassified.pdfType, classified.pdfType);
|
||||
assert.equal(asyncClassified.pageCount, classified.pageCount);
|
||||
assert.equal(asyncClassified.confidence, classified.confidence);
|
||||
assert.deepEqual(asyncClassified.pagesNeedingOcr, classified.pagesNeedingOcr);
|
||||
console.log(' classifyPdfAsync: OK');
|
||||
|
||||
// extractPagesMarkdownAsync matches the sync result
|
||||
const asyncAllPages = await extractPagesMarkdownAsync(fixture);
|
||||
assert.equal(asyncAllPages.pages.length, allPages.pages.length);
|
||||
assert.deepEqual(
|
||||
asyncAllPages.pages.map(p => p.markdown),
|
||||
allPages.pages.map(p => p.markdown),
|
||||
);
|
||||
assert.equal(asyncAllPages.isComplex, allPages.isComplex);
|
||||
console.log(' extractPagesMarkdownAsync: OK');
|
||||
|
||||
// selected pages preserve caller order
|
||||
const asyncPicked = await extractPagesMarkdownAsync(fixture, [2, 0]);
|
||||
assert.equal(asyncPicked.pages.length, 2);
|
||||
assert.equal(asyncPicked.pages[0].page, 2);
|
||||
assert.equal(asyncPicked.pages[1].page, 0);
|
||||
console.log(' extractPagesMarkdownAsync with pages: OK');
|
||||
|
||||
// input buffer is copied at call time: mutating it immediately after the
|
||||
// call must not affect the in-flight parse
|
||||
const scratch = Buffer.from(fixture);
|
||||
const inFlight = processPdfAsync(scratch);
|
||||
scratch.fill(0);
|
||||
const fromMutated = await inFlight;
|
||||
assert.equal(fromMutated.markdown, result.markdown);
|
||||
console.log(' processPdfAsync input copied at call time: OK');
|
||||
|
||||
// concurrent async calls all settle
|
||||
const [c1, c2, c3] = await Promise.all([
|
||||
processPdfAsync(fixture),
|
||||
classifyPdfAsync(fixture),
|
||||
extractPagesMarkdownAsync(fixture),
|
||||
]);
|
||||
assert.equal(c1.pdfType, 'TextBased');
|
||||
assert.equal(c2.pdfType, 'TextBased');
|
||||
assert.equal(c3.pages.length, 3);
|
||||
console.log(' concurrent async calls: OK');
|
||||
|
||||
// --- Error handling ---
|
||||
console.log('Testing error handling...');
|
||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
|
||||
await assert.rejects(processPdfAsync(Buffer.from('not a pdf')), /process_pdf/);
|
||||
await assert.rejects(classifyPdfAsync(Buffer.from('')), /classify_pdf/);
|
||||
await assert.rejects(extractPagesMarkdownAsync(Buffer.from('')), /extract_pages_markdown/);
|
||||
console.log(' error handling: OK');
|
||||
|
||||
console.log('\nAll NAPI tests passed!');
|
||||
|
||||
@@ -10,9 +10,6 @@ class PdfResult:
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed page numbers that need OCR."""
|
||||
ocr_reasons_by_page: list["PageOcrReasons"]
|
||||
"""Machine-readable OCR reasons by 1-indexed page."""
|
||||
title: Optional[str]
|
||||
confidence: float
|
||||
is_complex_layout: bool
|
||||
@@ -20,13 +17,6 @@ class PdfResult:
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool
|
||||
|
||||
class PageOcrReasons:
|
||||
"""OCR reasons for a single 1-indexed page."""
|
||||
page: int
|
||||
"""1-indexed page number."""
|
||||
reasons: list[str]
|
||||
"""Machine-readable OCR reason identifiers."""
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
@@ -51,28 +41,12 @@ class TextItem:
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
mcid: Optional[int]
|
||||
"""Marked Content ID from the content stream's BDC/BMC operator, None when
|
||||
the text is not part of marked content. Join with the (page, mcid) pairs
|
||||
from extract_structure_elements to attach structure-tree roles in tagged
|
||||
PDFs."""
|
||||
|
||||
class StructureElement:
|
||||
"""One structure-tree element reference from a tagged PDF."""
|
||||
page: int
|
||||
"""1-indexed page number (matches TextItem.page)."""
|
||||
mcid: int
|
||||
"""Marked Content ID from the page's content stream (matches TextItem.mcid)."""
|
||||
role: str
|
||||
"""Standard structure type name ("H1".."H6", "P", "Table", "TD", ...)."""
|
||||
|
||||
class RegionText:
|
||||
"""Extracted text for a single region."""
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
"""True when the text should not be trusted."""
|
||||
ocr_reason: Optional[str]
|
||||
"""Machine-readable OCR reason when the cause is known."""
|
||||
|
||||
class PageRegionTexts:
|
||||
"""Extracted text for one page's regions."""
|
||||
@@ -88,8 +62,6 @@ class PageMarkdown:
|
||||
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
|
||||
needs_ocr: bool
|
||||
"""True when text on this page is unreliable and OCR should be used instead."""
|
||||
ocr_reason: Optional[str]
|
||||
"""Machine-readable OCR reason when the cause is known."""
|
||||
|
||||
class PagesExtractionResult:
|
||||
"""Per-page markdown output with document-wide layout classification."""
|
||||
@@ -101,8 +73,6 @@ class PagesExtractionResult:
|
||||
"""1-indexed pages where multi-column layout was detected."""
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed pages that need OCR."""
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
"""Machine-readable OCR reasons by 1-indexed page."""
|
||||
is_complex: bool
|
||||
"""True if any page has tables or multi-column layout."""
|
||||
|
||||
@@ -146,27 +116,6 @@ def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] =
|
||||
"""Extract text with position information from bytes."""
|
||||
...
|
||||
|
||||
def extract_structure_elements(path: str, pages: Optional[list[int]] = None) -> list[StructureElement]:
|
||||
"""Extract structure-tree element references from a tagged PDF file.
|
||||
|
||||
Returns one entry per marked-content reference, resolved to its 1-indexed
|
||||
page, MCID, and structure type name ("H1".."H6", "P", "Table", ...), sorted
|
||||
by (page, mcid). Returns an empty list when the PDF is not tagged.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
pages: Optional list of 1-indexed pages (matching ``TextItem.page``).
|
||||
When ``None`` (default), the whole document is returned.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_structure_elements_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[StructureElement]:
|
||||
"""Extract structure-tree element references from tagged PDF bytes.
|
||||
|
||||
See :func:`extract_structure_elements` for details.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_text_in_regions(
|
||||
path: str,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
|
||||
+3
-3
@@ -4,9 +4,9 @@ build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
# Keep package versions in sync with `python3 scripts/version.py <version>`.
|
||||
# CI publishes automatically when the synchronized change lands on main.
|
||||
version = "1.14.0"
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.5"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
@@ -1,351 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run a paired pdf-inspector OpenDataLoader benchmark and report deltas."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SCORE_KEYS = (
|
||||
"overall_mean",
|
||||
"nid_mean",
|
||||
"nid_s_mean",
|
||||
"teds_mean",
|
||||
"teds_s_mean",
|
||||
"mhs_mean",
|
||||
"mhs_s_mean",
|
||||
)
|
||||
|
||||
|
||||
def _non_negative_int(value: str) -> int:
|
||||
parsed = int(value)
|
||||
if parsed < 0:
|
||||
raise argparse.ArgumentTypeError("must be non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _non_negative_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed) or parsed < 0.0:
|
||||
raise argparse.ArgumentTypeError("must be finite and non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _finite_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed):
|
||||
raise argparse.ArgumentTypeError("must be finite")
|
||||
return parsed
|
||||
|
||||
|
||||
def _scores(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
score = evaluation.get("metrics", {}).get("score", {})
|
||||
return {key: float(score[key]) for key in SCORE_KEYS if score.get(key) is not None}
|
||||
|
||||
|
||||
def _documents(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
documents: dict[str, float] = {}
|
||||
for document in evaluation.get("documents", []):
|
||||
overall = document.get("scores", {}).get("overall")
|
||||
if overall is not None:
|
||||
documents[str(document["document_id"])] = float(overall)
|
||||
return documents
|
||||
|
||||
|
||||
def compare_evaluations(
|
||||
baseline: dict[str, Any],
|
||||
candidate: dict[str, Any],
|
||||
reference: dict[str, Any] | None = None,
|
||||
*,
|
||||
top: int = 10,
|
||||
) -> dict[str, Any]:
|
||||
"""Build aggregate and per-document deltas from evaluator JSON payloads."""
|
||||
baseline_scores = _scores(baseline)
|
||||
candidate_scores = _scores(candidate)
|
||||
metric_deltas = {
|
||||
key: candidate_scores[key] - baseline_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in baseline_scores and key in candidate_scores
|
||||
}
|
||||
|
||||
baseline_documents = _documents(baseline)
|
||||
candidate_documents = _documents(candidate)
|
||||
shared = sorted(baseline_documents.keys() & candidate_documents.keys())
|
||||
document_deltas = [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"baseline": baseline_documents[document_id],
|
||||
"candidate": candidate_documents[document_id],
|
||||
"delta": candidate_documents[document_id] - baseline_documents[document_id],
|
||||
}
|
||||
for document_id in shared
|
||||
]
|
||||
epsilon = 1e-12
|
||||
improvements = sorted(document_deltas, key=lambda item: item["delta"], reverse=True)
|
||||
regressions = sorted(document_deltas, key=lambda item: item["delta"])
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"baseline": baseline_scores,
|
||||
"candidate": candidate_scores,
|
||||
"deltas": metric_deltas,
|
||||
"missing_predictions": {
|
||||
"baseline": int(baseline.get("metrics", {}).get("missing_predictions", 0)),
|
||||
"candidate": int(candidate.get("metrics", {}).get("missing_predictions", 0)),
|
||||
},
|
||||
"documents": {
|
||||
"shared": len(shared),
|
||||
"improved": sum(item["delta"] > epsilon for item in document_deltas),
|
||||
"regressed": sum(item["delta"] < -epsilon for item in document_deltas),
|
||||
"unchanged": sum(abs(item["delta"]) <= epsilon for item in document_deltas),
|
||||
"largest_improvements": [
|
||||
item for item in improvements if item["delta"] > epsilon
|
||||
][:top],
|
||||
"largest_regressions": [
|
||||
item for item in regressions if item["delta"] < -epsilon
|
||||
][:top],
|
||||
"worst_regression": next(
|
||||
(item for item in regressions if item["delta"] < -epsilon), None
|
||||
),
|
||||
},
|
||||
}
|
||||
if reference is not None:
|
||||
reference_scores = _scores(reference)
|
||||
result["reference"] = reference_scores
|
||||
result["candidate_vs_reference"] = {
|
||||
key: candidate_scores[key] - reference_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in candidate_scores and key in reference_scores
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def evaluate_gates(
|
||||
comparison: dict[str, Any],
|
||||
*,
|
||||
min_overall_delta: float,
|
||||
max_document_regression: float | None,
|
||||
max_missing: int,
|
||||
require_reference_lead: bool,
|
||||
) -> list[str]:
|
||||
"""Return human-readable gate failures; an empty list means pass."""
|
||||
failures: list[str] = []
|
||||
overall_delta = comparison["deltas"].get("overall_mean")
|
||||
if overall_delta is None or overall_delta < min_overall_delta:
|
||||
failures.append(
|
||||
f"overall delta {overall_delta!r} is below {min_overall_delta:+.6f}"
|
||||
)
|
||||
candidate_missing = comparison["missing_predictions"]["candidate"]
|
||||
if candidate_missing > max_missing:
|
||||
failures.append(
|
||||
f"candidate has {candidate_missing} missing predictions (maximum {max_missing})"
|
||||
)
|
||||
if max_document_regression is not None:
|
||||
regression = comparison["documents"].get("worst_regression")
|
||||
if regression is not None and regression["delta"] < -max_document_regression:
|
||||
failures.append(
|
||||
"largest document regression "
|
||||
f"{regression['document_id']}={regression['delta']:+.6f} "
|
||||
f"exceeds {-max_document_regression:+.6f}"
|
||||
)
|
||||
if require_reference_lead:
|
||||
reference_delta = comparison.get("candidate_vs_reference", {}).get("overall_mean")
|
||||
if reference_delta is None:
|
||||
failures.append("reference overall score is unavailable")
|
||||
elif reference_delta < 0.0:
|
||||
failures.append(
|
||||
f"candidate trails reference overall by {reference_delta!r}"
|
||||
)
|
||||
return failures
|
||||
|
||||
|
||||
def _run(command: list[str], *, cwd: Path, env: dict[str, str] | None = None) -> None:
|
||||
print("+", " ".join(command), flush=True)
|
||||
subprocess.run(command, cwd=cwd, env=env, check=True)
|
||||
|
||||
|
||||
def _run_engine(
|
||||
*,
|
||||
bench_dir: Path,
|
||||
python: Path,
|
||||
binary: Path,
|
||||
label: str,
|
||||
scratch_root: Path,
|
||||
) -> dict[str, Any]:
|
||||
env = os.environ.copy()
|
||||
env["PDF_INSPECTOR_BINARY"] = str(binary)
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
if source.exists():
|
||||
if source.is_dir():
|
||||
shutil.rmtree(source)
|
||||
else:
|
||||
source.unlink()
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/pdf_parser.py",
|
||||
"--engine",
|
||||
"pdf-inspector",
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
env=env,
|
||||
)
|
||||
|
||||
if not source.is_dir():
|
||||
raise RuntimeError(f"parser did not produce predictions: {source}")
|
||||
destination = scratch_root / label
|
||||
shutil.copytree(source, destination)
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/evaluator.py",
|
||||
"--prediction-root",
|
||||
str(scratch_root),
|
||||
"--engine",
|
||||
label,
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
)
|
||||
with (destination / "evaluation.json").open(encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def _print_report(comparison: dict[str, Any]) -> None:
|
||||
print("\nMetric baseline candidate delta")
|
||||
print("-------------------- ---------- ---------- ----------")
|
||||
for key in SCORE_KEYS:
|
||||
if key not in comparison["deltas"]:
|
||||
continue
|
||||
print(
|
||||
f"{key:<20} {comparison['baseline'][key]:>10.6f} "
|
||||
f"{comparison['candidate'][key]:>10.6f} "
|
||||
f"{comparison['deltas'][key]:>+10.6f}"
|
||||
)
|
||||
if "reference" in comparison:
|
||||
delta = comparison["candidate_vs_reference"].get("overall_mean")
|
||||
reference = comparison["reference"].get("overall_mean")
|
||||
reference_display = f"{reference:.6f}" if reference is not None else "n/a"
|
||||
delta_display = f"{delta:+.6f}" if delta is not None else "n/a"
|
||||
print(f"\nReference overall: {reference_display}; candidate delta: {delta_display}")
|
||||
|
||||
documents = comparison["documents"]
|
||||
print(
|
||||
"\nDocuments: "
|
||||
f"{documents['improved']} improved, {documents['regressed']} regressed, "
|
||||
f"{documents['unchanged']} unchanged ({documents['shared']} shared)"
|
||||
)
|
||||
for heading, key in (
|
||||
("Largest improvements", "largest_improvements"),
|
||||
("Largest regressions", "largest_regressions"),
|
||||
):
|
||||
print(f"\n{heading}:")
|
||||
rows = documents[key]
|
||||
if not rows:
|
||||
print(" none")
|
||||
for row in rows:
|
||||
print(
|
||||
f" {row['document_id']}: {row['delta']:+.6f} "
|
||||
f"({row['baseline']:.6f} -> {row['candidate']:.6f})"
|
||||
)
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--bench-dir", type=Path, required=True)
|
||||
parser.add_argument("--baseline", type=Path, required=True)
|
||||
parser.add_argument("--candidate", type=Path, required=True)
|
||||
parser.add_argument("--python", type=Path)
|
||||
parser.add_argument("--reference-evaluation", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--top", type=_non_negative_int, default=10)
|
||||
parser.add_argument("--min-overall-delta", type=_finite_float, default=0.0)
|
||||
parser.add_argument("--max-document-regression", type=_non_negative_float)
|
||||
parser.add_argument("--max-missing", type=_non_negative_int, default=0)
|
||||
parser.add_argument("--require-reference-lead", action="store_true")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
bench_dir = args.bench_dir.resolve()
|
||||
baseline = args.baseline.resolve()
|
||||
candidate = args.candidate.resolve()
|
||||
# Keep the virtualenv launcher path intact. Resolving its symlink would
|
||||
# invoke the underlying system interpreter without the benchmark's site
|
||||
# packages.
|
||||
python = (args.python or bench_dir / ".venv" / "bin" / "python").absolute()
|
||||
for path, description in (
|
||||
(bench_dir / "src" / "pdf_parser.py", "OpenDataLoader parser"),
|
||||
(bench_dir / "src" / "evaluator.py", "OpenDataLoader evaluator"),
|
||||
(baseline, "baseline binary"),
|
||||
(candidate, "candidate binary"),
|
||||
(python, "Python interpreter"),
|
||||
):
|
||||
if not path.exists():
|
||||
raise SystemExit(f"{description} not found: {path}")
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="pdf-inspector-opendataloader-") as temporary:
|
||||
scratch_root = Path(temporary)
|
||||
baseline_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=baseline,
|
||||
label="baseline",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
candidate_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=candidate,
|
||||
label="candidate",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
|
||||
reference = None
|
||||
if args.reference_evaluation is not None:
|
||||
with args.reference_evaluation.resolve().open(encoding="utf-8") as handle:
|
||||
reference = json.load(handle)
|
||||
|
||||
comparison = compare_evaluations(
|
||||
baseline_evaluation,
|
||||
candidate_evaluation,
|
||||
reference,
|
||||
top=args.top,
|
||||
)
|
||||
|
||||
_print_report(comparison)
|
||||
if args.json_output is not None:
|
||||
args.json_output.resolve().write_text(
|
||||
json.dumps(comparison, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=args.min_overall_delta,
|
||||
max_document_regression=args.max_document_regression,
|
||||
max_missing=args.max_missing,
|
||||
require_reference_lead=args.require_reference_lead,
|
||||
)
|
||||
if failures:
|
||||
print("\nBenchmark gate failed:", file=sys.stderr)
|
||||
for failure in failures:
|
||||
print(f" - {failure}", file=sys.stderr)
|
||||
return 1
|
||||
print("\nBenchmark gate passed.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,351 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare pdf-inspector evidence with optional MuPDF structured text.
|
||||
|
||||
This is an experiment and diagnostic tool, not an extraction fallback. It runs
|
||||
MuPDF's deterministic ``stext.json`` backend without OCR and highlights pages
|
||||
where that backend exposes materially different text or layout evidence.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from collections import Counter
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any, Iterable
|
||||
|
||||
|
||||
TOKEN_PATTERN = re.compile(r"[^\W_]+(?:[\u2019'][^\W_]+)*", re.UNICODE)
|
||||
|
||||
|
||||
def _tokens(texts: Iterable[str]) -> Counter[str]:
|
||||
tokens: Counter[str] = Counter()
|
||||
for text in texts:
|
||||
for token in TOKEN_PATTERN.findall(text.casefold()):
|
||||
# Lone letters are frequently bullets, chart labels, or fragmented
|
||||
# glyphs. Digits remain useful even when they are one character.
|
||||
if len(token) > 1 or token.isdigit():
|
||||
tokens[token] += 1
|
||||
return tokens
|
||||
|
||||
|
||||
def _repeated_x_anchors(xs: Iterable[float], *, tolerance: float = 4.0) -> int:
|
||||
buckets = Counter(round(float(x) / tolerance) for x in xs)
|
||||
return sum(count >= 3 for count in buckets.values())
|
||||
|
||||
|
||||
def local_pages(payload: dict[str, Any]) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize positioned ``pdf2md --items-json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
for item in payload.get("items", []):
|
||||
page_number = int(item["page"])
|
||||
page = pages.setdefault(
|
||||
page_number,
|
||||
{"texts": [], "xs": [], "text_items": 0, "image_items": 0},
|
||||
)
|
||||
if item.get("item_type") == "image":
|
||||
page["image_items"] += 1
|
||||
continue
|
||||
text = str(item.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
page["xs"].append(float(item.get("x", 0.0)))
|
||||
page["text_items"] += 1
|
||||
return pages
|
||||
|
||||
|
||||
def alternate_pages(
|
||||
payload: dict[str, Any] | list[dict[str, Any]],
|
||||
) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize MuPDF ``stext.json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
raw_pages = payload if isinstance(payload, list) else payload.get("pages", [])
|
||||
for index, raw_page in enumerate(raw_pages, start=1):
|
||||
page_number = int(raw_page.get("number", index))
|
||||
page = {
|
||||
"texts": [],
|
||||
"xs": [],
|
||||
"text_blocks": 0,
|
||||
"text_lines": 0,
|
||||
"image_blocks": 0,
|
||||
}
|
||||
for block in raw_page.get("blocks", []):
|
||||
if block.get("type") == "image":
|
||||
page["image_blocks"] += 1
|
||||
continue
|
||||
if block.get("type") != "text":
|
||||
continue
|
||||
page["text_blocks"] += 1
|
||||
for line in block.get("lines", []):
|
||||
text = str(line.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
bbox = line.get("bbox", {})
|
||||
page["xs"].append(float(bbox.get("x", line.get("x", 0.0))))
|
||||
page["text_lines"] += 1
|
||||
pages[page_number] = page
|
||||
return pages
|
||||
|
||||
|
||||
def compare_page(
|
||||
local: dict[str, Any],
|
||||
alternate: dict[str, Any],
|
||||
*,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
"""Compare semantic and coarse layout evidence for one page."""
|
||||
local_tokens = _tokens(local.get("texts", []))
|
||||
alternate_tokens = _tokens(alternate.get("texts", []))
|
||||
shared = local_tokens & alternate_tokens
|
||||
alternate_only = alternate_tokens - local_tokens
|
||||
local_only = local_tokens - alternate_tokens
|
||||
local_total = sum(local_tokens.values())
|
||||
alternate_total = sum(alternate_tokens.values())
|
||||
shared_total = sum(shared.values())
|
||||
alternate_only_total = sum(alternate_only.values())
|
||||
local_only_total = sum(local_only.values())
|
||||
net_token_gain = alternate_total - local_total
|
||||
alternate_only_ratio = alternate_only_total / max(alternate_total, 1)
|
||||
|
||||
local_anchors = _repeated_x_anchors(local.get("xs", []))
|
||||
alternate_anchors = _repeated_x_anchors(alternate.get("xs", []))
|
||||
anchor_gain = alternate_anchors - local_anchors
|
||||
image_gain = int(alternate.get("image_blocks", 0)) - int(
|
||||
local.get("image_items", 0)
|
||||
)
|
||||
|
||||
reasons: list[str] = []
|
||||
if local_total == 0 and alternate_total >= max(5, min_token_gain // 2):
|
||||
reasons.append("local_text_empty")
|
||||
elif (
|
||||
net_token_gain >= min_token_gain
|
||||
and alternate_only_ratio >= min_alternate_only_ratio
|
||||
):
|
||||
reasons.append("alternate_has_more_text")
|
||||
if anchor_gain >= min_anchor_gain:
|
||||
reasons.append("alternate_has_more_alignment_anchors")
|
||||
if image_gain > 0:
|
||||
reasons.append("alternate_has_more_image_blocks")
|
||||
|
||||
if reasons:
|
||||
classification = "investigate_alternate_evidence"
|
||||
elif local_total - alternate_total >= min_token_gain:
|
||||
classification = "local_has_more_text"
|
||||
elif alternate_only_total + local_only_total:
|
||||
classification = "different_segmentation_or_decoding"
|
||||
else:
|
||||
classification = "equivalent_text_evidence"
|
||||
|
||||
return {
|
||||
"classification": classification,
|
||||
"reasons": reasons,
|
||||
"tokens": {
|
||||
"local": local_total,
|
||||
"alternate": alternate_total,
|
||||
"shared": shared_total,
|
||||
"net_alternate_gain": net_token_gain,
|
||||
"alternate_only": alternate_only_total,
|
||||
"local_only": local_only_total,
|
||||
"alternate_only_ratio": alternate_only_ratio,
|
||||
"alternate_only_sample": sorted(alternate_only)[:12],
|
||||
"local_only_sample": sorted(local_only)[:12],
|
||||
},
|
||||
"layout": {
|
||||
"local_text_items": int(local.get("text_items", 0)),
|
||||
"local_image_items": int(local.get("image_items", 0)),
|
||||
"local_repeated_x_anchors": local_anchors,
|
||||
"alternate_text_blocks": int(alternate.get("text_blocks", 0)),
|
||||
"alternate_text_lines": int(alternate.get("text_lines", 0)),
|
||||
"alternate_image_blocks": int(alternate.get("image_blocks", 0)),
|
||||
"alternate_repeated_x_anchors": alternate_anchors,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def compare_documents(
|
||||
local_payload: dict[str, Any],
|
||||
alternate_payload: dict[str, Any] | list[dict[str, Any]],
|
||||
*,
|
||||
min_token_gain: int = 20,
|
||||
min_alternate_only_ratio: float = 0.15,
|
||||
min_anchor_gain: int = 2,
|
||||
) -> dict[str, Any]:
|
||||
"""Return a page-level evidence report for already extracted payloads."""
|
||||
local = local_pages(local_payload)
|
||||
alternate = alternate_pages(alternate_payload)
|
||||
page_numbers = sorted(local.keys() | alternate.keys())
|
||||
pages = []
|
||||
for page_number in page_numbers:
|
||||
result = compare_page(
|
||||
local.get(page_number, {}),
|
||||
alternate.get(page_number, {}),
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
result["page"] = page_number
|
||||
pages.append(result)
|
||||
|
||||
flagged = [
|
||||
page
|
||||
for page in pages
|
||||
if page["classification"] == "investigate_alternate_evidence"
|
||||
]
|
||||
return {
|
||||
"summary": {
|
||||
"pages": len(pages),
|
||||
"flagged_pages": len(flagged),
|
||||
"flagged_page_numbers": [page["page"] for page in flagged],
|
||||
"local_tokens": sum(page["tokens"]["local"] for page in pages),
|
||||
"alternate_tokens": sum(page["tokens"]["alternate"] for page in pages),
|
||||
"alternate_only_tokens": sum(
|
||||
page["tokens"]["alternate_only"] for page in pages
|
||||
),
|
||||
},
|
||||
"pages": pages,
|
||||
}
|
||||
|
||||
|
||||
def _json_command(command: list[str]) -> Any:
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
check=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as error:
|
||||
detail = error.stderr.strip() or error.stdout.strip() or "no diagnostic output"
|
||||
raise RuntimeError(f"command failed: {' '.join(command)}\n{detail}") from error
|
||||
try:
|
||||
return json.loads(completed.stdout)
|
||||
except json.JSONDecodeError as error:
|
||||
raise RuntimeError(
|
||||
f"command did not return JSON: {' '.join(command)}: {error}"
|
||||
) from error
|
||||
|
||||
|
||||
def probe_pdf(
|
||||
pdf: Path,
|
||||
*,
|
||||
pdf2md: Path,
|
||||
mutool: Path,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
local_payload = _json_command([str(pdf2md), str(pdf), "--items-json"])
|
||||
# `stext.json` is MuPDF's native structured text output. The OCR formats
|
||||
# are intentionally not used so this remains a deterministic no-model
|
||||
# comparison.
|
||||
alternate_payload = _json_command(
|
||||
[str(mutool), "draw", "-q", "-F", "stext.json", "-o", "-", str(pdf)]
|
||||
)
|
||||
report = compare_documents(
|
||||
local_payload,
|
||||
alternate_payload,
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
report["pdf"] = str(pdf)
|
||||
return report
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("pdf", type=Path, nargs="+")
|
||||
parser.add_argument("--pdf2md", type=Path, default=Path("target/release/pdf2md"))
|
||||
parser.add_argument("--mutool", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--min-token-gain", type=int, default=20)
|
||||
parser.add_argument("--min-alternate-only-ratio", type=float, default=0.15)
|
||||
parser.add_argument("--min-anchor-gain", type=int, default=2)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def _print_report(result: dict[str, Any]) -> None:
|
||||
summary = result["summary"]
|
||||
print(f"\n{result['pdf']}")
|
||||
print(
|
||||
f" {summary['flagged_pages']}/{summary['pages']} pages flagged; "
|
||||
f"tokens local={summary['local_tokens']} alternate={summary['alternate_tokens']} "
|
||||
f"alternate-only={summary['alternate_only_tokens']}"
|
||||
)
|
||||
for page in result["pages"]:
|
||||
if page["classification"] != "investigate_alternate_evidence":
|
||||
continue
|
||||
reasons = ", ".join(page["reasons"])
|
||||
tokens = page["tokens"]
|
||||
print(
|
||||
f" page {page['page']}: {reasons}; "
|
||||
f"net tokens={tokens['net_alternate_gain']:+d}, "
|
||||
f"alternate-only={tokens['alternate_only']}"
|
||||
)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
pdf2md = args.pdf2md.absolute()
|
||||
mutool = args.mutool or (Path(found) if (found := shutil.which("mutool")) else None)
|
||||
if not pdf2md.is_file():
|
||||
print(f"error: pdf2md binary not found: {pdf2md}", file=sys.stderr)
|
||||
return 2
|
||||
if mutool is None or not mutool.is_file():
|
||||
print("error: mutool not found; install MuPDF or pass --mutool", file=sys.stderr)
|
||||
return 2
|
||||
if (
|
||||
args.min_token_gain < 0
|
||||
or args.min_anchor_gain < 0
|
||||
or not 0.0 <= args.min_alternate_only_ratio <= 1.0
|
||||
):
|
||||
print("error: thresholds must be non-negative and ratio must be in [0, 1]", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
results = []
|
||||
for pdf in args.pdf:
|
||||
path = pdf.absolute()
|
||||
if not path.is_file():
|
||||
print(f"error: PDF not found: {path}", file=sys.stderr)
|
||||
return 2
|
||||
try:
|
||||
result = probe_pdf(
|
||||
path,
|
||||
pdf2md=pdf2md,
|
||||
mutool=mutool,
|
||||
min_token_gain=args.min_token_gain,
|
||||
min_alternate_only_ratio=args.min_alternate_only_ratio,
|
||||
min_anchor_gain=args.min_anchor_gain,
|
||||
)
|
||||
except RuntimeError as error:
|
||||
print(f"error: {error}", file=sys.stderr)
|
||||
return 1
|
||||
results.append(result)
|
||||
_print_report(result)
|
||||
|
||||
payload = {
|
||||
"schema_version": 1,
|
||||
"experiment": "optional_mupdf_stext_evidence",
|
||||
"ocr": False,
|
||||
"thresholds": {
|
||||
"min_token_gain": args.min_token_gain,
|
||||
"min_alternate_only_ratio": args.min_alternate_only_ratio,
|
||||
"min_anchor_gain": args.min_anchor_gain,
|
||||
},
|
||||
"documents": results,
|
||||
}
|
||||
if args.json_output:
|
||||
args.json_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.json_output.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,203 +0,0 @@
|
||||
import io
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from contextlib import redirect_stderr, redirect_stdout
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from bench_opendataloader import (
|
||||
_arguments,
|
||||
_print_report,
|
||||
_run_engine,
|
||||
compare_evaluations,
|
||||
evaluate_gates,
|
||||
)
|
||||
|
||||
|
||||
def evaluation(overall, documents, *, missing=0):
|
||||
return {
|
||||
"metrics": {
|
||||
"score": {
|
||||
"overall_mean": overall,
|
||||
"nid_mean": overall + 0.01,
|
||||
},
|
||||
"missing_predictions": missing,
|
||||
},
|
||||
"documents": [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"scores": {"overall": score},
|
||||
}
|
||||
for document_id, score in documents.items()
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class ComparisonTests(unittest.TestCase):
|
||||
def test_reports_metric_and_document_deltas(self):
|
||||
baseline = evaluation(0.80, {"a": 0.8, "b": 0.6, "c": 0.7})
|
||||
candidate = evaluation(0.82, {"a": 0.9, "b": 0.5, "c": 0.7})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, top=1)
|
||||
|
||||
self.assertAlmostEqual(result["deltas"]["overall_mean"], 0.02)
|
||||
self.assertEqual(result["documents"]["improved"], 1)
|
||||
self.assertEqual(result["documents"]["regressed"], 1)
|
||||
self.assertEqual(result["documents"]["unchanged"], 1)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_improvements"][0]["document_id"], "a"
|
||||
)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_regressions"][0]["document_id"], "b"
|
||||
)
|
||||
|
||||
def test_reference_delta_is_reported(self):
|
||||
baseline = evaluation(0.80, {})
|
||||
candidate = evaluation(0.82, {})
|
||||
reference = evaluation(0.81, {})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, reference)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
result["candidate_vs_reference"]["overall_mean"], 0.01
|
||||
)
|
||||
|
||||
def test_gates_cover_aggregate_document_missing_and_reference(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.79, {"a": 0.7}, missing=1),
|
||||
evaluation(0.81, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 4)
|
||||
|
||||
def test_regression_gate_is_independent_of_report_limit(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.80, {"a": 0.7}),
|
||||
top=0,
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=False,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 1)
|
||||
self.assertIn("largest document regression", failures[0])
|
||||
|
||||
def test_report_handles_reference_without_overall_score(self):
|
||||
result = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
{"metrics": {"score": {"nid_mean": 0.81}}},
|
||||
)
|
||||
|
||||
output = io.StringIO()
|
||||
with redirect_stdout(output):
|
||||
_print_report(result)
|
||||
|
||||
self.assertIn("Reference overall: n/a; candidate delta: n/a", output.getvalue())
|
||||
|
||||
def test_reference_gate_reports_missing_score_as_unavailable(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=None,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(failures, ["reference overall score is unavailable"])
|
||||
|
||||
def test_arguments_reject_negative_counts_and_allow_zero_top(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
self.assertEqual(_arguments(required + ["--top", "0"]).top, 0)
|
||||
for option in ("--top", "--max-document-regression", "--max-missing"):
|
||||
with self.subTest(option=option), redirect_stderr(io.StringIO()):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, "-1"])
|
||||
|
||||
def test_arguments_reject_nonfinite_float_thresholds(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
for option in ("--min-overall-delta", "--max-document-regression"):
|
||||
for value in ("nan", "inf", "-inf"):
|
||||
with self.subTest(option=option, value=value), redirect_stderr(
|
||||
io.StringIO()
|
||||
):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, value])
|
||||
|
||||
def test_run_engine_clears_stale_predictions_before_parser(self):
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
bench_dir = root / "bench"
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
source.mkdir(parents=True)
|
||||
(source / "stale.md").write_text("stale", encoding="utf-8")
|
||||
scratch = root / "scratch"
|
||||
scratch.mkdir()
|
||||
|
||||
def fake_run(command, *, cwd, env=None):
|
||||
if any(part.endswith("pdf_parser.py") for part in command):
|
||||
self.assertFalse(source.exists())
|
||||
(source / "markdown").mkdir(parents=True)
|
||||
(source / "markdown" / "new.md").write_text(
|
||||
"new", encoding="utf-8"
|
||||
)
|
||||
else:
|
||||
destination = scratch / "candidate"
|
||||
(destination / "evaluation.json").write_text(
|
||||
json.dumps(evaluation(0.82, {})), encoding="utf-8"
|
||||
)
|
||||
|
||||
with patch("bench_opendataloader._run", side_effect=fake_run):
|
||||
result = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=Path("python"),
|
||||
binary=Path("pdf2md"),
|
||||
label="candidate",
|
||||
scratch_root=scratch,
|
||||
)
|
||||
|
||||
self.assertEqual(result["metrics"]["score"]["overall_mean"], 0.82)
|
||||
self.assertFalse((source / "stale.md").exists())
|
||||
self.assertFalse((scratch / "candidate" / "stale.md").exists())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,103 +0,0 @@
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from probe_backend_evidence import compare_documents
|
||||
|
||||
|
||||
def local_payload(items):
|
||||
return {"items": items}
|
||||
|
||||
|
||||
def item(page, text, x=10, item_type="text"):
|
||||
return {"page": page, "text": text, "x": x, "item_type": item_type}
|
||||
|
||||
|
||||
def alternate_payload(pages):
|
||||
return {"pages": pages}
|
||||
|
||||
|
||||
def page(lines, *, images=0):
|
||||
blocks = [
|
||||
{
|
||||
"type": "text",
|
||||
"lines": [
|
||||
{"text": text, "bbox": {"x": x, "y": index * 10, "w": 80, "h": 8}}
|
||||
for index, (text, x) in enumerate(lines)
|
||||
],
|
||||
}
|
||||
]
|
||||
blocks.extend({"type": "image"} for _ in range(images))
|
||||
return {"blocks": blocks}
|
||||
|
||||
|
||||
class EvidenceComparisonTests(unittest.TestCase):
|
||||
def test_accepts_real_top_level_page_array(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = [page([("alpha beta gamma", 10)])]
|
||||
|
||||
result = compare_documents(local, alternate)["pages"][0]
|
||||
|
||||
self.assertEqual(result["tokens"]["alternate"], 3)
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 1)
|
||||
|
||||
def test_flags_material_alternate_text_gain(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = alternate_payload(
|
||||
[page([("alpha beta gamma delta epsilon zeta", 10)])]
|
||||
)
|
||||
|
||||
report = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=3,
|
||||
min_alternate_only_ratio=0.2,
|
||||
)
|
||||
|
||||
result = report["pages"][0]
|
||||
self.assertEqual(result["classification"], "investigate_alternate_evidence")
|
||||
self.assertIn("alternate_has_more_text", result["reasons"])
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 4)
|
||||
|
||||
def test_repeated_alignment_and_image_evidence_are_reported(self):
|
||||
local = local_payload([item(1, "one two", 10)])
|
||||
alternate = alternate_payload(
|
||||
[
|
||||
page(
|
||||
[
|
||||
("one two", 10),
|
||||
("row three", 100),
|
||||
("row four", 100),
|
||||
("row five", 100),
|
||||
],
|
||||
images=1,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
result = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=99,
|
||||
min_anchor_gain=1,
|
||||
)["pages"][0]
|
||||
|
||||
self.assertIn("alternate_has_more_alignment_anchors", result["reasons"])
|
||||
self.assertIn("alternate_has_more_image_blocks", result["reasons"])
|
||||
self.assertEqual(result["layout"]["alternate_repeated_x_anchors"], 1)
|
||||
|
||||
def test_token_segmentation_difference_does_not_imply_more_evidence(self):
|
||||
local = local_payload([item(1, "Revenue 2025")])
|
||||
alternate = alternate_payload([page([("Revenue 2024", 10)])])
|
||||
|
||||
result = compare_documents(local, alternate, min_token_gain=2)["pages"][0]
|
||||
|
||||
self.assertEqual(result["classification"], "different_segmentation_or_decoding")
|
||||
self.assertEqual(result["reasons"], [])
|
||||
self.assertEqual(result["tokens"]["alternate_only_sample"], ["2024"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,111 +0,0 @@
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from version import PLATFORM_PACKAGES, check_versions, set_versions
|
||||
|
||||
|
||||
class VersionTests(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.temporary = tempfile.TemporaryDirectory()
|
||||
self.root = Path(self.temporary.name)
|
||||
(self.root / "napi").mkdir()
|
||||
(self.root / "site").mkdir()
|
||||
(self.root / "wasm").mkdir()
|
||||
|
||||
self._write_manifest("Cargo.toml", "package", "0.1.0")
|
||||
self._write_manifest("pyproject.toml", "project", "0.1.0")
|
||||
self._write_manifest("napi/Cargo.toml", "package", "0.1.0")
|
||||
self._write_manifest("wasm/Cargo.toml", "package", "0.1.0")
|
||||
|
||||
package = {
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "0.1.0",
|
||||
"optionalDependencies": {
|
||||
dependency: "0.1.0" for dependency in PLATFORM_PACKAGES
|
||||
},
|
||||
}
|
||||
(self.root / "napi/package.json").write_text(
|
||||
json.dumps(package), encoding="utf-8"
|
||||
)
|
||||
(self.root / "napi/bun.lock").write_text(
|
||||
"\n".join(
|
||||
f' "{dependency}": "0.1.0",'
|
||||
for dependency in PLATFORM_PACKAGES
|
||||
)
|
||||
+ "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(self.root / "site/index.html").write_text(
|
||||
'https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@0.1.0/'
|
||||
'pdf_inspector_wasm.js\n',
|
||||
encoding="utf-8",
|
||||
)
|
||||
self._write_lock(
|
||||
"napi/Cargo.lock", ("pdf-inspector", "pdf-inspector-napi")
|
||||
)
|
||||
self._write_lock(
|
||||
"wasm/Cargo.lock", ("pdf-inspector", "pdf-inspector-wasm")
|
||||
)
|
||||
|
||||
def tearDown(self):
|
||||
self.temporary.cleanup()
|
||||
|
||||
def _write_manifest(self, relative, section, version):
|
||||
(self.root / relative).write_text(
|
||||
f'[{section}]\nname = "fixture"\nversion = "{version}"\n',
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
def _write_lock(self, relative, packages):
|
||||
content = "\n".join(
|
||||
f'[[package]]\nname = "{package}"\nversion = "0.1.0"\n'
|
||||
for package in packages
|
||||
)
|
||||
(self.root / relative).write_text(content, encoding="utf-8")
|
||||
|
||||
def test_updates_every_version_location(self):
|
||||
set_versions("1.14.0", self.root)
|
||||
|
||||
self.assertEqual(check_versions(self.root), "1.14.0")
|
||||
|
||||
def test_reports_a_divergent_package(self):
|
||||
self._write_manifest("wasm/Cargo.toml", "package", "0.2.0")
|
||||
|
||||
with self.assertRaisesRegex(ValueError, "WASM package: 0.2.0"):
|
||||
check_versions(self.root)
|
||||
|
||||
def test_rejects_an_invalid_version(self):
|
||||
with self.assertRaisesRegex(ValueError, "Invalid semantic version"):
|
||||
set_versions("next", self.root)
|
||||
|
||||
def test_rejects_numeric_prerelease_with_leading_zero(self):
|
||||
before = (self.root / "Cargo.toml").read_text(encoding="utf-8")
|
||||
|
||||
with self.assertRaisesRegex(ValueError, "Invalid semantic version"):
|
||||
set_versions("1.2.3-01", self.root)
|
||||
|
||||
self.assertEqual(
|
||||
(self.root / "Cargo.toml").read_text(encoding="utf-8"), before
|
||||
)
|
||||
|
||||
def test_preflight_failure_does_not_partially_update(self):
|
||||
before = (self.root / "Cargo.toml").read_text(encoding="utf-8")
|
||||
(self.root / "site/index.html").write_text(
|
||||
"missing module URL\n", encoding="utf-8"
|
||||
)
|
||||
|
||||
with self.assertRaisesRegex(ValueError, "Missing pinned WASM package URL"):
|
||||
set_versions("1.14.0", self.root)
|
||||
|
||||
self.assertEqual(
|
||||
(self.root / "Cargo.toml").read_text(encoding="utf-8"), before
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,262 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Keep every pdf-inspector package on one release version."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
PRERELEASE_IDENTIFIER = (
|
||||
r"(?:0|[1-9]\d*|[0-9A-Za-z-]*[A-Za-z-][0-9A-Za-z-]*)"
|
||||
)
|
||||
SEMVER = re.compile(
|
||||
r"^(0|[1-9]\d*)\.(0|[1-9]\d*)\.(0|[1-9]\d*)"
|
||||
rf"(?:-{PRERELEASE_IDENTIFIER}(?:\.{PRERELEASE_IDENTIFIER})*)?"
|
||||
r"(?:\+[0-9A-Za-z-]+(?:\.[0-9A-Za-z-]+)*)?$"
|
||||
)
|
||||
VERSION_LINE = re.compile(r'^(\s*version\s*=\s*")[^"]+(".*)$')
|
||||
SECTION_LINE = re.compile(r"^\s*\[([^]]+)]\s*$")
|
||||
PLATFORM_PACKAGES = (
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc",
|
||||
)
|
||||
TOML_VERSIONS = (
|
||||
("Rust crate", Path("Cargo.toml"), "package"),
|
||||
("Python package", Path("pyproject.toml"), "project"),
|
||||
("NAPI crate", Path("napi/Cargo.toml"), "package"),
|
||||
("WASM package", Path("wasm/Cargo.toml"), "package"),
|
||||
)
|
||||
LOCK_VERSIONS = (
|
||||
("NAPI lock: core", Path("napi/Cargo.lock"), "pdf-inspector"),
|
||||
("NAPI lock: binding", Path("napi/Cargo.lock"), "pdf-inspector-napi"),
|
||||
("WASM lock: core", Path("wasm/Cargo.lock"), "pdf-inspector"),
|
||||
("WASM lock: binding", Path("wasm/Cargo.lock"), "pdf-inspector-wasm"),
|
||||
)
|
||||
SITE_WASM_VERSION = re.compile(
|
||||
r"(@firecrawl/pdf-inspector-wasm@)([^/\"]+)(/pdf_inspector_wasm\.js)"
|
||||
)
|
||||
|
||||
|
||||
def _read_section_version(path: Path, section: str) -> str:
|
||||
active = False
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
section_match = SECTION_LINE.match(line)
|
||||
if section_match:
|
||||
active = section_match.group(1) == section
|
||||
elif active:
|
||||
version_match = VERSION_LINE.match(line)
|
||||
if version_match:
|
||||
return line.split('"', 2)[1]
|
||||
raise ValueError(f"No version found in [{section}] of {path}")
|
||||
|
||||
|
||||
def _write_section_version(path: Path, section: str, version: str) -> None:
|
||||
lines = path.read_text(encoding="utf-8").splitlines(keepends=True)
|
||||
active = False
|
||||
for index, line in enumerate(lines):
|
||||
section_match = SECTION_LINE.match(line)
|
||||
if section_match:
|
||||
active = section_match.group(1) == section
|
||||
elif active:
|
||||
version_match = VERSION_LINE.match(line)
|
||||
if version_match:
|
||||
newline = "\n" if line.endswith("\n") else ""
|
||||
replacement = (
|
||||
f"{version_match.group(1)}{version}"
|
||||
f"{version_match.group(2).rstrip()}"
|
||||
)
|
||||
lines[index] = (
|
||||
f"{replacement}{newline}"
|
||||
)
|
||||
path.write_text("".join(lines), encoding="utf-8")
|
||||
return
|
||||
raise ValueError(f"No version found in [{section}] of {path}")
|
||||
|
||||
|
||||
def _package_block(lines: list[str], package: str) -> tuple[int, int]:
|
||||
for start, line in enumerate(lines):
|
||||
if line.strip() != "[[package]]":
|
||||
continue
|
||||
end = next(
|
||||
(
|
||||
index
|
||||
for index in range(start + 1, len(lines))
|
||||
if lines[index].strip() == "[[package]]"
|
||||
),
|
||||
len(lines),
|
||||
)
|
||||
if any(line.strip() == f'name = "{package}"' for line in lines[start:end]):
|
||||
return start, end
|
||||
raise ValueError(f"No lockfile entry found for {package}")
|
||||
|
||||
|
||||
def _read_lock_version(path: Path, package: str) -> str:
|
||||
lines = path.read_text(encoding="utf-8").splitlines()
|
||||
start, end = _package_block(lines, package)
|
||||
for line in lines[start:end]:
|
||||
version_match = VERSION_LINE.match(line)
|
||||
if version_match:
|
||||
return line.split('"', 2)[1]
|
||||
raise ValueError(f"No version found for {package} in {path}")
|
||||
|
||||
|
||||
def _write_lock_version(path: Path, package: str, version: str) -> None:
|
||||
lines = path.read_text(encoding="utf-8").splitlines(keepends=True)
|
||||
start, end = _package_block(lines, package)
|
||||
for index in range(start, end):
|
||||
version_match = VERSION_LINE.match(lines[index])
|
||||
if version_match:
|
||||
newline = "\n" if lines[index].endswith("\n") else ""
|
||||
lines[index] = (
|
||||
f'{version_match.group(1)}{version}{version_match.group(2).rstrip()}'
|
||||
f"{newline}"
|
||||
)
|
||||
path.write_text("".join(lines), encoding="utf-8")
|
||||
return
|
||||
raise ValueError(f"No version found for {package} in {path}")
|
||||
|
||||
|
||||
def _node_versions(root: Path) -> dict[str, str]:
|
||||
package = json.loads((root / "napi/package.json").read_text(encoding="utf-8"))
|
||||
versions = {"Node package": package["version"]}
|
||||
optional = package.get("optionalDependencies", {})
|
||||
for dependency in PLATFORM_PACKAGES:
|
||||
if dependency not in optional:
|
||||
raise ValueError(f"Missing Node optional dependency: {dependency}")
|
||||
versions[f"Node optional dependency: {dependency}"] = optional[dependency]
|
||||
return versions
|
||||
|
||||
|
||||
def _bun_versions(root: Path) -> dict[str, str]:
|
||||
text = (root / "napi/bun.lock").read_text(encoding="utf-8")
|
||||
versions = {}
|
||||
for dependency in PLATFORM_PACKAGES:
|
||||
match = re.search(
|
||||
rf'"{re.escape(dependency)}": "([^"]+)"[,]', text
|
||||
)
|
||||
if not match:
|
||||
raise ValueError(f"Missing Bun lock dependency: {dependency}")
|
||||
versions[f"Bun lock: {dependency}"] = match.group(1)
|
||||
return versions
|
||||
|
||||
|
||||
def _site_wasm_version(root: Path) -> str:
|
||||
text = (root / "site/index.html").read_text(encoding="utf-8")
|
||||
match = SITE_WASM_VERSION.search(text)
|
||||
if not match:
|
||||
raise ValueError("Missing pinned WASM package URL in site/index.html")
|
||||
return match.group(2)
|
||||
|
||||
|
||||
def package_versions(root: Path = ROOT) -> dict[str, str]:
|
||||
versions = {
|
||||
label: _read_section_version(root / relative, section)
|
||||
for label, relative, section in TOML_VERSIONS
|
||||
}
|
||||
versions.update(_node_versions(root))
|
||||
versions.update(_bun_versions(root))
|
||||
versions["Website WASM module"] = _site_wasm_version(root)
|
||||
versions.update(
|
||||
{
|
||||
label: _read_lock_version(root / relative, package)
|
||||
for label, relative, package in LOCK_VERSIONS
|
||||
}
|
||||
)
|
||||
return versions
|
||||
|
||||
|
||||
def check_versions(root: Path = ROOT) -> str:
|
||||
versions = package_versions(root)
|
||||
expected = versions["Rust crate"]
|
||||
if not SEMVER.fullmatch(expected):
|
||||
raise ValueError(f"Rust crate has an invalid semantic version: {expected}")
|
||||
mismatches = {
|
||||
label: version for label, version in versions.items() if version != expected
|
||||
}
|
||||
if mismatches:
|
||||
details = "\n".join(
|
||||
f" - {label}: {version}" for label, version in mismatches.items()
|
||||
)
|
||||
raise ValueError(f"Expected every package to use {expected}:\n{details}")
|
||||
return expected
|
||||
|
||||
|
||||
def set_versions(version: str, root: Path = ROOT) -> None:
|
||||
if not SEMVER.fullmatch(version):
|
||||
raise ValueError(f"Invalid semantic version: {version}")
|
||||
|
||||
# Validate every expected location before writing the first file. This
|
||||
# prevents a stale manifest or generated file from leaving a partial bump.
|
||||
package_versions(root)
|
||||
|
||||
for _, relative, section in TOML_VERSIONS:
|
||||
_write_section_version(root / relative, section, version)
|
||||
|
||||
package_path = root / "napi/package.json"
|
||||
package = json.loads(package_path.read_text(encoding="utf-8"))
|
||||
package["version"] = version
|
||||
optional = package.get("optionalDependencies", {})
|
||||
for dependency in PLATFORM_PACKAGES:
|
||||
if dependency not in optional:
|
||||
raise ValueError(f"Missing Node optional dependency: {dependency}")
|
||||
optional[dependency] = version
|
||||
package_path.write_text(json.dumps(package, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
bun_path = root / "napi/bun.lock"
|
||||
bun_text = bun_path.read_text(encoding="utf-8")
|
||||
for dependency in PLATFORM_PACKAGES:
|
||||
pattern = rf'("{re.escape(dependency)}": ")[^"]+("[,])'
|
||||
bun_text, count = re.subn(
|
||||
pattern, rf"\g<1>{version}\g<2>", bun_text, count=1
|
||||
)
|
||||
if count != 1:
|
||||
raise ValueError(f"Missing Bun lock dependency: {dependency}")
|
||||
bun_path.write_text(bun_text, encoding="utf-8")
|
||||
|
||||
site_path = root / "site/index.html"
|
||||
site_text = site_path.read_text(encoding="utf-8")
|
||||
site_text, count = SITE_WASM_VERSION.subn(
|
||||
rf"\g<1>{version}\g<3>", site_text, count=1
|
||||
)
|
||||
if count != 1:
|
||||
raise ValueError("Missing pinned WASM package URL in site/index.html")
|
||||
site_path.write_text(site_text, encoding="utf-8")
|
||||
|
||||
for _, relative, package_name in LOCK_VERSIONS:
|
||||
_write_lock_version(root / relative, package_name, version)
|
||||
|
||||
check_versions(root)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("version", nargs="?", help="new shared semantic version")
|
||||
parser.add_argument(
|
||||
"--check", action="store_true", help="fail if package versions have diverged"
|
||||
)
|
||||
arguments = parser.parse_args()
|
||||
if arguments.check == bool(arguments.version):
|
||||
parser.error("provide either a version or --check")
|
||||
|
||||
try:
|
||||
if arguments.check:
|
||||
version = check_versions()
|
||||
print(f"All packages use {version}")
|
||||
else:
|
||||
set_versions(arguments.version)
|
||||
print(f"Updated all packages to {arguments.version}")
|
||||
except ValueError as error:
|
||||
parser.exit(1, f"{error}\n")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+353
-1097
File diff suppressed because it is too large
Load Diff
@@ -63,7 +63,6 @@ fn json_escape(s: &str) -> String {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
|
||||
+5
-33
@@ -2,8 +2,8 @@
|
||||
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages_with_password, process_pdf_with_options, LayoutComplexity,
|
||||
PdfOptions, PdfType, ProcessMode, TextItem,
|
||||
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
|
||||
PdfType, ProcessMode, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
use std::env;
|
||||
@@ -103,18 +103,9 @@ fn format_items_json(items: &[TextItem]) -> String {
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_items_json(
|
||||
pdf_path: &str,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<String, pdf_inspector::PdfError> {
|
||||
extract_text_with_positions_pages_with_password(pdf_path, page_filter, password)
|
||||
.map(|items| format_items_json(&items))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{extract_items_json, format_items_json};
|
||||
use super::format_items_json;
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
@@ -146,24 +137,6 @@ mod tests {
|
||||
assert!(json.contains(r#""item_type":"text""#));
|
||||
assert!(json.contains(r#""mcid":7"#));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn items_json_uses_supplied_pdf_password() {
|
||||
let path = "tests/fixtures/encrypted-secret123.pdf";
|
||||
|
||||
let without_password = extract_items_json(path, None, None);
|
||||
assert!(
|
||||
without_password.is_err(),
|
||||
"encrypted fixture unexpectedly extracted without a password"
|
||||
);
|
||||
|
||||
let json = extract_items_json(path, None, Some("secret123"))
|
||||
.expect("correct password should decrypt positioned text");
|
||||
assert!(
|
||||
json.contains("Procurement"),
|
||||
"decrypted item JSON should contain fixture text, got {json}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
@@ -217,7 +190,6 @@ fn print_layout_info(layout: &LayoutComplexity) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
@@ -284,8 +256,8 @@ fn main() {
|
||||
});
|
||||
|
||||
if items_json_output {
|
||||
match extract_items_json(pdf_path, page_filter.as_ref(), password.as_deref()) {
|
||||
Ok(json) => println!("{}", json),
|
||||
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
|
||||
Ok(items) => println!("{}", format_items_json(&items)),
|
||||
Err(e) => {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
process::exit(1);
|
||||
|
||||
+1
-65
@@ -1659,15 +1659,7 @@ fn hex_val(b: u8) -> Option<u8> {
|
||||
/// Standard page: 612x792 points (US Letter) = ~485,000 sq points
|
||||
/// At 2x resolution that's ~1.9M pixels, so we use 250K pixels as threshold
|
||||
/// (accounting for varying DPI and page sizes)
|
||||
/// Returns `(has_images, total_image_area, has_template_image)` for a page.
|
||||
/// `has_template_image` means a single large (>50% page coverage)
|
||||
/// background image — the signal `classify_pdf`/`detect_pdf_type` uses to
|
||||
/// route a page to OCR regardless of any incidental native text drawn over
|
||||
/// it. Exposed at crate visibility so extraction-side per-page `needs_ocr`
|
||||
/// computation (`extract_pages_markdown_mem`) can consult the same signal
|
||||
/// instead of maintaining its own, independent notion of "needs OCR" that
|
||||
/// can silently disagree with detection — see #227.
|
||||
pub(crate) fn analyze_page_images(doc: &Document, page_id: ObjectId) -> (bool, u64, bool) {
|
||||
fn analyze_page_images(doc: &Document, page_id: ObjectId) -> (bool, u64, bool) {
|
||||
// Threshold: image covering roughly half a page at 150+ DPI
|
||||
// 612 * 792 / 2 * (150/72)^2 ≈ 1M pixels, but we'll be conservative
|
||||
const TEMPLATE_IMAGE_THRESHOLD: u64 = 500_000; // 500K pixels
|
||||
@@ -1749,62 +1741,6 @@ pub(crate) fn analyze_page_images(doc: &Document, page_id: ObjectId) -> (bool, u
|
||||
(has_images, total_area, has_template_image)
|
||||
}
|
||||
|
||||
/// Computes both `(needs_ocr_for_template_image, has_vector_text)` for a
|
||||
/// page from a single shared `analyze_page_content` pass — that call
|
||||
/// decompresses and scans every content stream (page + XObjects) plus
|
||||
/// image coverage, so `extract_pages_markdown_mem` must not invoke it
|
||||
/// twice per page (once per signal) the way `detect_from_document` avoids
|
||||
/// by caching its per-page `PageAnalysis`.
|
||||
///
|
||||
/// `needs_ocr_for_template_image` is true when a page's template image
|
||||
/// should be treated as a scan needing OCR — a single full-page background
|
||||
/// image with little/no real text — rather than a text page that happens
|
||||
/// to carry a watermark, letterhead, or figure. Mirrors the two distinct
|
||||
/// signals classification uses to route a template-image page to OCR:
|
||||
///
|
||||
/// 1. `looks_like_scan`: image_count <= 1, few text operators (<50), and
|
||||
/// low alphanumeric diversity in raw string operands (unless decodable
|
||||
/// CID/ToUnicode fonts explain that away) — the gate used for
|
||||
/// `pages_with_template_images` and Mixed-type per-page routing.
|
||||
/// 2. Insufficient real text volume, using `DetectionConfig::default()`'s
|
||||
/// `min_text_ops_per_page` (3) — the same threshold Mixed-type per-page
|
||||
/// routing applies via `text_operator_count < config.min_text_ops_per_page
|
||||
/// && has_images` (simplified here since a template image implies
|
||||
/// `has_images`). Deliberately *not* the higher `effective_min_ops`
|
||||
/// floor (`min_text_ops_per_page.max(10)`) that whole-document
|
||||
/// `PdfType::ImageBased`/`Scanned` classification uses for
|
||||
/// `pages_with_text` — that's a cross-page aggregate decision this
|
||||
/// per-page function has no way to replicate exactly, and the lower
|
||||
/// per-page threshold is the one a single page's own signals can
|
||||
/// actually agree with.
|
||||
///
|
||||
/// `has_vector_text` is true when a page has vector-outlined text (glyphs
|
||||
/// drawn as paths rather than shown via text-showing operators) —
|
||||
/// `detect_from_document`'s Mixed-type per-page routing always sends
|
||||
/// these pages to OCR, independent of any template-image check, since
|
||||
/// outlined glyphs can't be extracted as text at all.
|
||||
///
|
||||
/// Exposed at crate visibility so `extract_pages_markdown_mem` can apply
|
||||
/// the same gates classification needs elsewhere instead of treating the
|
||||
/// raw signals alone as sufficient — see #227/#231.
|
||||
pub(crate) fn page_ocr_signals(doc: &Document, page_id: ObjectId) -> (bool, bool) {
|
||||
let analysis = analyze_page_content(doc, page_id);
|
||||
|
||||
let needs_ocr_for_template_image = if !analysis.has_template_image {
|
||||
false
|
||||
} else {
|
||||
let alphanum_low = analysis.unique_alphanum_chars < 10
|
||||
&& !(analysis.has_decodable_text_fonts && analysis.text_operator_count >= 10);
|
||||
let looks_like_scan =
|
||||
analysis.image_count <= 1 && analysis.text_operator_count < 50 && alphanum_low;
|
||||
let insufficient_text =
|
||||
analysis.text_operator_count < DetectionConfig::default().min_text_ops_per_page;
|
||||
looks_like_scan || insufficient_text
|
||||
};
|
||||
|
||||
(needs_ocr_for_template_image, analysis.has_vector_text)
|
||||
}
|
||||
|
||||
/// Recursively collect image dimensions from XObject resources,
|
||||
/// including images nested inside Form XObjects.
|
||||
fn collect_images_from_resources(
|
||||
|
||||
@@ -1,649 +0,0 @@
|
||||
//! Built-in glyph metrics for the 14 standard PDF fonts.
|
||||
//!
|
||||
//! PDFs may omit `/Widths` for non-embedded base-14 fonts (Times, Helvetica,
|
||||
//! Courier, Symbol, ZapfDingbats); per the PDF spec the reader must supply
|
||||
//! the metrics. Without them every text item gets width 0, which breaks
|
||||
//! space synthesis, sub/superscript detection, and table column detection
|
||||
//! (common in 1990s dvips/Distiller output).
|
||||
//!
|
||||
//! Tables are generated from the Adobe Core 14 AFM files (via reportlab's
|
||||
//! `_fontdata`), keyed by Unicode char, sorted for binary search.
|
||||
//! Generator: scratchpad/gen_base14.py (session tooling, not checked in).
|
||||
|
||||
/// Width in 1000ths of an em for `c` in the given base-14 font, or `None`
|
||||
/// if the font is not one of the base 14 (after name normalization) or the
|
||||
/// char has no glyph in its AFM.
|
||||
pub(crate) fn base14_char_width(base_font: &str, c: char) -> Option<u16> {
|
||||
let table = base14_table(base_font)?;
|
||||
// AFM tables key visible glyphs only; alias the invisible variants the
|
||||
// cp1252 fallback can produce so they get the metric of their visible
|
||||
// counterpart instead of the generic default.
|
||||
let c = match c {
|
||||
'\u{00A0}' => ' ', // no-break space -> space
|
||||
'\u{00AD}' => '-', // soft hyphen -> hyphen
|
||||
_ => c,
|
||||
};
|
||||
table
|
||||
.binary_search_by_key(&c, |&(ch, _)| ch)
|
||||
.ok()
|
||||
.map(|i| table[i].1)
|
||||
}
|
||||
|
||||
/// True when the base font name normalizes to one of the standard 14 fonts.
|
||||
pub(crate) fn is_base14_font(base_font: &str) -> bool {
|
||||
base14_table(base_font).is_some()
|
||||
}
|
||||
|
||||
/// Code → Unicode through the font's BUILT-IN encoding, for the base-14
|
||||
/// fonts whose repertoire is not Latin (Symbol, ZapfDingbats). Their glyphs
|
||||
/// live at byte positions that have nothing to do with cp1252 (Symbol 0x61
|
||||
/// renders α, Zapf 0x21 renders ✁), so advance widths must be resolved
|
||||
/// through this mapping — the renderer draws these glyphs regardless of how
|
||||
/// the text decoder transliterates them. Returns `None` for the Latin text
|
||||
/// fonts, which follow standard single-byte encodings.
|
||||
pub(crate) fn builtin_encoding_char(base_font: &str, code: u8) -> Option<char> {
|
||||
let table = base14_table(base_font)?;
|
||||
let enc: &[(u8, char)] = if std::ptr::eq(table, SYMBOL) {
|
||||
SYMBOL_ENCODING
|
||||
} else if std::ptr::eq(table, ZAPFDINGBATS) {
|
||||
ZAPFDINGBATS_ENCODING
|
||||
} else {
|
||||
return None;
|
||||
};
|
||||
enc.binary_search_by_key(&code, |&(b, _)| b)
|
||||
.ok()
|
||||
.map(|i| enc[i].1)
|
||||
}
|
||||
|
||||
/// Map a BaseFont name (possibly subset-prefixed, e.g. "ABCDEF+Times-Bold",
|
||||
/// or a common alias like "Arial" / "TimesNewRomanPSMT") to its width table.
|
||||
fn base14_table(base_font: &str) -> Option<&'static [(char, u16)]> {
|
||||
// Strip subset prefix "ABCDEF+"
|
||||
let name = match base_font.split_once('+') {
|
||||
Some((prefix, rest))
|
||||
if prefix.len() == 6 && prefix.chars().all(|c| c.is_ascii_uppercase()) =>
|
||||
{
|
||||
rest
|
||||
}
|
||||
_ => base_font,
|
||||
};
|
||||
let lower = name.to_ascii_lowercase();
|
||||
let bold = lower.contains("bold");
|
||||
let italic = lower.contains("italic") || lower.contains("oblique");
|
||||
if lower.contains("courier") {
|
||||
return Some(match (bold, italic) {
|
||||
(false, false) => COURIER,
|
||||
(true, false) => COURIER_BOLD,
|
||||
(false, true) => COURIER_OBLIQUE,
|
||||
(true, true) => COURIER_BOLDOBLIQUE,
|
||||
});
|
||||
}
|
||||
if lower.contains("helvetica") || lower.contains("arial") {
|
||||
return Some(match (bold, italic) {
|
||||
(false, false) => HELVETICA,
|
||||
(true, false) => HELVETICA_BOLD,
|
||||
(false, true) => HELVETICA_OBLIQUE,
|
||||
(true, true) => HELVETICA_BOLDOBLIQUE,
|
||||
});
|
||||
}
|
||||
if lower.contains("times") {
|
||||
return Some(match (bold, italic) {
|
||||
(false, false) => TIMES_ROMAN,
|
||||
(true, false) => TIMES_BOLD,
|
||||
(false, true) => TIMES_ITALIC,
|
||||
(true, true) => TIMES_BOLDITALIC,
|
||||
});
|
||||
}
|
||||
// Symbol and ZapfDingbats have unique glyph repertoires, so only exact
|
||||
// names (plus the common MT/ITC aliases) qualify — a custom font that
|
||||
// merely mentions "Symbol" in its name must not get these metrics.
|
||||
match lower.as_str() {
|
||||
"zapfdingbats" | "dingbats" | "itczapfdingbats" | "zapfdingbatsitc" => {
|
||||
return Some(ZAPFDINGBATS)
|
||||
}
|
||||
"symbol" | "symbolmt" | "symbolitc" => return Some(SYMBOL),
|
||||
_ => {}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[rustfmt::skip]
|
||||
static COURIER: &[(char, u16)] = &[
|
||||
(' ', 600), ('!', 600), ('"', 600), ('#', 600), ('$', 600), ('%', 600),
|
||||
('&', 600), ('\'', 600), ('(', 600), (')', 600), ('*', 600), ('+', 600),
|
||||
(',', 600), ('-', 600), ('.', 600), ('/', 600), ('0', 600), ('1', 600),
|
||||
('2', 600), ('3', 600), ('4', 600), ('5', 600), ('6', 600), ('7', 600),
|
||||
('8', 600), ('9', 600), (':', 600), (';', 600), ('<', 600), ('=', 600),
|
||||
('>', 600), ('?', 600), ('@', 600), ('A', 600), ('B', 600), ('C', 600),
|
||||
('D', 600), ('E', 600), ('F', 600), ('G', 600), ('H', 600), ('I', 600),
|
||||
('J', 600), ('K', 600), ('L', 600), ('M', 600), ('N', 600), ('O', 600),
|
||||
('P', 600), ('Q', 600), ('R', 600), ('S', 600), ('T', 600), ('U', 600),
|
||||
('V', 600), ('W', 600), ('X', 600), ('Y', 600), ('Z', 600), ('[', 600),
|
||||
('\\', 600), (']', 600), ('^', 600), ('_', 600), ('`', 600), ('a', 600),
|
||||
('b', 600), ('c', 600), ('d', 600), ('e', 600), ('f', 600), ('g', 600),
|
||||
('h', 600), ('i', 600), ('j', 600), ('k', 600), ('l', 600), ('m', 600),
|
||||
('n', 600), ('o', 600), ('p', 600), ('q', 600), ('r', 600), ('s', 600),
|
||||
('t', 600), ('u', 600), ('v', 600), ('w', 600), ('x', 600), ('y', 600),
|
||||
('z', 600), ('{', 600), ('|', 600), ('}', 600), ('~', 600), ('\u{00A1}', 600),
|
||||
('\u{00A2}', 600), ('\u{00A3}', 600), ('\u{00A4}', 600), ('\u{00A5}', 600), ('\u{00A6}', 600), ('\u{00A7}', 600),
|
||||
('\u{00A8}', 600), ('\u{00A9}', 600), ('\u{00AA}', 600), ('\u{00AB}', 600), ('\u{00AC}', 600), ('\u{00AE}', 600),
|
||||
('\u{00AF}', 600), ('\u{00B0}', 600), ('\u{00B1}', 600), ('\u{00B2}', 600), ('\u{00B3}', 600), ('\u{00B4}', 600),
|
||||
('\u{00B5}', 600), ('\u{00B6}', 600), ('\u{00B7}', 600), ('\u{00B8}', 600), ('\u{00B9}', 600), ('\u{00BA}', 600),
|
||||
('\u{00BB}', 600), ('\u{00BC}', 600), ('\u{00BD}', 600), ('\u{00BE}', 600), ('\u{00BF}', 600), ('\u{00C0}', 600),
|
||||
('\u{00C1}', 600), ('\u{00C2}', 600), ('\u{00C3}', 600), ('\u{00C4}', 600), ('\u{00C5}', 600), ('\u{00C6}', 600),
|
||||
('\u{00C7}', 600), ('\u{00C8}', 600), ('\u{00C9}', 600), ('\u{00CA}', 600), ('\u{00CB}', 600), ('\u{00CC}', 600),
|
||||
('\u{00CD}', 600), ('\u{00CE}', 600), ('\u{00CF}', 600), ('\u{00D0}', 600), ('\u{00D1}', 600), ('\u{00D2}', 600),
|
||||
('\u{00D3}', 600), ('\u{00D4}', 600), ('\u{00D5}', 600), ('\u{00D6}', 600), ('\u{00D7}', 600), ('\u{00D8}', 600),
|
||||
('\u{00D9}', 600), ('\u{00DA}', 600), ('\u{00DB}', 600), ('\u{00DC}', 600), ('\u{00DD}', 600), ('\u{00DE}', 600),
|
||||
('\u{00DF}', 600), ('\u{00E0}', 600), ('\u{00E1}', 600), ('\u{00E2}', 600), ('\u{00E3}', 600), ('\u{00E4}', 600),
|
||||
('\u{00E5}', 600), ('\u{00E6}', 600), ('\u{00E7}', 600), ('\u{00E8}', 600), ('\u{00E9}', 600), ('\u{00EA}', 600),
|
||||
('\u{00EB}', 600), ('\u{00EC}', 600), ('\u{00ED}', 600), ('\u{00EE}', 600), ('\u{00EF}', 600), ('\u{00F0}', 600),
|
||||
('\u{00F1}', 600), ('\u{00F2}', 600), ('\u{00F3}', 600), ('\u{00F4}', 600), ('\u{00F5}', 600), ('\u{00F6}', 600),
|
||||
('\u{00F7}', 600), ('\u{00F8}', 600), ('\u{00F9}', 600), ('\u{00FA}', 600), ('\u{00FB}', 600), ('\u{00FC}', 600),
|
||||
('\u{00FD}', 600), ('\u{00FE}', 600), ('\u{00FF}', 600), ('\u{0131}', 600), ('\u{0141}', 600), ('\u{0142}', 600),
|
||||
('\u{0152}', 600), ('\u{0153}', 600), ('\u{0160}', 600), ('\u{0161}', 600), ('\u{0178}', 600), ('\u{017D}', 600),
|
||||
('\u{017E}', 600), ('\u{0192}', 600), ('\u{02C6}', 600), ('\u{02C7}', 600), ('\u{02D8}', 600), ('\u{02D9}', 600),
|
||||
('\u{02DA}', 600), ('\u{02DB}', 600), ('\u{02DC}', 600), ('\u{02DD}', 600), ('\u{2013}', 600), ('\u{2014}', 600),
|
||||
('\u{2018}', 600), ('\u{2019}', 600), ('\u{201A}', 600), ('\u{201C}', 600), ('\u{201D}', 600), ('\u{201E}', 600),
|
||||
('\u{2020}', 600), ('\u{2021}', 600), ('\u{2022}', 600), ('\u{2026}', 600), ('\u{2030}', 600), ('\u{2039}', 600),
|
||||
('\u{203A}', 600), ('\u{2044}', 600), ('\u{20AC}', 600), ('\u{2122}', 600), ('\u{2212}', 600), ('\u{FB01}', 600),
|
||||
('\u{FB02}', 600),
|
||||
];
|
||||
|
||||
static COURIER_BOLD: &[(char, u16)] = COURIER;
|
||||
|
||||
static COURIER_OBLIQUE: &[(char, u16)] = COURIER;
|
||||
|
||||
static COURIER_BOLDOBLIQUE: &[(char, u16)] = COURIER;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA: &[(char, u16)] = &[
|
||||
(' ', 278), ('!', 278), ('"', 355), ('#', 556), ('$', 556), ('%', 889),
|
||||
('&', 667), ('\'', 191), ('(', 333), (')', 333), ('*', 389), ('+', 584),
|
||||
(',', 278), ('-', 333), ('.', 278), ('/', 278), ('0', 556), ('1', 556),
|
||||
('2', 556), ('3', 556), ('4', 556), ('5', 556), ('6', 556), ('7', 556),
|
||||
('8', 556), ('9', 556), (':', 278), (';', 278), ('<', 584), ('=', 584),
|
||||
('>', 584), ('?', 556), ('@', 1015), ('A', 667), ('B', 667), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 722), ('I', 278),
|
||||
('J', 500), ('K', 667), ('L', 556), ('M', 833), ('N', 722), ('O', 778),
|
||||
('P', 667), ('Q', 778), ('R', 722), ('S', 667), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 944), ('X', 667), ('Y', 667), ('Z', 611), ('[', 278),
|
||||
('\\', 278), (']', 278), ('^', 469), ('_', 556), ('`', 333), ('a', 556),
|
||||
('b', 556), ('c', 500), ('d', 556), ('e', 556), ('f', 278), ('g', 556),
|
||||
('h', 556), ('i', 222), ('j', 222), ('k', 500), ('l', 222), ('m', 833),
|
||||
('n', 556), ('o', 556), ('p', 556), ('q', 556), ('r', 333), ('s', 500),
|
||||
('t', 278), ('u', 556), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 500), ('{', 334), ('|', 260), ('}', 334), ('~', 584), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 556), ('\u{00A3}', 556), ('\u{00A4}', 556), ('\u{00A5}', 556), ('\u{00A6}', 260), ('\u{00A7}', 556),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 737), ('\u{00AA}', 370), ('\u{00AB}', 556), ('\u{00AC}', 584), ('\u{00AE}', 737),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 584), ('\u{00B2}', 333), ('\u{00B3}', 333), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 556), ('\u{00B6}', 537), ('\u{00B7}', 278), ('\u{00B8}', 333), ('\u{00B9}', 333), ('\u{00BA}', 365),
|
||||
('\u{00BB}', 556), ('\u{00BC}', 834), ('\u{00BD}', 834), ('\u{00BE}', 834), ('\u{00BF}', 611), ('\u{00C0}', 667),
|
||||
('\u{00C1}', 667), ('\u{00C2}', 667), ('\u{00C3}', 667), ('\u{00C4}', 667), ('\u{00C5}', 667), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 278),
|
||||
('\u{00CD}', 278), ('\u{00CE}', 278), ('\u{00CF}', 278), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 584), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 667), ('\u{00DE}', 667),
|
||||
('\u{00DF}', 611), ('\u{00E0}', 556), ('\u{00E1}', 556), ('\u{00E2}', 556), ('\u{00E3}', 556), ('\u{00E4}', 556),
|
||||
('\u{00E5}', 556), ('\u{00E6}', 889), ('\u{00E7}', 500), ('\u{00E8}', 556), ('\u{00E9}', 556), ('\u{00EA}', 556),
|
||||
('\u{00EB}', 556), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 556),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 556), ('\u{00F3}', 556), ('\u{00F4}', 556), ('\u{00F5}', 556), ('\u{00F6}', 556),
|
||||
('\u{00F7}', 584), ('\u{00F8}', 611), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 556), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 556), ('\u{0142}', 222),
|
||||
('\u{0152}', 1000), ('\u{0153}', 944), ('\u{0160}', 667), ('\u{0161}', 500), ('\u{0178}', 667), ('\u{017D}', 611),
|
||||
('\u{017E}', 500), ('\u{0192}', 556), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 556), ('\u{2014}', 1000),
|
||||
('\u{2018}', 222), ('\u{2019}', 222), ('\u{201A}', 222), ('\u{201C}', 333), ('\u{201D}', 333), ('\u{201E}', 333),
|
||||
('\u{2020}', 556), ('\u{2021}', 556), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 556), ('\u{2122}', 1000), ('\u{2212}', 584), ('\u{FB01}', 500),
|
||||
('\u{FB02}', 500),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA_BOLD: &[(char, u16)] = &[
|
||||
(' ', 278), ('!', 333), ('"', 474), ('#', 556), ('$', 556), ('%', 889),
|
||||
('&', 722), ('\'', 238), ('(', 333), (')', 333), ('*', 389), ('+', 584),
|
||||
(',', 278), ('-', 333), ('.', 278), ('/', 278), ('0', 556), ('1', 556),
|
||||
('2', 556), ('3', 556), ('4', 556), ('5', 556), ('6', 556), ('7', 556),
|
||||
('8', 556), ('9', 556), (':', 333), (';', 333), ('<', 584), ('=', 584),
|
||||
('>', 584), ('?', 611), ('@', 975), ('A', 722), ('B', 722), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 722), ('I', 278),
|
||||
('J', 556), ('K', 722), ('L', 611), ('M', 833), ('N', 722), ('O', 778),
|
||||
('P', 667), ('Q', 778), ('R', 722), ('S', 667), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 944), ('X', 667), ('Y', 667), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 584), ('_', 556), ('`', 333), ('a', 556),
|
||||
('b', 611), ('c', 556), ('d', 611), ('e', 556), ('f', 333), ('g', 611),
|
||||
('h', 611), ('i', 278), ('j', 278), ('k', 556), ('l', 278), ('m', 889),
|
||||
('n', 611), ('o', 611), ('p', 611), ('q', 611), ('r', 389), ('s', 556),
|
||||
('t', 333), ('u', 611), ('v', 556), ('w', 778), ('x', 556), ('y', 556),
|
||||
('z', 500), ('{', 389), ('|', 280), ('}', 389), ('~', 584), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 556), ('\u{00A3}', 556), ('\u{00A4}', 556), ('\u{00A5}', 556), ('\u{00A6}', 280), ('\u{00A7}', 556),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 737), ('\u{00AA}', 370), ('\u{00AB}', 556), ('\u{00AC}', 584), ('\u{00AE}', 737),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 584), ('\u{00B2}', 333), ('\u{00B3}', 333), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 611), ('\u{00B6}', 556), ('\u{00B7}', 278), ('\u{00B8}', 333), ('\u{00B9}', 333), ('\u{00BA}', 365),
|
||||
('\u{00BB}', 556), ('\u{00BC}', 834), ('\u{00BD}', 834), ('\u{00BE}', 834), ('\u{00BF}', 611), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 278),
|
||||
('\u{00CD}', 278), ('\u{00CE}', 278), ('\u{00CF}', 278), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 584), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 667), ('\u{00DE}', 667),
|
||||
('\u{00DF}', 611), ('\u{00E0}', 556), ('\u{00E1}', 556), ('\u{00E2}', 556), ('\u{00E3}', 556), ('\u{00E4}', 556),
|
||||
('\u{00E5}', 556), ('\u{00E6}', 889), ('\u{00E7}', 556), ('\u{00E8}', 556), ('\u{00E9}', 556), ('\u{00EA}', 556),
|
||||
('\u{00EB}', 556), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 611),
|
||||
('\u{00F1}', 611), ('\u{00F2}', 611), ('\u{00F3}', 611), ('\u{00F4}', 611), ('\u{00F5}', 611), ('\u{00F6}', 611),
|
||||
('\u{00F7}', 584), ('\u{00F8}', 611), ('\u{00F9}', 611), ('\u{00FA}', 611), ('\u{00FB}', 611), ('\u{00FC}', 611),
|
||||
('\u{00FD}', 556), ('\u{00FE}', 611), ('\u{00FF}', 556), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 1000), ('\u{0153}', 944), ('\u{0160}', 667), ('\u{0161}', 556), ('\u{0178}', 667), ('\u{017D}', 611),
|
||||
('\u{017E}', 500), ('\u{0192}', 556), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 556), ('\u{2014}', 1000),
|
||||
('\u{2018}', 278), ('\u{2019}', 278), ('\u{201A}', 278), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 556), ('\u{2021}', 556), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 556), ('\u{2122}', 1000), ('\u{2212}', 584), ('\u{FB01}', 611),
|
||||
('\u{FB02}', 611),
|
||||
];
|
||||
|
||||
static HELVETICA_OBLIQUE: &[(char, u16)] = HELVETICA;
|
||||
|
||||
static HELVETICA_BOLDOBLIQUE: &[(char, u16)] = HELVETICA_BOLD;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_ROMAN: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('"', 408), ('#', 500), ('$', 500), ('%', 833),
|
||||
('&', 778), ('\'', 180), ('(', 333), (')', 333), ('*', 500), ('+', 564),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 278), (';', 278), ('<', 564), ('=', 564),
|
||||
('>', 564), ('?', 444), ('@', 921), ('A', 722), ('B', 667), ('C', 667),
|
||||
('D', 722), ('E', 611), ('F', 556), ('G', 722), ('H', 722), ('I', 333),
|
||||
('J', 389), ('K', 722), ('L', 611), ('M', 889), ('N', 722), ('O', 722),
|
||||
('P', 556), ('Q', 722), ('R', 667), ('S', 556), ('T', 611), ('U', 722),
|
||||
('V', 722), ('W', 944), ('X', 722), ('Y', 722), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 469), ('_', 500), ('`', 333), ('a', 444),
|
||||
('b', 500), ('c', 444), ('d', 500), ('e', 444), ('f', 333), ('g', 500),
|
||||
('h', 500), ('i', 278), ('j', 278), ('k', 500), ('l', 278), ('m', 778),
|
||||
('n', 500), ('o', 500), ('p', 500), ('q', 500), ('r', 333), ('s', 389),
|
||||
('t', 278), ('u', 500), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 444), ('{', 480), ('|', 200), ('}', 480), ('~', 541), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 200), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 760), ('\u{00AA}', 276), ('\u{00AB}', 500), ('\u{00AC}', 564), ('\u{00AE}', 760),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 564), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 500), ('\u{00B6}', 453), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 310),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 444), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 889),
|
||||
('\u{00C7}', 667), ('\u{00C8}', 611), ('\u{00C9}', 611), ('\u{00CA}', 611), ('\u{00CB}', 611), ('\u{00CC}', 333),
|
||||
('\u{00CD}', 333), ('\u{00CE}', 333), ('\u{00CF}', 333), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 722),
|
||||
('\u{00D3}', 722), ('\u{00D4}', 722), ('\u{00D5}', 722), ('\u{00D6}', 722), ('\u{00D7}', 564), ('\u{00D8}', 722),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 722), ('\u{00DE}', 556),
|
||||
('\u{00DF}', 500), ('\u{00E0}', 444), ('\u{00E1}', 444), ('\u{00E2}', 444), ('\u{00E3}', 444), ('\u{00E4}', 444),
|
||||
('\u{00E5}', 444), ('\u{00E6}', 667), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 500), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 564), ('\u{00F8}', 500), ('\u{00F9}', 500), ('\u{00FA}', 500), ('\u{00FB}', 500), ('\u{00FC}', 500),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 500), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 889), ('\u{0153}', 722), ('\u{0160}', 556), ('\u{0161}', 389), ('\u{0178}', 722), ('\u{017D}', 611),
|
||||
('\u{017E}', 444), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 1000),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 444), ('\u{201D}', 444), ('\u{201E}', 444),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 980), ('\u{2212}', 564), ('\u{FB01}', 556),
|
||||
('\u{FB02}', 556),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_BOLD: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('"', 555), ('#', 500), ('$', 500), ('%', 1000),
|
||||
('&', 833), ('\'', 278), ('(', 333), (')', 333), ('*', 500), ('+', 570),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 333), (';', 333), ('<', 570), ('=', 570),
|
||||
('>', 570), ('?', 500), ('@', 930), ('A', 722), ('B', 667), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 778), ('I', 389),
|
||||
('J', 500), ('K', 778), ('L', 667), ('M', 944), ('N', 722), ('O', 778),
|
||||
('P', 611), ('Q', 778), ('R', 722), ('S', 556), ('T', 667), ('U', 722),
|
||||
('V', 722), ('W', 1000), ('X', 722), ('Y', 722), ('Z', 667), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 581), ('_', 500), ('`', 333), ('a', 500),
|
||||
('b', 556), ('c', 444), ('d', 556), ('e', 444), ('f', 333), ('g', 500),
|
||||
('h', 556), ('i', 278), ('j', 333), ('k', 556), ('l', 278), ('m', 833),
|
||||
('n', 556), ('o', 500), ('p', 556), ('q', 556), ('r', 444), ('s', 389),
|
||||
('t', 333), ('u', 556), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 444), ('{', 394), ('|', 220), ('}', 394), ('~', 520), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 220), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 747), ('\u{00AA}', 300), ('\u{00AB}', 500), ('\u{00AC}', 570), ('\u{00AE}', 747),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 570), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 556), ('\u{00B6}', 540), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 330),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 500), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 389),
|
||||
('\u{00CD}', 389), ('\u{00CE}', 389), ('\u{00CF}', 389), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 570), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 722), ('\u{00DE}', 611),
|
||||
('\u{00DF}', 556), ('\u{00E0}', 500), ('\u{00E1}', 500), ('\u{00E2}', 500), ('\u{00E3}', 500), ('\u{00E4}', 500),
|
||||
('\u{00E5}', 500), ('\u{00E6}', 722), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 570), ('\u{00F8}', 500), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 556), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 667), ('\u{0142}', 278),
|
||||
('\u{0152}', 1000), ('\u{0153}', 722), ('\u{0160}', 556), ('\u{0161}', 389), ('\u{0178}', 722), ('\u{017D}', 667),
|
||||
('\u{017E}', 444), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 1000),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 1000), ('\u{2212}', 570), ('\u{FB01}', 556),
|
||||
('\u{FB02}', 556),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_ITALIC: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('"', 420), ('#', 500), ('$', 500), ('%', 833),
|
||||
('&', 778), ('\'', 214), ('(', 333), (')', 333), ('*', 500), ('+', 675),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 333), (';', 333), ('<', 675), ('=', 675),
|
||||
('>', 675), ('?', 500), ('@', 920), ('A', 611), ('B', 611), ('C', 667),
|
||||
('D', 722), ('E', 611), ('F', 611), ('G', 722), ('H', 722), ('I', 333),
|
||||
('J', 444), ('K', 667), ('L', 556), ('M', 833), ('N', 667), ('O', 722),
|
||||
('P', 611), ('Q', 722), ('R', 611), ('S', 500), ('T', 556), ('U', 722),
|
||||
('V', 611), ('W', 833), ('X', 611), ('Y', 556), ('Z', 556), ('[', 389),
|
||||
('\\', 278), (']', 389), ('^', 422), ('_', 500), ('`', 333), ('a', 500),
|
||||
('b', 500), ('c', 444), ('d', 500), ('e', 444), ('f', 278), ('g', 500),
|
||||
('h', 500), ('i', 278), ('j', 278), ('k', 444), ('l', 278), ('m', 722),
|
||||
('n', 500), ('o', 500), ('p', 500), ('q', 500), ('r', 389), ('s', 389),
|
||||
('t', 278), ('u', 500), ('v', 444), ('w', 667), ('x', 444), ('y', 444),
|
||||
('z', 389), ('{', 400), ('|', 275), ('}', 400), ('~', 541), ('\u{00A1}', 389),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 275), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 760), ('\u{00AA}', 276), ('\u{00AB}', 500), ('\u{00AC}', 675), ('\u{00AE}', 760),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 675), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 500), ('\u{00B6}', 523), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 310),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 500), ('\u{00C0}', 611),
|
||||
('\u{00C1}', 611), ('\u{00C2}', 611), ('\u{00C3}', 611), ('\u{00C4}', 611), ('\u{00C5}', 611), ('\u{00C6}', 889),
|
||||
('\u{00C7}', 667), ('\u{00C8}', 611), ('\u{00C9}', 611), ('\u{00CA}', 611), ('\u{00CB}', 611), ('\u{00CC}', 333),
|
||||
('\u{00CD}', 333), ('\u{00CE}', 333), ('\u{00CF}', 333), ('\u{00D0}', 722), ('\u{00D1}', 667), ('\u{00D2}', 722),
|
||||
('\u{00D3}', 722), ('\u{00D4}', 722), ('\u{00D5}', 722), ('\u{00D6}', 722), ('\u{00D7}', 675), ('\u{00D8}', 722),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 556), ('\u{00DE}', 611),
|
||||
('\u{00DF}', 500), ('\u{00E0}', 500), ('\u{00E1}', 500), ('\u{00E2}', 500), ('\u{00E3}', 500), ('\u{00E4}', 500),
|
||||
('\u{00E5}', 500), ('\u{00E6}', 667), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 500), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 675), ('\u{00F8}', 500), ('\u{00F9}', 500), ('\u{00FA}', 500), ('\u{00FB}', 500), ('\u{00FC}', 500),
|
||||
('\u{00FD}', 444), ('\u{00FE}', 500), ('\u{00FF}', 444), ('\u{0131}', 278), ('\u{0141}', 556), ('\u{0142}', 278),
|
||||
('\u{0152}', 944), ('\u{0153}', 667), ('\u{0160}', 500), ('\u{0161}', 389), ('\u{0178}', 556), ('\u{017D}', 556),
|
||||
('\u{017E}', 389), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 889),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 556), ('\u{201D}', 556), ('\u{201E}', 556),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 889), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 980), ('\u{2212}', 675), ('\u{FB01}', 500),
|
||||
('\u{FB02}', 500),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_BOLDITALIC: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 389), ('"', 555), ('#', 500), ('$', 500), ('%', 833),
|
||||
('&', 778), ('\'', 278), ('(', 333), (')', 333), ('*', 500), ('+', 570),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 333), (';', 333), ('<', 570), ('=', 570),
|
||||
('>', 570), ('?', 500), ('@', 832), ('A', 667), ('B', 667), ('C', 667),
|
||||
('D', 722), ('E', 667), ('F', 667), ('G', 722), ('H', 778), ('I', 389),
|
||||
('J', 500), ('K', 667), ('L', 611), ('M', 889), ('N', 722), ('O', 722),
|
||||
('P', 611), ('Q', 722), ('R', 667), ('S', 556), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 889), ('X', 667), ('Y', 611), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 570), ('_', 500), ('`', 333), ('a', 500),
|
||||
('b', 500), ('c', 444), ('d', 500), ('e', 444), ('f', 333), ('g', 500),
|
||||
('h', 556), ('i', 278), ('j', 278), ('k', 500), ('l', 278), ('m', 778),
|
||||
('n', 556), ('o', 500), ('p', 500), ('q', 500), ('r', 389), ('s', 389),
|
||||
('t', 278), ('u', 556), ('v', 444), ('w', 667), ('x', 500), ('y', 444),
|
||||
('z', 389), ('{', 348), ('|', 220), ('}', 348), ('~', 570), ('\u{00A1}', 389),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 220), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 747), ('\u{00AA}', 266), ('\u{00AB}', 500), ('\u{00AC}', 606), ('\u{00AE}', 747),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 570), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 576), ('\u{00B6}', 500), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 300),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 500), ('\u{00C0}', 667),
|
||||
('\u{00C1}', 667), ('\u{00C2}', 667), ('\u{00C3}', 667), ('\u{00C4}', 667), ('\u{00C5}', 667), ('\u{00C6}', 944),
|
||||
('\u{00C7}', 667), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 389),
|
||||
('\u{00CD}', 389), ('\u{00CE}', 389), ('\u{00CF}', 389), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 722),
|
||||
('\u{00D3}', 722), ('\u{00D4}', 722), ('\u{00D5}', 722), ('\u{00D6}', 722), ('\u{00D7}', 570), ('\u{00D8}', 722),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 611), ('\u{00DE}', 611),
|
||||
('\u{00DF}', 500), ('\u{00E0}', 500), ('\u{00E1}', 500), ('\u{00E2}', 500), ('\u{00E3}', 500), ('\u{00E4}', 500),
|
||||
('\u{00E5}', 500), ('\u{00E6}', 722), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 570), ('\u{00F8}', 500), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 444), ('\u{00FE}', 500), ('\u{00FF}', 444), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 944), ('\u{0153}', 722), ('\u{0160}', 556), ('\u{0161}', 389), ('\u{0178}', 611), ('\u{017D}', 611),
|
||||
('\u{017E}', 389), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 1000),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 1000), ('\u{2212}', 606), ('\u{FB01}', 556),
|
||||
('\u{FB02}', 556),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static SYMBOL: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('#', 500), ('%', 833), ('&', 778), ('(', 333),
|
||||
(')', 333), ('+', 549), (',', 250), ('.', 250), ('/', 278), ('0', 500),
|
||||
('1', 500), ('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500),
|
||||
('7', 500), ('8', 500), ('9', 500), (':', 278), (';', 278), ('<', 549),
|
||||
('=', 549), ('>', 549), ('?', 444), ('[', 333), (']', 333), ('_', 500),
|
||||
('{', 480), ('|', 200), ('}', 480), ('\u{00AC}', 713), ('\u{00B0}', 400), ('\u{00B1}', 549),
|
||||
('\u{00B5}', 576), ('\u{00D7}', 549), ('\u{00F7}', 549), ('\u{0192}', 500), ('\u{0391}', 722), ('\u{0392}', 667),
|
||||
('\u{0393}', 603), ('\u{0395}', 611), ('\u{0396}', 611), ('\u{0397}', 722), ('\u{0398}', 741), ('\u{0399}', 333),
|
||||
('\u{039A}', 722), ('\u{039B}', 686), ('\u{039C}', 889), ('\u{039D}', 722), ('\u{039E}', 645), ('\u{039F}', 722),
|
||||
('\u{03A0}', 768), ('\u{03A1}', 556), ('\u{03A3}', 592), ('\u{03A4}', 611), ('\u{03A5}', 690), ('\u{03A6}', 763),
|
||||
('\u{03A7}', 722), ('\u{03A8}', 795), ('\u{03B1}', 631), ('\u{03B2}', 549), ('\u{03B3}', 411), ('\u{03B4}', 494),
|
||||
('\u{03B5}', 439), ('\u{03B6}', 494), ('\u{03B7}', 603), ('\u{03B8}', 521), ('\u{03B9}', 329), ('\u{03BA}', 549),
|
||||
('\u{03BB}', 549), ('\u{03BD}', 521), ('\u{03BE}', 493), ('\u{03BF}', 549), ('\u{03C0}', 549), ('\u{03C1}', 549),
|
||||
('\u{03C2}', 439), ('\u{03C3}', 603), ('\u{03C4}', 439), ('\u{03C5}', 576), ('\u{03C6}', 521), ('\u{03C7}', 549),
|
||||
('\u{03C8}', 686), ('\u{03C9}', 686), ('\u{03D1}', 631), ('\u{03D2}', 620), ('\u{03D5}', 603), ('\u{03D6}', 713),
|
||||
('\u{2022}', 460), ('\u{2026}', 1000), ('\u{2032}', 247), ('\u{2033}', 411), ('\u{2044}', 167), ('\u{20AC}', 750),
|
||||
('\u{2111}', 686), ('\u{2118}', 987), ('\u{211C}', 795), ('\u{2126}', 768), ('\u{2135}', 823), ('\u{2190}', 987),
|
||||
('\u{2191}', 603), ('\u{2192}', 987), ('\u{2193}', 603), ('\u{2194}', 1042), ('\u{21B5}', 658), ('\u{21D0}', 987),
|
||||
('\u{21D1}', 603), ('\u{21D2}', 987), ('\u{21D3}', 603), ('\u{21D4}', 1042), ('\u{2200}', 713), ('\u{2202}', 494),
|
||||
('\u{2203}', 549), ('\u{2205}', 823), ('\u{2206}', 612), ('\u{2207}', 713), ('\u{2208}', 713), ('\u{2209}', 713),
|
||||
('\u{220B}', 439), ('\u{220F}', 823), ('\u{2211}', 713), ('\u{2212}', 549), ('\u{2217}', 500), ('\u{221A}', 549),
|
||||
('\u{221D}', 713), ('\u{221E}', 713), ('\u{2220}', 768), ('\u{2227}', 603), ('\u{2228}', 603), ('\u{2229}', 768),
|
||||
('\u{222A}', 768), ('\u{222B}', 274), ('\u{2234}', 863), ('\u{223C}', 549), ('\u{2245}', 549), ('\u{2248}', 549),
|
||||
('\u{2260}', 549), ('\u{2261}', 549), ('\u{2264}', 549), ('\u{2265}', 549), ('\u{2282}', 713), ('\u{2283}', 713),
|
||||
('\u{2284}', 713), ('\u{2286}', 713), ('\u{2287}', 713), ('\u{2295}', 768), ('\u{2297}', 768), ('\u{22A5}', 658),
|
||||
('\u{22C5}', 250), ('\u{2320}', 686), ('\u{2321}', 686), ('\u{2329}', 329), ('\u{232A}', 329), ('\u{25CA}', 494),
|
||||
('\u{2660}', 753), ('\u{2663}', 753), ('\u{2665}', 753), ('\u{2666}', 753), ('\u{F6D9}', 790), ('\u{F6DA}', 790),
|
||||
('\u{F6DB}', 890), ('\u{F8E5}', 500), ('\u{F8E6}', 603), ('\u{F8E7}', 1000), ('\u{F8E8}', 790), ('\u{F8E9}', 790),
|
||||
('\u{F8EA}', 786), ('\u{F8EB}', 384), ('\u{F8EC}', 384), ('\u{F8ED}', 384), ('\u{F8EE}', 384), ('\u{F8EF}', 384),
|
||||
('\u{F8F0}', 384), ('\u{F8F1}', 494), ('\u{F8F2}', 494), ('\u{F8F3}', 494), ('\u{F8F4}', 494), ('\u{F8F5}', 686),
|
||||
('\u{F8F6}', 384), ('\u{F8F7}', 384), ('\u{F8F8}', 384), ('\u{F8F9}', 384), ('\u{F8FA}', 384), ('\u{F8FB}', 384),
|
||||
('\u{F8FC}', 494), ('\u{F8FD}', 494), ('\u{F8FE}', 494), ('\u{F8FF}', 790),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static ZAPFDINGBATS: &[(char, u16)] = &[
|
||||
(' ', 278), ('\u{2192}', 838), ('\u{2194}', 1016), ('\u{2195}', 458), ('\u{2460}', 788), ('\u{2461}', 788),
|
||||
('\u{2462}', 788), ('\u{2463}', 788), ('\u{2464}', 788), ('\u{2465}', 788), ('\u{2466}', 788), ('\u{2467}', 788),
|
||||
('\u{2468}', 788), ('\u{2469}', 788), ('\u{25A0}', 761), ('\u{25B2}', 892), ('\u{25BC}', 892), ('\u{25C6}', 788),
|
||||
('\u{25CF}', 791), ('\u{25D7}', 438), ('\u{2605}', 816), ('\u{260E}', 719), ('\u{261B}', 960), ('\u{261E}', 939),
|
||||
('\u{2660}', 626), ('\u{2663}', 776), ('\u{2665}', 694), ('\u{2666}', 595), ('\u{2701}', 974), ('\u{2702}', 961),
|
||||
('\u{2703}', 974), ('\u{2704}', 980), ('\u{2706}', 789), ('\u{2707}', 790), ('\u{2708}', 791), ('\u{2709}', 690),
|
||||
('\u{270C}', 549), ('\u{270D}', 855), ('\u{270E}', 911), ('\u{270F}', 933), ('\u{2710}', 911), ('\u{2711}', 945),
|
||||
('\u{2712}', 974), ('\u{2713}', 755), ('\u{2714}', 846), ('\u{2715}', 762), ('\u{2716}', 761), ('\u{2717}', 571),
|
||||
('\u{2718}', 677), ('\u{2719}', 763), ('\u{271A}', 760), ('\u{271B}', 759), ('\u{271C}', 754), ('\u{271D}', 494),
|
||||
('\u{271E}', 552), ('\u{271F}', 537), ('\u{2720}', 577), ('\u{2721}', 692), ('\u{2722}', 786), ('\u{2723}', 788),
|
||||
('\u{2724}', 788), ('\u{2725}', 790), ('\u{2726}', 793), ('\u{2727}', 794), ('\u{2729}', 823), ('\u{272A}', 789),
|
||||
('\u{272B}', 841), ('\u{272C}', 823), ('\u{272D}', 833), ('\u{272E}', 816), ('\u{272F}', 831), ('\u{2730}', 923),
|
||||
('\u{2731}', 744), ('\u{2732}', 723), ('\u{2733}', 749), ('\u{2734}', 790), ('\u{2735}', 792), ('\u{2736}', 695),
|
||||
('\u{2737}', 776), ('\u{2738}', 768), ('\u{2739}', 792), ('\u{273A}', 759), ('\u{273B}', 707), ('\u{273C}', 708),
|
||||
('\u{273D}', 682), ('\u{273E}', 701), ('\u{273F}', 826), ('\u{2740}', 815), ('\u{2741}', 789), ('\u{2742}', 789),
|
||||
('\u{2743}', 707), ('\u{2744}', 687), ('\u{2745}', 696), ('\u{2746}', 689), ('\u{2747}', 786), ('\u{2748}', 787),
|
||||
('\u{2749}', 713), ('\u{274A}', 791), ('\u{274B}', 785), ('\u{274D}', 873), ('\u{274F}', 762), ('\u{2750}', 762),
|
||||
('\u{2751}', 759), ('\u{2752}', 759), ('\u{2756}', 784), ('\u{2758}', 138), ('\u{2759}', 277), ('\u{275A}', 415),
|
||||
('\u{275B}', 392), ('\u{275C}', 392), ('\u{275D}', 668), ('\u{275E}', 668), ('\u{2761}', 732), ('\u{2762}', 544),
|
||||
('\u{2763}', 544), ('\u{2764}', 910), ('\u{2765}', 667), ('\u{2766}', 760), ('\u{2767}', 760), ('\u{2768}', 390),
|
||||
('\u{2769}', 390), ('\u{276A}', 317), ('\u{276B}', 317), ('\u{276C}', 276), ('\u{276D}', 276), ('\u{276E}', 509),
|
||||
('\u{276F}', 509), ('\u{2770}', 410), ('\u{2771}', 410), ('\u{2772}', 234), ('\u{2773}', 234), ('\u{2774}', 334),
|
||||
('\u{2775}', 334), ('\u{2776}', 788), ('\u{2777}', 788), ('\u{2778}', 788), ('\u{2779}', 788), ('\u{277A}', 788),
|
||||
('\u{277B}', 788), ('\u{277C}', 788), ('\u{277D}', 788), ('\u{277E}', 788), ('\u{277F}', 788), ('\u{2780}', 788),
|
||||
('\u{2781}', 788), ('\u{2782}', 788), ('\u{2783}', 788), ('\u{2784}', 788), ('\u{2785}', 788), ('\u{2786}', 788),
|
||||
('\u{2787}', 788), ('\u{2788}', 788), ('\u{2789}', 788), ('\u{278A}', 788), ('\u{278B}', 788), ('\u{278C}', 788),
|
||||
('\u{278D}', 788), ('\u{278E}', 788), ('\u{278F}', 788), ('\u{2790}', 788), ('\u{2791}', 788), ('\u{2792}', 788),
|
||||
('\u{2793}', 788), ('\u{2794}', 894), ('\u{2798}', 748), ('\u{2799}', 924), ('\u{279A}', 748), ('\u{279B}', 918),
|
||||
('\u{279C}', 927), ('\u{279D}', 928), ('\u{279E}', 928), ('\u{279F}', 834), ('\u{27A0}', 873), ('\u{27A1}', 828),
|
||||
('\u{27A2}', 924), ('\u{27A3}', 924), ('\u{27A4}', 917), ('\u{27A5}', 930), ('\u{27A6}', 931), ('\u{27A7}', 463),
|
||||
('\u{27A8}', 883), ('\u{27A9}', 836), ('\u{27AA}', 836), ('\u{27AB}', 867), ('\u{27AC}', 867), ('\u{27AD}', 696),
|
||||
('\u{27AE}', 696), ('\u{27AF}', 874), ('\u{27B1}', 874), ('\u{27B2}', 760), ('\u{27B3}', 946), ('\u{27B4}', 771),
|
||||
('\u{27B5}', 865), ('\u{27B6}', 771), ('\u{27B7}', 888), ('\u{27B8}', 967), ('\u{27B9}', 888), ('\u{27BA}', 831),
|
||||
('\u{27BB}', 873), ('\u{27BC}', 927), ('\u{27BD}', 970), ('\u{27BE}', 918),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static SYMBOL_ENCODING: &[(u8, char)] = &[
|
||||
(0x20, ' '), (0x21, '!'), (0x22, '\u{2200}'), (0x23, '#'), (0x24, '\u{2203}'), (0x25, '%'),
|
||||
(0x26, '&'), (0x27, '\u{220B}'), (0x28, '('), (0x29, ')'), (0x2A, '\u{2217}'), (0x2B, '+'),
|
||||
(0x2C, ','), (0x2D, '\u{2212}'), (0x2E, '.'), (0x2F, '/'), (0x30, '0'), (0x31, '1'),
|
||||
(0x32, '2'), (0x33, '3'), (0x34, '4'), (0x35, '5'), (0x36, '6'), (0x37, '7'),
|
||||
(0x38, '8'), (0x39, '9'), (0x3A, ':'), (0x3B, ';'), (0x3C, '<'), (0x3D, '='),
|
||||
(0x3E, '>'), (0x3F, '?'), (0x40, '\u{2245}'), (0x41, '\u{0391}'), (0x42, '\u{0392}'), (0x43, '\u{03A7}'),
|
||||
(0x44, '\u{2206}'), (0x45, '\u{0395}'), (0x46, '\u{03A6}'), (0x47, '\u{0393}'), (0x48, '\u{0397}'), (0x49, '\u{0399}'),
|
||||
(0x4A, '\u{03D1}'), (0x4B, '\u{039A}'), (0x4C, '\u{039B}'), (0x4D, '\u{039C}'), (0x4E, '\u{039D}'), (0x4F, '\u{039F}'),
|
||||
(0x50, '\u{03A0}'), (0x51, '\u{0398}'), (0x52, '\u{03A1}'), (0x53, '\u{03A3}'), (0x54, '\u{03A4}'), (0x55, '\u{03A5}'),
|
||||
(0x56, '\u{03C2}'), (0x57, '\u{2126}'), (0x58, '\u{039E}'), (0x59, '\u{03A8}'), (0x5A, '\u{0396}'), (0x5B, '['),
|
||||
(0x5C, '\u{2234}'), (0x5D, ']'), (0x5E, '\u{22A5}'), (0x5F, '_'), (0x60, '\u{F8E5}'), (0x61, '\u{03B1}'),
|
||||
(0x62, '\u{03B2}'), (0x63, '\u{03C7}'), (0x64, '\u{03B4}'), (0x65, '\u{03B5}'), (0x66, '\u{03C6}'), (0x67, '\u{03B3}'),
|
||||
(0x68, '\u{03B7}'), (0x69, '\u{03B9}'), (0x6A, '\u{03D5}'), (0x6B, '\u{03BA}'), (0x6C, '\u{03BB}'), (0x6D, '\u{00B5}'),
|
||||
(0x6E, '\u{03BD}'), (0x6F, '\u{03BF}'), (0x70, '\u{03C0}'), (0x71, '\u{03B8}'), (0x72, '\u{03C1}'), (0x73, '\u{03C3}'),
|
||||
(0x74, '\u{03C4}'), (0x75, '\u{03C5}'), (0x76, '\u{03D6}'), (0x77, '\u{03C9}'), (0x78, '\u{03BE}'), (0x79, '\u{03C8}'),
|
||||
(0x7A, '\u{03B6}'), (0x7B, '{'), (0x7C, '|'), (0x7D, '}'), (0x7E, '\u{223C}'), (0xA0, '\u{20AC}'),
|
||||
(0xA1, '\u{03D2}'), (0xA2, '\u{2032}'), (0xA3, '\u{2264}'), (0xA4, '\u{2044}'), (0xA5, '\u{221E}'), (0xA6, '\u{0192}'),
|
||||
(0xA7, '\u{2663}'), (0xA8, '\u{2666}'), (0xA9, '\u{2665}'), (0xAA, '\u{2660}'), (0xAB, '\u{2194}'), (0xAC, '\u{2190}'),
|
||||
(0xAD, '\u{2191}'), (0xAE, '\u{2192}'), (0xAF, '\u{2193}'), (0xB0, '\u{00B0}'), (0xB1, '\u{00B1}'), (0xB2, '\u{2033}'),
|
||||
(0xB3, '\u{2265}'), (0xB4, '\u{00D7}'), (0xB5, '\u{221D}'), (0xB6, '\u{2202}'), (0xB7, '\u{2022}'), (0xB8, '\u{00F7}'),
|
||||
(0xB9, '\u{2260}'), (0xBA, '\u{2261}'), (0xBB, '\u{2248}'), (0xBC, '\u{2026}'), (0xBD, '\u{F8E6}'), (0xBE, '\u{F8E7}'),
|
||||
(0xBF, '\u{21B5}'), (0xC0, '\u{2135}'), (0xC1, '\u{2111}'), (0xC2, '\u{211C}'), (0xC3, '\u{2118}'), (0xC4, '\u{2297}'),
|
||||
(0xC5, '\u{2295}'), (0xC6, '\u{2205}'), (0xC7, '\u{2229}'), (0xC8, '\u{222A}'), (0xC9, '\u{2283}'), (0xCA, '\u{2287}'),
|
||||
(0xCB, '\u{2284}'), (0xCC, '\u{2282}'), (0xCD, '\u{2286}'), (0xCE, '\u{2208}'), (0xCF, '\u{2209}'), (0xD0, '\u{2220}'),
|
||||
(0xD1, '\u{2207}'), (0xD2, '\u{F6DA}'), (0xD3, '\u{F6D9}'), (0xD4, '\u{F6DB}'), (0xD5, '\u{220F}'), (0xD6, '\u{221A}'),
|
||||
(0xD7, '\u{22C5}'), (0xD8, '\u{00AC}'), (0xD9, '\u{2227}'), (0xDA, '\u{2228}'), (0xDB, '\u{21D4}'), (0xDC, '\u{21D0}'),
|
||||
(0xDD, '\u{21D1}'), (0xDE, '\u{21D2}'), (0xDF, '\u{21D3}'), (0xE0, '\u{25CA}'), (0xE1, '\u{2329}'), (0xE2, '\u{F8E8}'),
|
||||
(0xE3, '\u{F8E9}'), (0xE4, '\u{F8EA}'), (0xE5, '\u{2211}'), (0xE6, '\u{F8EB}'), (0xE7, '\u{F8EC}'), (0xE8, '\u{F8ED}'),
|
||||
(0xE9, '\u{F8EE}'), (0xEA, '\u{F8EF}'), (0xEB, '\u{F8F0}'), (0xEC, '\u{F8F1}'), (0xED, '\u{F8F2}'), (0xEE, '\u{F8F3}'),
|
||||
(0xEF, '\u{F8F4}'), (0xF1, '\u{232A}'), (0xF2, '\u{222B}'), (0xF3, '\u{2320}'), (0xF4, '\u{F8F5}'), (0xF5, '\u{2321}'),
|
||||
(0xF6, '\u{F8F6}'), (0xF7, '\u{F8F7}'), (0xF8, '\u{F8F8}'), (0xF9, '\u{F8F9}'), (0xFA, '\u{F8FA}'), (0xFB, '\u{F8FB}'),
|
||||
(0xFC, '\u{F8FC}'), (0xFD, '\u{F8FD}'), (0xFE, '\u{F8FE}'),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static ZAPFDINGBATS_ENCODING: &[(u8, char)] = &[
|
||||
(0x20, ' '), (0x21, '\u{2701}'), (0x22, '\u{2702}'), (0x23, '\u{2703}'), (0x24, '\u{2704}'), (0x25, '\u{260E}'),
|
||||
(0x26, '\u{2706}'), (0x27, '\u{2707}'), (0x28, '\u{2708}'), (0x29, '\u{2709}'), (0x2A, '\u{261B}'), (0x2B, '\u{261E}'),
|
||||
(0x2C, '\u{270C}'), (0x2D, '\u{270D}'), (0x2E, '\u{270E}'), (0x2F, '\u{270F}'), (0x30, '\u{2710}'), (0x31, '\u{2711}'),
|
||||
(0x32, '\u{2712}'), (0x33, '\u{2713}'), (0x34, '\u{2714}'), (0x35, '\u{2715}'), (0x36, '\u{2716}'), (0x37, '\u{2717}'),
|
||||
(0x38, '\u{2718}'), (0x39, '\u{2719}'), (0x3A, '\u{271A}'), (0x3B, '\u{271B}'), (0x3C, '\u{271C}'), (0x3D, '\u{271D}'),
|
||||
(0x3E, '\u{271E}'), (0x3F, '\u{271F}'), (0x40, '\u{2720}'), (0x41, '\u{2721}'), (0x42, '\u{2722}'), (0x43, '\u{2723}'),
|
||||
(0x44, '\u{2724}'), (0x45, '\u{2725}'), (0x46, '\u{2726}'), (0x47, '\u{2727}'), (0x48, '\u{2605}'), (0x49, '\u{2729}'),
|
||||
(0x4A, '\u{272A}'), (0x4B, '\u{272B}'), (0x4C, '\u{272C}'), (0x4D, '\u{272D}'), (0x4E, '\u{272E}'), (0x4F, '\u{272F}'),
|
||||
(0x50, '\u{2730}'), (0x51, '\u{2731}'), (0x52, '\u{2732}'), (0x53, '\u{2733}'), (0x54, '\u{2734}'), (0x55, '\u{2735}'),
|
||||
(0x56, '\u{2736}'), (0x57, '\u{2737}'), (0x58, '\u{2738}'), (0x59, '\u{2739}'), (0x5A, '\u{273A}'), (0x5B, '\u{273B}'),
|
||||
(0x5C, '\u{273C}'), (0x5D, '\u{273D}'), (0x5E, '\u{273E}'), (0x5F, '\u{273F}'), (0x60, '\u{2740}'), (0x61, '\u{2741}'),
|
||||
(0x62, '\u{2742}'), (0x63, '\u{2743}'), (0x64, '\u{2744}'), (0x65, '\u{2745}'), (0x66, '\u{2746}'), (0x67, '\u{2747}'),
|
||||
(0x68, '\u{2748}'), (0x69, '\u{2749}'), (0x6A, '\u{274A}'), (0x6B, '\u{274B}'), (0x6C, '\u{25CF}'), (0x6D, '\u{274D}'),
|
||||
(0x6E, '\u{25A0}'), (0x6F, '\u{274F}'), (0x70, '\u{2750}'), (0x71, '\u{2751}'), (0x72, '\u{2752}'), (0x73, '\u{25B2}'),
|
||||
(0x74, '\u{25BC}'), (0x75, '\u{25C6}'), (0x76, '\u{2756}'), (0x77, '\u{25D7}'), (0x78, '\u{2758}'), (0x79, '\u{2759}'),
|
||||
(0x7A, '\u{275A}'), (0x7B, '\u{275B}'), (0x7C, '\u{275C}'), (0x7D, '\u{275D}'), (0x7E, '\u{275E}'), (0x80, '\u{2768}'),
|
||||
(0x81, '\u{2769}'), (0x82, '\u{276A}'), (0x83, '\u{276B}'), (0x84, '\u{276C}'), (0x85, '\u{276D}'), (0x86, '\u{276E}'),
|
||||
(0x87, '\u{276F}'), (0x88, '\u{2770}'), (0x89, '\u{2771}'), (0x8A, '\u{2772}'), (0x8B, '\u{2773}'), (0x8C, '\u{2774}'),
|
||||
(0x8D, '\u{2775}'), (0xA1, '\u{2761}'), (0xA2, '\u{2762}'), (0xA3, '\u{2763}'), (0xA4, '\u{2764}'), (0xA5, '\u{2765}'),
|
||||
(0xA6, '\u{2766}'), (0xA7, '\u{2767}'), (0xA8, '\u{2663}'), (0xA9, '\u{2666}'), (0xAA, '\u{2665}'), (0xAB, '\u{2660}'),
|
||||
(0xAC, '\u{2460}'), (0xAD, '\u{2461}'), (0xAE, '\u{2462}'), (0xAF, '\u{2463}'), (0xB0, '\u{2464}'), (0xB1, '\u{2465}'),
|
||||
(0xB2, '\u{2466}'), (0xB3, '\u{2467}'), (0xB4, '\u{2468}'), (0xB5, '\u{2469}'), (0xB6, '\u{2776}'), (0xB7, '\u{2777}'),
|
||||
(0xB8, '\u{2778}'), (0xB9, '\u{2779}'), (0xBA, '\u{277A}'), (0xBB, '\u{277B}'), (0xBC, '\u{277C}'), (0xBD, '\u{277D}'),
|
||||
(0xBE, '\u{277E}'), (0xBF, '\u{277F}'), (0xC0, '\u{2780}'), (0xC1, '\u{2781}'), (0xC2, '\u{2782}'), (0xC3, '\u{2783}'),
|
||||
(0xC4, '\u{2784}'), (0xC5, '\u{2785}'), (0xC6, '\u{2786}'), (0xC7, '\u{2787}'), (0xC8, '\u{2788}'), (0xC9, '\u{2789}'),
|
||||
(0xCA, '\u{278A}'), (0xCB, '\u{278B}'), (0xCC, '\u{278C}'), (0xCD, '\u{278D}'), (0xCE, '\u{278E}'), (0xCF, '\u{278F}'),
|
||||
(0xD0, '\u{2790}'), (0xD1, '\u{2791}'), (0xD2, '\u{2792}'), (0xD3, '\u{2793}'), (0xD4, '\u{2794}'), (0xD5, '\u{2192}'),
|
||||
(0xD6, '\u{2194}'), (0xD7, '\u{2195}'), (0xD8, '\u{2798}'), (0xD9, '\u{2799}'), (0xDA, '\u{279A}'), (0xDB, '\u{279B}'),
|
||||
(0xDC, '\u{279C}'), (0xDD, '\u{279D}'), (0xDE, '\u{279E}'), (0xDF, '\u{279F}'), (0xE0, '\u{27A0}'), (0xE1, '\u{27A1}'),
|
||||
(0xE2, '\u{27A2}'), (0xE3, '\u{27A3}'), (0xE4, '\u{27A4}'), (0xE5, '\u{27A5}'), (0xE6, '\u{27A6}'), (0xE7, '\u{27A7}'),
|
||||
(0xE8, '\u{27A8}'), (0xE9, '\u{27A9}'), (0xEA, '\u{27AA}'), (0xEB, '\u{27AB}'), (0xEC, '\u{27AC}'), (0xED, '\u{27AD}'),
|
||||
(0xEE, '\u{27AE}'), (0xEF, '\u{27AF}'), (0xF1, '\u{27B1}'), (0xF2, '\u{27B2}'), (0xF3, '\u{27B3}'), (0xF4, '\u{27B4}'),
|
||||
(0xF5, '\u{27B5}'), (0xF6, '\u{27B6}'), (0xF7, '\u{27B7}'), (0xF8, '\u{27B8}'), (0xF9, '\u{27B9}'), (0xFA, '\u{27BA}'),
|
||||
(0xFB, '\u{27BB}'), (0xFC, '\u{27BC}'), (0xFD, '\u{27BD}'), (0xFE, '\u{27BE}'),
|
||||
];
|
||||
|
||||
/// Every width table, for exhaustive testing.
|
||||
#[cfg(test)]
|
||||
static ALL_TABLES: &[(&str, &[(char, u16)])] = &[
|
||||
("COURIER", COURIER),
|
||||
("COURIER_BOLD", COURIER_BOLD),
|
||||
("COURIER_OBLIQUE", COURIER_OBLIQUE),
|
||||
("COURIER_BOLDOBLIQUE", COURIER_BOLDOBLIQUE),
|
||||
("HELVETICA", HELVETICA),
|
||||
("HELVETICA_BOLD", HELVETICA_BOLD),
|
||||
("HELVETICA_OBLIQUE", HELVETICA_OBLIQUE),
|
||||
("HELVETICA_BOLDOBLIQUE", HELVETICA_BOLDOBLIQUE),
|
||||
("TIMES_ROMAN", TIMES_ROMAN),
|
||||
("TIMES_BOLD", TIMES_BOLD),
|
||||
("TIMES_ITALIC", TIMES_ITALIC),
|
||||
("TIMES_BOLDITALIC", TIMES_BOLDITALIC),
|
||||
("SYMBOL", SYMBOL),
|
||||
("ZAPFDINGBATS", ZAPFDINGBATS),
|
||||
];
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn times_roman_ascii_widths() {
|
||||
assert_eq!(base14_char_width("Times-Roman", ' '), Some(250));
|
||||
assert_eq!(base14_char_width("Times-Roman", 'M'), Some(889));
|
||||
assert_eq!(base14_char_width("Times-Roman", 'i'), Some(278));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subset_prefix_and_aliases_normalize() {
|
||||
assert_eq!(
|
||||
base14_char_width("ABCDEF+Times-Bold", ' '),
|
||||
base14_char_width("Times-Bold", ' ')
|
||||
);
|
||||
assert!(base14_char_width("ArialMT", 'a').is_some());
|
||||
assert!(base14_char_width("TimesNewRomanPSMT", 'a').is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_base14_returns_none() {
|
||||
assert_eq!(base14_char_width("DejaVuSans", 'a'), None);
|
||||
assert!(!is_base14_font("Garamond"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builtin_encoding_resolves_symbol_and_zapf_codes() {
|
||||
// Symbol 0x61 renders alpha; Zapf 0x21 renders U+2701.
|
||||
assert_eq!(builtin_encoding_char("Symbol", 0x61), Some('\u{03B1}'));
|
||||
assert_eq!(builtin_encoding_char("Symbol", 0xA5), Some('\u{221E}'));
|
||||
assert_eq!(
|
||||
builtin_encoding_char("ZapfDingbats", 0x21),
|
||||
Some('\u{2701}')
|
||||
);
|
||||
// Latin text fonts follow standard encodings — no builtin override.
|
||||
assert_eq!(builtin_encoding_char("Times-Roman", 0x61), None);
|
||||
// The resolved chars have real AFM widths.
|
||||
let alpha_w = base14_char_width("Symbol", '\u{03B1}');
|
||||
assert!(alpha_w.is_some() && alpha_w != Some(500));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encoding_tables_are_sorted_for_binary_search() {
|
||||
for table in [SYMBOL_ENCODING, ZAPFDINGBATS_ENCODING] {
|
||||
assert!(table.windows(2).all(|w| w[0].0 < w[1].0));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tables_are_sorted_for_binary_search() {
|
||||
// Every table is queried by binary search, so all of them must be
|
||||
// sorted — not just a sample.
|
||||
for (name, table) in ALL_TABLES {
|
||||
assert!(
|
||||
table.windows(2).all(|w| w[0].0 < w[1].0),
|
||||
"{name} is not sorted"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -14,9 +14,9 @@ use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, build_type3_scales, compute_string_width_ts,
|
||||
descriptor_style_flags, extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes,
|
||||
CMapDecisionCache, FontStyleCache,
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, descriptor_style_flags,
|
||||
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
FontStyleCache,
|
||||
};
|
||||
use super::underline::UnderlineLine;
|
||||
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
|
||||
@@ -41,17 +41,6 @@ fn strip_pdf_comments(data: &[u8]) -> Vec<u8> {
|
||||
while i < data.len() {
|
||||
let b = data[i];
|
||||
match b {
|
||||
// Inside a string literal, a backslash escapes the next byte —
|
||||
// `\(`, `\)`, and `\\` must not touch the nesting depth, or a
|
||||
// later `%` glyph inside a string gets stripped as a comment,
|
||||
// corrupting the stream.
|
||||
b'\\' if in_string > 0 => {
|
||||
result.push(b);
|
||||
if let Some(&next) = data.get(i + 1) {
|
||||
result.push(next);
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
b'(' if !in_hex_string => {
|
||||
in_string += 1;
|
||||
result.push(b);
|
||||
@@ -173,11 +162,10 @@ pub(crate) fn extract_page_text_items(
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
|
||||
// Build font encoding maps from Differences arrays
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
|
||||
|
||||
// Build font width info for accurate text positioning
|
||||
let font_widths = build_font_widths(doc, &fonts);
|
||||
let type3_scales = build_type3_scales(doc, &fonts);
|
||||
|
||||
// Build maps of font resource names to their base font names and ToUnicode object refs
|
||||
let mut font_base_names: std::collections::HashMap<String, String> =
|
||||
@@ -514,8 +502,7 @@ pub(crate) fn extract_page_text_items(
|
||||
) {
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
@@ -686,8 +673,7 @@ pub(crate) fn extract_page_text_items(
|
||||
} else {
|
||||
rotation_votes.rotated += 1;
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let base_font = font_base_names
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
@@ -794,8 +780,7 @@ pub(crate) fn extract_page_text_items(
|
||||
} else {
|
||||
rotation_votes.rotated += 1;
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = w_ts_opt
|
||||
.map(|w_ts| {
|
||||
@@ -947,8 +932,7 @@ pub(crate) fn extract_page_text_items(
|
||||
} else {
|
||||
rotation_votes.rotated += 1;
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
// Width in device space from text matrix delta
|
||||
let delta_ts = text_matrix[4] - start_tm[4];
|
||||
@@ -1865,26 +1849,4 @@ BT 30 700 Tm <41> Tj ET";
|
||||
"ET should be preserved after comment stripping"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_pdf_comments_escaped_parens() {
|
||||
// An escaped `\)` must not close the string: the `%` after it is
|
||||
// still string content, not a comment (subset fonts routinely map
|
||||
// glyphs to `%` and to escaped parens in the same TJ array).
|
||||
let input = b"[ (a\\)b) 1 (%) 1 (c) ] TJ\n";
|
||||
let output = strip_pdf_comments(input);
|
||||
assert_eq!(output, input.to_vec());
|
||||
|
||||
// Same for an escaped `\(` — must not open a phantom string that
|
||||
// shields a real comment.
|
||||
let input = b"(x\\(y) Tj % real comment\nET\n";
|
||||
let output = strip_pdf_comments(input);
|
||||
assert_eq!(output, b"(x\\(y) Tj \nET\n");
|
||||
|
||||
// Escaped backslash before a real close-paren: `\\` ends the escape,
|
||||
// the `)` does close the string, and the comment is stripped.
|
||||
let input = b"(x\\\\) Tj % comment\nET\n";
|
||||
let output = strip_pdf_comments(input);
|
||||
assert_eq!(output, b"(x\\\\) Tj \nET\n");
|
||||
}
|
||||
}
|
||||
|
||||
+13
-388
@@ -138,93 +138,6 @@ pub(crate) fn build_font_widths(
|
||||
widths
|
||||
}
|
||||
|
||||
/// Visual-size scale factors for Type3 fonts, keyed by resource name.
|
||||
///
|
||||
/// A Type3 font's glyph space maps to text space through FontMatrix, so the
|
||||
/// visual height of its glyphs is `nominal_size × |matrix_y| × FontBBox
|
||||
/// height`. For a well-behaved font (matrix 0.001, bbox ≈ 1000 units) that
|
||||
/// factor is ≈ 1.0 and the nominal size is already right. TeX PK bitmap
|
||||
/// fonts (dvips → Distiller) instead use FontMatrix [1 0 0 -1 0 0] with
|
||||
/// nominal sizes like 0.12, which makes every downstream font-size heuristic
|
||||
/// (drop caps, sub/superscripts, small-font tables, line heights) see
|
||||
/// nonsense. Fonts without a usable FontBBox are omitted (treated as 1.0).
|
||||
pub(crate) fn build_type3_scales(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
) -> HashMap<String, f32> {
|
||||
let mut scales = HashMap::new();
|
||||
for (font_name, font_dict) in fonts {
|
||||
let is_type3 = font_dict
|
||||
.get(b"Subtype")
|
||||
.ok()
|
||||
.and_then(|o| o.as_name().ok())
|
||||
.is_some_and(|n| n == b"Type3");
|
||||
if !is_type3 {
|
||||
continue;
|
||||
}
|
||||
// Array elements may themselves be indirect references per PDF
|
||||
// syntax — resolve before reading the numeric value.
|
||||
let num = |o: &Object| {
|
||||
let resolved = match o {
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(inner) => inner,
|
||||
Err(_) => return 0.0,
|
||||
},
|
||||
other => other,
|
||||
};
|
||||
match resolved {
|
||||
Object::Integer(i) => *i as f32,
|
||||
Object::Real(r) => *r,
|
||||
_ => 0.0,
|
||||
}
|
||||
};
|
||||
let Some(matrix) = font_dict
|
||||
.get(b"FontMatrix")
|
||||
.ok()
|
||||
.and_then(|o| resolve_array(doc, o))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let Some(bbox) = font_dict
|
||||
.get(b"FontBBox")
|
||||
.ok()
|
||||
.and_then(|o| resolve_array(doc, o))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
if matrix.len() < 4 || bbox.len() < 4 {
|
||||
continue;
|
||||
}
|
||||
let scale_y = (num(&matrix[2]).powi(2) + num(&matrix[3]).powi(2)).sqrt();
|
||||
let bbox_h = (num(&bbox[3]) - num(&bbox[1])).abs();
|
||||
let scale = bbox_h * scale_y;
|
||||
|
||||
// `scale` is the glyph box measured in text-space units. For a
|
||||
// self-consistent font it lands near 1.0 — the FontMatrix is the
|
||||
// reciprocal of the glyph-space em by construction — so the Tf
|
||||
// operand is already the rendered size and must be left alone.
|
||||
// A modest deviation is normal and must NOT trigger rescaling:
|
||||
// FontBBox is the glyph bounding box, not the em box, so it is
|
||||
// routinely somewhat smaller (descender..ascender ≈ 0.7) or larger
|
||||
// (tall accents > 1.0).
|
||||
//
|
||||
// Only a wildly inconsistent font gets renormalized. dvips/PK
|
||||
// bitmap fonts declare [1 0 0 -1 0 0] with glyphs spanning
|
||||
// hundreds of units, giving scale ≈ 159 against a nominal size of
|
||||
// 0.12pt — there the declared size carries no information. The
|
||||
// band is deliberately wide so that only that class qualifies,
|
||||
// while any matrix scale (including non-standard ones like 0.005
|
||||
// with a full-em bbox, scale = 5.0) is judged on the product
|
||||
// rather than on the matrix alone.
|
||||
const CONSISTENT_LO: f32 = 0.25;
|
||||
const CONSISTENT_HI: f32 = 4.0;
|
||||
if scale.is_finite() && scale > 0.0 && !(CONSISTENT_LO..=CONSISTENT_HI).contains(&scale) {
|
||||
scales.insert(String::from_utf8_lossy(font_name).to_string(), scale);
|
||||
}
|
||||
}
|
||||
scales
|
||||
}
|
||||
|
||||
/// Parse font widths from a font dictionary, dispatching by Subtype
|
||||
pub(crate) fn parse_font_widths(
|
||||
doc: &Document,
|
||||
@@ -236,71 +149,11 @@ pub(crate) fn parse_font_widths(
|
||||
|
||||
match subtype_name {
|
||||
b"Type0" => parse_type0_widths(doc, font_dict),
|
||||
b"Type1" | b"TrueType" | b"MMType1" => parse_simple_font_widths(doc, font_dict)
|
||||
.or_else(|| base14_fallback_widths(doc, font_dict)),
|
||||
b"Type3" => parse_simple_font_widths(doc, font_dict),
|
||||
b"Type1" | b"TrueType" | b"MMType1" | b"Type3" => parse_simple_font_widths(doc, font_dict),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Fallback metrics for non-embedded base-14 fonts whose dictionary omits
|
||||
/// `/FirstChar`/`/Widths` (legal per the PDF spec — the reader must supply
|
||||
/// standard-font metrics). Without this, every glyph advances 0 and all
|
||||
/// downstream gap-based logic (space synthesis, script detection, table
|
||||
/// columns) collapses — common in 1990s dvips/Distiller PDFs.
|
||||
///
|
||||
/// Widths are resolved per code through the font's Differences encoding when
|
||||
/// present, falling back to the same single-byte decode the text extractor
|
||||
/// uses (cp1252-style smart punctuation for 0x80..=0x9F, Latin-1 elsewhere) —
|
||||
/// so the width of a code always matches the char we extract for it.
|
||||
fn base14_fallback_widths(doc: &Document, font_dict: &lopdf::Dictionary) -> Option<FontWidthInfo> {
|
||||
let base_font = font_dict
|
||||
.get(b"BaseFont")
|
||||
.ok()
|
||||
.and_then(|o| o.as_name().ok())
|
||||
.map(|n| String::from_utf8_lossy(n).to_string())?;
|
||||
if !crate::extractor::base14::is_base14_font(&base_font) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let enc_map = parse_font_encoding(doc, font_dict)
|
||||
.map(|r| r.map)
|
||||
.unwrap_or_default();
|
||||
|
||||
let mut widths = HashMap::new();
|
||||
for code in 0u16..=255 {
|
||||
// Resolution order: Differences override, then the font's BUILT-IN
|
||||
// encoding (Symbol/ZapfDingbats glyphs live at positions unrelated
|
||||
// to cp1252 — the renderer draws α for Symbol 0x61 no matter how
|
||||
// the text decoder transliterates it, so the advance must be α's),
|
||||
// then the cp1252-style fallback used by the text decoder.
|
||||
let ch = enc_map
|
||||
.get(&(code as u8))
|
||||
.copied()
|
||||
.or_else(|| crate::extractor::base14::builtin_encoding_char(&base_font, code as u8))
|
||||
.unwrap_or_else(|| decode_single_byte_fallback_char(code as u8, true));
|
||||
if let Some(w) = crate::extractor::base14::base14_char_width(&base_font, ch) {
|
||||
widths.insert(code, w);
|
||||
}
|
||||
}
|
||||
let space_width = widths.get(&32).copied().unwrap_or(250);
|
||||
|
||||
debug!(
|
||||
" base14 fallback widths for {} ({} codes mapped)",
|
||||
base_font,
|
||||
widths.len()
|
||||
);
|
||||
|
||||
Some(FontWidthInfo {
|
||||
widths,
|
||||
default_width: 500,
|
||||
space_width,
|
||||
is_cid: false,
|
||||
units_scale: 0.001,
|
||||
wmode: 0,
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse widths for simple fonts (Type1, TrueType, MMType1, Type3)
|
||||
/// Reads FirstChar, LastChar, and Widths array.
|
||||
/// For Type3 fonts, reads FontMatrix to determine the correct units_scale.
|
||||
@@ -644,13 +497,9 @@ pub(crate) fn get_operand_bytes(obj: &Object) -> Option<&[u8]> {
|
||||
/// Build encoding maps for all fonts on a page.
|
||||
/// Returns `(encodings, has_gid_fonts)` where `has_gid_fonts` is true when
|
||||
/// any font uses raw glyph ID names (gidNNNNN) that can't be decoded.
|
||||
/// Gid names whose codes the font's own ToUnicode CMap maps are decodable
|
||||
/// and do not set the flag (LibreOffice subsets write /gidNNNN Differences
|
||||
/// names alongside a complete ToUnicode CMap).
|
||||
pub(crate) fn build_font_encodings(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
cmaps: &FontCMaps,
|
||||
) -> (PageFontEncodings, bool) {
|
||||
let mut encodings = PageFontEncodings::new();
|
||||
let mut has_gid_fonts = false;
|
||||
@@ -659,9 +508,7 @@ pub(crate) fn build_font_encodings(
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
|
||||
if let Some(result) = parse_font_encoding(doc, font_dict) {
|
||||
if !result.gid_codes.is_empty()
|
||||
&& !tounicode_maps_codes(font_dict, cmaps, &result.gid_codes)
|
||||
{
|
||||
if result.gid_glyph_count > 0 {
|
||||
has_gid_fonts = true;
|
||||
}
|
||||
if !result.map.is_empty() {
|
||||
@@ -673,34 +520,6 @@ pub(crate) fn build_font_encodings(
|
||||
(encodings, has_gid_fonts)
|
||||
}
|
||||
|
||||
/// True when the font's ToUnicode CMap maps the gid-named character codes,
|
||||
/// so the Differences entries still decode through the CMap.
|
||||
fn tounicode_maps_codes(font_dict: &lopdf::Dictionary, cmaps: &FontCMaps, codes: &[u8]) -> bool {
|
||||
let Some(obj_ref) = font_dict
|
||||
.get(b"ToUnicode")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
else {
|
||||
return false;
|
||||
};
|
||||
let Some(entry) = cmaps.get_by_obj(obj_ref.0) else {
|
||||
return false;
|
||||
};
|
||||
// At least one gid code usably mapped means the CMap addresses these
|
||||
// codes; remaining unmapped codes are subset leftovers (e.g. the
|
||||
// component glyphs of an emoji ZWJ sequence mapped whole on its first
|
||||
// code). A mapping is usable only when extraction would accept it —
|
||||
// empty or U+FFFD results are rejected there as invalid. Fonts whose
|
||||
// CMap ignores the gid codes entirely stay flagged, and the downstream
|
||||
// garbage/encoding checks still catch partial damage.
|
||||
codes.iter().any(|&code| {
|
||||
entry
|
||||
.primary
|
||||
.lookup(code as u16)
|
||||
.is_some_and(|s| !s.is_empty() && !s.contains('\u{FFFD}'))
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse font encoding from a font dictionary
|
||||
pub(crate) fn parse_font_encoding(
|
||||
doc: &Document,
|
||||
@@ -739,10 +558,11 @@ pub(crate) fn parse_font_encoding(
|
||||
/// Result of parsing an encoding dictionary's Differences array.
|
||||
pub(crate) struct EncodingResult {
|
||||
pub map: FontEncodingMap,
|
||||
/// Character codes whose glyph names match the `gidNNNNN` pattern (raw
|
||||
/// glyph IDs). These reference the original font's glyph table and are
|
||||
/// only decodable when the font's ToUnicode CMap maps the code.
|
||||
pub gid_codes: Vec<u8>,
|
||||
/// Number of glyph names matching the `gidNNNNN` pattern (raw glyph IDs).
|
||||
/// These indicate a font with unresolvable encoding — the glyph IDs
|
||||
/// reference the original font's glyph table, but without the original
|
||||
/// font's cmap there is no way to map them to Unicode.
|
||||
pub gid_glyph_count: u32,
|
||||
}
|
||||
|
||||
/// Parse an encoding dictionary with Differences array
|
||||
@@ -768,7 +588,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
let mut encoding_map = FontEncodingMap::new();
|
||||
let mut current_code: u8 = 0;
|
||||
let mut ligature_count = 0u32;
|
||||
let mut gid_codes: Vec<u8> = Vec::new();
|
||||
let mut gid_glyph_count = 0u32;
|
||||
|
||||
for item in diff_array {
|
||||
match item {
|
||||
@@ -794,7 +614,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
&& glyph_name.len() >= 4
|
||||
&& glyph_name[3..].chars().all(|c| c.is_ascii_digit())
|
||||
{
|
||||
gid_codes.push(current_code);
|
||||
gid_glyph_count += 1;
|
||||
}
|
||||
if let Some(ch) = mapped_char {
|
||||
encoding_map.insert(current_code, ch);
|
||||
@@ -818,16 +638,16 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
);
|
||||
}
|
||||
|
||||
if !gid_codes.is_empty() {
|
||||
if gid_glyph_count > 0 {
|
||||
debug!(
|
||||
" Differences: {} gid-encoded glyphs (decodable only via ToUnicode)",
|
||||
gid_codes.len()
|
||||
" Differences: {} gid-encoded glyphs (unresolvable without original font)",
|
||||
gid_glyph_count
|
||||
);
|
||||
}
|
||||
|
||||
Some(EncodingResult {
|
||||
map: encoding_map,
|
||||
gid_codes,
|
||||
gid_glyph_count,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1617,92 +1437,6 @@ fn score_text(text: &str) -> i32 {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn type3_scale_resolves_indirect_matrix_and_bbox_numbers() {
|
||||
use lopdf::{dictionary, Document, Object};
|
||||
// FontMatrix/FontBBox elements may be indirect references per PDF
|
||||
// syntax; the scale must use their resolved values, not zero.
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let matrix_d = doc.add_object(Object::Real(-1.0));
|
||||
let bbox_top = doc.add_object(Object::Integer(3));
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type3",
|
||||
"FontMatrix" => vec![
|
||||
Object::Integer(1),
|
||||
Object::Integer(0),
|
||||
Object::Integer(0),
|
||||
Object::Reference(matrix_d),
|
||||
Object::Integer(0),
|
||||
Object::Integer(0),
|
||||
],
|
||||
"FontBBox" => vec![
|
||||
Object::Integer(1),
|
||||
Object::Integer(-156),
|
||||
Object::Integer(37),
|
||||
Object::Reference(bbox_top),
|
||||
],
|
||||
};
|
||||
let mut fonts = std::collections::BTreeMap::new();
|
||||
fonts.insert(b"T2".to_vec(), &font_dict);
|
||||
let scales = super::build_type3_scales(&doc, &fonts);
|
||||
let scale = scales.get("T2").copied().unwrap_or(1.0);
|
||||
// bbox height 159 x |matrix_y| 1.0
|
||||
assert!(
|
||||
(scale - 159.0).abs() < 0.5,
|
||||
"scale should use resolved indirect values, got {scale}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Build a one-font Type3 document and return its computed scale, if any.
|
||||
#[cfg(test)]
|
||||
fn type3_scale_for(matrix_y: f32, bbox_lo: i64, bbox_hi: i64) -> Option<f32> {
|
||||
use lopdf::{dictionary, Document, Object};
|
||||
let doc = Document::with_version("1.4");
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type3",
|
||||
"FontMatrix" => vec![
|
||||
Object::Real(matrix_y), Object::Integer(0), Object::Integer(0),
|
||||
Object::Real(matrix_y), Object::Integer(0), Object::Integer(0),
|
||||
],
|
||||
"FontBBox" => vec![
|
||||
Object::Integer(0), Object::Integer(bbox_lo),
|
||||
Object::Integer(600), Object::Integer(bbox_hi),
|
||||
],
|
||||
};
|
||||
let mut fonts = std::collections::BTreeMap::new();
|
||||
fonts.insert(b"T9".to_vec(), &font_dict);
|
||||
super::build_type3_scales(&doc, &fonts).get("T9").copied()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_skips_self_consistent_fonts() {
|
||||
// Conventional 1/1000 matrix with a descender..ascender bbox of 700
|
||||
// units: scale 0.7. The Tf operand is already the rendered size, so
|
||||
// renormalizing would report every size at 0.7x.
|
||||
assert_eq!(type3_scale_for(0.001, -200, 500), None);
|
||||
// Tall-accent bbox slightly over the em (1100 units, scale 1.1).
|
||||
assert_eq!(type3_scale_for(0.001, -100, 1000), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_applies_to_inconsistent_fonts_at_any_matrix_scale() {
|
||||
// Non-standard but valid matrix (0.005) with a full-em bbox:
|
||||
// scale 5.0, so the declared size is off by 5x and must be fixed.
|
||||
let s = type3_scale_for(0.005, 0, 1000).expect("0.005 matrix should rescale");
|
||||
assert!((s - 5.0).abs() < 0.01, "got {s}");
|
||||
// dvips/PK bitmap pattern: unit matrix, glyphs spanning ~159 units.
|
||||
let s = type3_scale_for(1.0, -156, 3).expect("PK pattern should rescale");
|
||||
assert!((s - 159.0).abs() < 0.5, "got {s}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_ignores_degenerate_bbox() {
|
||||
// [0 0 0 0] is legal and carries no size information.
|
||||
assert_eq!(type3_scale_for(0.001, 0, 0), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn texcm_math_symbols_remap() {
|
||||
assert_eq!(
|
||||
@@ -2204,113 +1938,4 @@ mod tests {
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
fn gid_font_doc(bfchar: Option<&str>) -> (Document, lopdf::ObjectId) {
|
||||
use lopdf::Stream;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let cmap = format!(
|
||||
"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
{}
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end",
|
||||
bfchar.unwrap_or_default()
|
||||
);
|
||||
let tounicode_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
cmap.into_bytes(),
|
||||
)));
|
||||
let enc_id = doc.add_object(dictionary! {
|
||||
"Type" => "Encoding",
|
||||
"Differences" => vec![
|
||||
1.into(),
|
||||
Object::Name(b"gid1283".to_vec()),
|
||||
Object::Name(b"gid1464".to_vec()),
|
||||
],
|
||||
});
|
||||
let mut font = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "ABCDEF+OpenSymbol",
|
||||
"Encoding" => Object::Reference(enc_id),
|
||||
};
|
||||
if bfchar.is_some() {
|
||||
font.set("ToUnicode", Object::Reference(tounicode_id));
|
||||
}
|
||||
let font_id = doc.add_object(font);
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn gid_flagged(bfchar: Option<&str>) -> bool {
|
||||
let (doc, page_id) = gid_font_doc(bfchar);
|
||||
let cmaps = FontCMaps::from_doc(&doc);
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap();
|
||||
let (_, has_gid_fonts) = build_font_encodings(&doc, &fonts, &cmaps);
|
||||
has_gid_fonts
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_covering_tounicode_are_not_flagged() {
|
||||
// LibreOffice subsets write /gidNNNN Differences names alongside a
|
||||
// ToUnicode CMap that decodes those codes; the page must not be
|
||||
// flagged as unresolvable (which would suppress the whole document's
|
||||
// markdown when every page carries such a font).
|
||||
assert!(!gid_flagged(Some("<01> <2022>\n<02> <25E6>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_partial_tounicode_are_not_flagged() {
|
||||
// An emoji ZWJ sequence maps whole on its first code; the remaining
|
||||
// component-glyph codes are subset leftovers, not damage.
|
||||
assert!(!gid_flagged(Some(
|
||||
"<01> <D83DDC68200DD83DDC69200DD83DDC67>"
|
||||
)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_without_tounicode_are_flagged() {
|
||||
assert!(
|
||||
gid_flagged(None),
|
||||
"gid glyphs without ToUnicode are unresolvable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_disjoint_tounicode_are_flagged() {
|
||||
// A ToUnicode that never addresses the gid codes leaves them
|
||||
// unresolvable.
|
||||
assert!(gid_flagged(Some("<10> <0041>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_replacement_char_tounicode_are_flagged() {
|
||||
// A mapping to U+FFFD is not usable — extraction rejects it as an
|
||||
// invalid CMap result — so it must not clear the gid flag.
|
||||
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
|
||||
}
|
||||
}
|
||||
|
||||
+32
-1183
File diff suppressed because it is too large
Load Diff
+7
-401
@@ -2,51 +2,11 @@
|
||||
|
||||
use crate::types::{ItemType, TextItem};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{resolve_array, resolve_dict};
|
||||
use super::get_number;
|
||||
|
||||
/// Upper bound on the number of form-field nodes visited during a single
|
||||
/// `extract_form_fields` pass. A crafted PDF can chain thousands of distinct
|
||||
/// `/Kids` fields to blow the stack even without an outright reference cycle,
|
||||
/// so we cap total traversal work in addition to detecting cycles.
|
||||
const MAX_FORM_FIELD_NODES: usize = 100_000;
|
||||
|
||||
/// Upper bound on `/Kids` recursion depth. Real AcroForm hierarchies are only
|
||||
/// a few levels deep (fields → child fields → widgets); a crafted PDF can chain
|
||||
/// tens of thousands of distinct fields into a linear `/Kids` list that would
|
||||
/// overflow the stack via depth-first recursion long before the node budget is
|
||||
/// reached. This depth cap bounds the stack independently of total node count.
|
||||
const MAX_FORM_FIELD_DEPTH: usize = 100;
|
||||
|
||||
/// Traversal budget for the AcroForm field walk. Bounds both the number of
|
||||
/// distinct nodes visited *and* the total number of `/Fields`/`/Kids` entries
|
||||
/// examined.
|
||||
///
|
||||
/// Counting `visited` alone is not enough: invalid entries (non-references) and
|
||||
/// duplicate references never grow `visited`, so an oversized array full of them
|
||||
/// would iterate to completion no matter how large. Charging every examined
|
||||
/// entry against the same budget makes it a real cap on traversal work.
|
||||
pub(crate) struct FieldWalkBudget {
|
||||
visited: HashSet<ObjectId>,
|
||||
examined: usize,
|
||||
}
|
||||
|
||||
impl FieldWalkBudget {
|
||||
fn new() -> Self {
|
||||
Self {
|
||||
visited: HashSet::new(),
|
||||
examined: 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// True once the budget is spent; callers must stop iterating and recursing.
|
||||
fn exhausted(&self) -> bool {
|
||||
self.visited.len() >= MAX_FORM_FIELD_NODES || self.examined >= MAX_FORM_FIELD_NODES
|
||||
}
|
||||
}
|
||||
|
||||
pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> Vec<TextItem> {
|
||||
let mut links = Vec::new();
|
||||
|
||||
@@ -186,104 +146,32 @@ pub(crate) fn extract_form_fields(
|
||||
Err(_) => return items,
|
||||
};
|
||||
|
||||
// Borrow the array rather than cloning it: a crafted `/Fields` can be huge,
|
||||
// and cloning would pay an O(n) allocation/copy before the budget check
|
||||
// below can stop the work.
|
||||
let fields = match acroform.get(b"Fields") {
|
||||
Ok(obj) => match resolve_array(doc, obj) {
|
||||
Some(arr) => arr,
|
||||
Some(arr) => arr.clone(),
|
||||
None => return items,
|
||||
},
|
||||
Err(_) => return items,
|
||||
};
|
||||
if fields.is_empty() {
|
||||
return items;
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
// Bound the walk so a crafted PDF cannot send us into unbounded recursion
|
||||
// via a `/Kids` cycle, a deep chain, or an oversized array of invalid or
|
||||
// duplicate entries.
|
||||
let mut budget = FieldWalkBudget::new();
|
||||
|
||||
for field_obj in fields {
|
||||
// Stop once the budget is spent so a `/Fields` array wider than the
|
||||
// budget can't burn CPU iterating entries whose walk would no-op. Charge
|
||||
// every entry (including invalid ones) against the budget.
|
||||
if budget.exhausted() {
|
||||
break;
|
||||
}
|
||||
budget.examined += 1;
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(
|
||||
doc,
|
||||
field_ref,
|
||||
None,
|
||||
"",
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
&mut budget,
|
||||
0,
|
||||
);
|
||||
walk_form_fields(doc, field_ref, None, "", page_map, &mut items);
|
||||
}
|
||||
}
|
||||
|
||||
items
|
||||
}
|
||||
|
||||
/// Map widget annotation objects back to the page whose `/Annots` array owns
|
||||
/// them. Some valid widgets omit `/P`, so the page tree is the only reliable
|
||||
/// ownership signal available for page-filtered extraction.
|
||||
fn annotation_page_map(
|
||||
doc: &Document,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
) -> HashMap<ObjectId, u32> {
|
||||
let mut annotation_pages = HashMap::new();
|
||||
for (&page_id, &page_num) in page_map {
|
||||
let Some(annotations) = doc
|
||||
.get_dictionary(page_id)
|
||||
.ok()
|
||||
.and_then(|page| page.get(b"Annots").ok())
|
||||
.and_then(|annotations| resolve_array(doc, annotations))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
for annotation in annotations {
|
||||
if let Ok(annotation_id) = annotation.as_reference() {
|
||||
annotation_pages.insert(annotation_id, page_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
annotation_pages
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
field_id: ObjectId,
|
||||
parent_ft: Option<&[u8]>,
|
||||
parent_name: &str,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
budget: &mut FieldWalkBudget,
|
||||
depth: usize,
|
||||
) {
|
||||
// Guard against `/Kids` cycles and pathologically large field trees.
|
||||
// Exceeding the depth cap means the chain is too deep to be a legitimate
|
||||
// form (and would overflow the stack); an exhausted budget means the tree is
|
||||
// too large. Both checks run *before* inserting so the visited set can never
|
||||
// grow past the budget.
|
||||
if depth > MAX_FORM_FIELD_DEPTH || budget.exhausted() {
|
||||
return;
|
||||
}
|
||||
// Revisiting an object ID means we hit a `/Kids` cycle.
|
||||
if !budget.visited.insert(field_id) {
|
||||
return;
|
||||
}
|
||||
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
Ok(d) => d,
|
||||
Err(_) => return,
|
||||
@@ -314,31 +202,11 @@ pub(crate) fn walk_form_fields(
|
||||
|
||||
// Check for /Kids — if present, recurse into children
|
||||
if let Ok(kids_obj) = field_dict.get(b"Kids") {
|
||||
// Iterate the borrowed array directly — cloning a crafted, oversized
|
||||
// `/Kids` would allocate and copy every entry before the budget check
|
||||
// below could stop the work.
|
||||
if let Some(kids) = resolve_array(doc, kids_obj) {
|
||||
for kid in kids {
|
||||
// Stop once the budget is spent so a `/Kids` array wider than the
|
||||
// budget can't burn CPU iterating entries whose walk would no-op.
|
||||
// Charge every entry (including invalid/duplicate ones) against
|
||||
// the budget so this is a true traversal-work cap.
|
||||
if budget.exhausted() {
|
||||
break;
|
||||
}
|
||||
budget.examined += 1;
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(
|
||||
doc,
|
||||
kid_ref,
|
||||
ft,
|
||||
&full_name,
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
budget,
|
||||
depth + 1,
|
||||
);
|
||||
walk_form_fields(doc, kid_ref, ft, &full_name, page_map, items);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -431,7 +299,6 @@ pub(crate) fn walk_form_fields(
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.and_then(|p| page_map.get(&p).copied())
|
||||
.or_else(|| annotation_pages.get(&field_id).copied())
|
||||
.unwrap_or(1);
|
||||
|
||||
let text = if full_name.is_empty() {
|
||||
@@ -457,264 +324,3 @@ pub(crate) fn walk_form_fields(
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::{dictionary, Object};
|
||||
|
||||
#[test]
|
||||
fn widget_without_page_reference_uses_owning_page_annotation() {
|
||||
let mut doc = Document::new();
|
||||
let widget_id = doc.add_object(dictionary! {
|
||||
"Type" => "Annot",
|
||||
"Subtype" => "Widget",
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("customer"),
|
||||
"V" => Object::string_literal("Alice"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
});
|
||||
let page_one_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
});
|
||||
let page_two_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Annots" => vec![Object::Reference(widget_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(widget_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::from([(page_one_id, 1), (page_two_id, 2)]);
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn kids_self_cycle_does_not_overflow_stack() {
|
||||
// A crafted AcroForm field that lists itself in `/Kids` must not send
|
||||
// the traversal into unbounded recursion.
|
||||
let mut doc = Document::new();
|
||||
let field_id = doc.new_object_id();
|
||||
doc.set_object(
|
||||
field_id,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("loop"),
|
||||
"Kids" => vec![Object::Reference(field_id)],
|
||||
},
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(field_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
// Completes (rather than overflowing the stack) and yields no items.
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn kids_mutual_cycle_terminates() {
|
||||
// Two fields that reference each other via `/Kids` form a cycle that
|
||||
// must also terminate.
|
||||
let mut doc = Document::new();
|
||||
let field_a = doc.new_object_id();
|
||||
let field_b = doc.new_object_id();
|
||||
doc.set_object(
|
||||
field_a,
|
||||
dictionary! {
|
||||
"T" => Object::string_literal("a"),
|
||||
"Kids" => vec![Object::Reference(field_b)],
|
||||
},
|
||||
);
|
||||
doc.set_object(
|
||||
field_b,
|
||||
dictionary! {
|
||||
"T" => Object::string_literal("b"),
|
||||
"Kids" => vec![Object::Reference(field_a)],
|
||||
},
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(field_a)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn deep_acyclic_kids_chain_does_not_overflow_stack() {
|
||||
// A long chain of *distinct* fields (no cycle) must also terminate:
|
||||
// the visited set alone would still recurse to the chain length, so
|
||||
// the depth cap is what prevents a stack overflow here.
|
||||
let mut doc = Document::new();
|
||||
let n = MAX_FORM_FIELD_DEPTH * 500;
|
||||
let ids: Vec<ObjectId> = (0..=n).map(|_| doc.new_object_id()).collect();
|
||||
for i in 0..n {
|
||||
doc.set_object(
|
||||
ids[i],
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"Kids" => vec![Object::Reference(ids[i + 1])],
|
||||
},
|
||||
);
|
||||
}
|
||||
// Leaf carries a value; it sits far below the depth cap so it is never
|
||||
// reached, proving traversal stops early rather than crashing.
|
||||
doc.set_object(
|
||||
ids[n],
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("leaf"),
|
||||
"V" => Object::string_literal("x"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(ids[0])],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_tree_traversal_stops_at_node_budget() {
|
||||
// A single field with a `/Kids` array wider than the node budget must
|
||||
// stop traversal at the cap rather than growing `visited` (and the work)
|
||||
// without bound. Each processed leaf emits one item, so the item count
|
||||
// is bounded by the budget and reaches right up to it (a couple of
|
||||
// slots go to the root and the boundary node charged against the cap).
|
||||
let mut doc = Document::new();
|
||||
let fanout = MAX_FORM_FIELD_NODES + 50;
|
||||
let leaf_ids: Vec<ObjectId> = (0..fanout).map(|_| doc.new_object_id()).collect();
|
||||
for &leaf in &leaf_ids {
|
||||
doc.set_object(
|
||||
leaf,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"V" => Object::string_literal("v"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
}
|
||||
let kids: Vec<Object> = leaf_ids.iter().map(|&id| Object::Reference(id)).collect();
|
||||
let root_id = doc.add_object(dictionary! {
|
||||
"T" => Object::string_literal("root"),
|
||||
"Kids" => kids,
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(root_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
// Extraction stops at the budget: bounded above by the cap, and it gets
|
||||
// right up to it (allowing a small delta for the root/boundary nodes
|
||||
// charged against the budget).
|
||||
assert!(items.len() <= MAX_FORM_FIELD_NODES);
|
||||
assert!(items.len() >= MAX_FORM_FIELD_NODES - 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_top_level_fields_stop_at_node_budget() {
|
||||
// A top-level `/Fields` array wider than the budget must also stop at
|
||||
// the cap: the item count is bounded by the budget and reaches right up
|
||||
// to it.
|
||||
let mut doc = Document::new();
|
||||
let fanout = MAX_FORM_FIELD_NODES + 50;
|
||||
let leaf_ids: Vec<ObjectId> = (0..fanout).map(|_| doc.new_object_id()).collect();
|
||||
for &leaf in &leaf_ids {
|
||||
doc.set_object(
|
||||
leaf,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"V" => Object::string_literal("v"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
}
|
||||
let fields: Vec<Object> = leaf_ids.iter().map(|&id| Object::Reference(id)).collect();
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => fields,
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert!(items.len() <= MAX_FORM_FIELD_NODES);
|
||||
assert!(items.len() >= MAX_FORM_FIELD_NODES - 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn duplicate_and_invalid_kids_entries_stop_at_budget() {
|
||||
// Duplicate references and non-reference junk never grow `visited`, so
|
||||
// without charging examined entries against the budget an oversized
|
||||
// array of them would iterate to completion. The walk must still
|
||||
// terminate and extract the single real leaf exactly once.
|
||||
let mut doc = Document::new();
|
||||
let leaf_id = doc.new_object_id();
|
||||
doc.set_object(
|
||||
leaf_id,
|
||||
dictionary! {
|
||||
"FT" => "Tx",
|
||||
"V" => Object::string_literal("v"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
},
|
||||
);
|
||||
// A `/Kids` array far wider than the budget: half duplicate references
|
||||
// to the same leaf, half invalid (null) entries.
|
||||
let mut kids: Vec<Object> = Vec::new();
|
||||
for i in 0..(MAX_FORM_FIELD_NODES * 2) {
|
||||
if i % 2 == 0 {
|
||||
kids.push(Object::Reference(leaf_id));
|
||||
} else {
|
||||
kids.push(Object::Null);
|
||||
}
|
||||
}
|
||||
let root_id = doc.add_object(dictionary! {
|
||||
"T" => Object::string_literal("root"),
|
||||
"Kids" => kids,
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(root_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::new();
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
assert_eq!(items.len(), 1);
|
||||
}
|
||||
}
|
||||
|
||||
+21
-861
@@ -2,12 +2,10 @@
|
||||
//!
|
||||
//! This module extracts text with position information for structure detection.
|
||||
|
||||
mod base14;
|
||||
pub(crate) mod content_stream;
|
||||
mod fonts;
|
||||
mod layout;
|
||||
mod links;
|
||||
mod reading_order;
|
||||
pub(crate) mod underline;
|
||||
mod xobjects;
|
||||
|
||||
@@ -28,15 +26,11 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
#[cfg(test)]
|
||||
use layout::filter_markdown_page_numbers;
|
||||
pub(crate) use layout::filter_markdown_page_numbers_with_removed_pages;
|
||||
pub use layout::group_into_lines;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
@@ -85,33 +79,17 @@ pub fn extract_text_with_positions_pages<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<Vec<TextItem>, PdfError> {
|
||||
let (items, _rects, _lines) =
|
||||
extract_text_with_positions_and_rects_with_password(path, page_filter, None)?;
|
||||
let (items, _rects, _lines) = extract_text_with_positions_and_rects(path, page_filter)?;
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
/// Extract text with positions from a file, limited to specific pages and
|
||||
/// decrypting with `password` when the PDF is encrypted.
|
||||
///
|
||||
/// `page_filter` is an optional set of 1-indexed page numbers to process.
|
||||
/// When `None`, all pages are processed.
|
||||
pub fn extract_text_with_positions_pages_with_password<P: AsRef<Path>>(
|
||||
/// Extract text with positions and rectangles from a file.
|
||||
pub(crate) fn extract_text_with_positions_and_rects<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<Vec<TextItem>, PdfError> {
|
||||
let (items, _rects, _lines) =
|
||||
extract_text_with_positions_and_rects_with_password(path, page_filter, password)?;
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
pub(crate) fn extract_text_with_positions_and_rects_with_password<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let (doc, _) = crate::load_document_from_path_with_password(&path, password)?;
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
@@ -160,92 +138,17 @@ pub(crate) fn extract_positioned_text_from_doc(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false, None)
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false)
|
||||
}
|
||||
|
||||
/// Extract selected pages and gather document-wide folio evidence only when a
|
||||
/// selected page contains an ambiguous contextual page-edge number. Errors on
|
||||
/// selected pages remain fatal; errors on context-only pages are skipped.
|
||||
pub(crate) fn extract_positioned_text_with_folio_context(
|
||||
/// Extract with option to include invisible (Tr=3) text.
|
||||
/// Used for Mixed/template PDFs where the OCR text layer is invisible.
|
||||
pub(crate) fn extract_positioned_text_include_invisible(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, false)
|
||||
}
|
||||
|
||||
/// Invisible-text variant of [`extract_positioned_text_with_folio_context`].
|
||||
pub(crate) fn extract_positioned_text_include_invisible_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, true)
|
||||
}
|
||||
|
||||
fn extract_positioned_text_with_folio_context_impl(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let Some(required_pages) = page_filter else {
|
||||
return extract_positioned_text_impl(doc, font_cmaps, None, include_invisible, None);
|
||||
};
|
||||
|
||||
let (
|
||||
(mut selected_items, mut selected_rects, mut selected_lines),
|
||||
mut page_thresholds,
|
||||
mut gid_encoded_pages,
|
||||
) = extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(required_pages),
|
||||
include_invisible,
|
||||
None,
|
||||
)?;
|
||||
if !layout::needs_document_page_number_context(&selected_items, doc.get_pages().len()) {
|
||||
return Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
));
|
||||
}
|
||||
|
||||
let context_pages: HashSet<u32> = doc
|
||||
.get_pages()
|
||||
.keys()
|
||||
.copied()
|
||||
.filter(|page| !required_pages.contains(page))
|
||||
.collect();
|
||||
let ((context_items, context_rects, context_lines), context_thresholds, context_gid_pages) =
|
||||
extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(&context_pages),
|
||||
include_invisible,
|
||||
Some(required_pages),
|
||||
)?;
|
||||
selected_items.extend(context_items);
|
||||
selected_rects.extend(context_rects);
|
||||
selected_lines.extend(context_lines);
|
||||
page_thresholds.extend(context_thresholds);
|
||||
gid_encoded_pages.extend(context_gid_pages);
|
||||
Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
))
|
||||
}
|
||||
|
||||
/// Extract all pages for document-wide analysis while allowing malformed
|
||||
/// unselected pages to be skipped. Any requested page still fails normally.
|
||||
pub(crate) fn extract_positioned_text_for_document_analysis(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
required_pages: &HashSet<u32>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, None, false, Some(required_pages))
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, true)
|
||||
}
|
||||
|
||||
fn extract_positioned_text_impl(
|
||||
@@ -253,7 +156,6 @@ fn extract_positioned_text_impl(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
required_pages: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let pages = doc.get_pages();
|
||||
let mut all_items = Vec::new();
|
||||
@@ -275,25 +177,15 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let page_result = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
);
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) = match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error) if required_pages.is_some_and(|required| !required.contains(page_num)) => {
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) =
|
||||
extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
@@ -423,9 +315,7 @@ fn extract_positioned_text_impl(
|
||||
}
|
||||
|
||||
// Extract AcroForm field values
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num)
|
||||
.into_iter()
|
||||
.filter(|item| page_filter.is_none_or(|filter| filter.contains(&item.page)));
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num);
|
||||
all_items.extend(form_items);
|
||||
|
||||
Ok((
|
||||
@@ -1627,736 +1517,6 @@ mod tests {
|
||||
assert_eq!(lines[1].text(), "Next line");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserving_all_text_keeps_numeric_page_footer() {
|
||||
let mut page_number = make_merge_item("42", 100.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
assert!(group_into_lines(vec![page_number.clone()]).is_empty());
|
||||
|
||||
let lines = group_into_lines_preserving_all_text(vec![page_number]);
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_numeric_run_near_page_edge_is_not_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Total 730 seats");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_page_footer_separated_from_label_is_removed() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut footer_label = make_merge_item("DOCUMENT FOOTER", 60.0, 100.0);
|
||||
footer_label.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, footer_label]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "DOCUMENT FOOTER");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decorative_marker_does_not_contextualize_numeric_page_footer() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut page_number = make_merge_item("42", 37.0, 10.0);
|
||||
page_number.y = 30.0;
|
||||
let mut footer_label = make_merge_item("Company report footer", 68.0, 120.0);
|
||||
footer_label.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, page_number, footer_label]);
|
||||
|
||||
assert!(lines.iter().all(|line| !line.text().contains("42")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_is_removed_in_a_short_document() {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item("42", 57.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![label, page_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_with_running_header_suffix_is_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("of", 73.0, 12.0),
|
||||
make_merge_item("100", 89.0, 18.0),
|
||||
make_merge_item("Report header", 111.0, 78.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report header");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_of_total_expression_is_removed_without_leaving_fragments() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 482.0, 27.0),
|
||||
make_merge_item("1", 513.0, 6.0),
|
||||
make_merge_item("of", 523.0, 10.0),
|
||||
make_merge_item("15", 537.0, 12.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 46.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn document_folio_filter_survives_per_page_layout_splitting() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
items.extend([label, page_number]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 3);
|
||||
assert!(filtered
|
||||
.iter()
|
||||
.all(|item| !matches!(item.text.as_str(), "42" | "43" | "44")));
|
||||
let mut lines = Vec::new();
|
||||
for page in 1..=3 {
|
||||
let page_items = filtered
|
||||
.iter()
|
||||
.filter(|item| item.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
lines.extend(
|
||||
group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
page_items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert!(lines.iter().all(|line| line.text() == "Page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_page_edge_runs_do_not_contextualize_folios() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut long_number = make_merge_item("12345", 43.0, 30.0);
|
||||
long_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, long_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12345");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structured_and_dense_numeric_page_edge_runs_are_preserved() {
|
||||
let mut list_marker = make_merge_item("11)", 25.0, 18.0);
|
||||
list_marker.y = 50.0;
|
||||
let mut chapter = make_merge_item("13", 47.0, 12.0);
|
||||
chapter.y = 50.0;
|
||||
|
||||
let mut isbn_prefix = make_merge_item("9", 25.0, 6.0);
|
||||
isbn_prefix.page = 2;
|
||||
isbn_prefix.y = 50.0;
|
||||
let mut isbn_mid = make_merge_item("780113", 35.0, 36.0);
|
||||
isbn_mid.page = 2;
|
||||
isbn_mid.y = 50.0;
|
||||
let mut isbn_end = make_merge_item("227426", 75.0, 36.0);
|
||||
isbn_end.page = 2;
|
||||
isbn_end.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![list_marker, chapter, isbn_prefix, isbn_mid, isbn_end]);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "11) 13");
|
||||
assert_eq!(lines[1].text(), "9 780113 227426");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn incrementing_numeric_body_column_is_not_treated_as_a_folio() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "13"), (2, "14"), (3, "15")] {
|
||||
let mut row_number = make_merge_item(value, 72.0, 12.0);
|
||||
row_number.page = page;
|
||||
row_number.y = 730.0;
|
||||
let mut name = make_merge_item("Person", 90.0, 42.0);
|
||||
name.page = page;
|
||||
name.y = 730.0;
|
||||
items.extend([row_number, name]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert_eq!(lines[0].text(), "13 Person");
|
||||
assert_eq!(lines[1].text(), "14 Person");
|
||||
assert_eq!(lines[2].text(), "15 Person");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn advancing_number_in_repeated_deep_margin_footer_is_removed() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_substantive_page_number_prose_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44"), (4, "45")] {
|
||||
let mut page_label = make_merge_item("Page", 25.0, 28.0);
|
||||
page_label.page = page;
|
||||
page_label.y = 30.0;
|
||||
let mut number = make_merge_item(value, 57.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut explanation = make_merge_item("explains the result", 73.0, 108.0);
|
||||
explanation.page = page;
|
||||
explanation.y = 30.0;
|
||||
items.extend([page_label, number, explanation]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
for (line, value) in lines.iter().zip(["42", "43", "44", "45"]) {
|
||||
assert_eq!(line.text(), format!("Page {value} explains the result"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_candidates_do_not_bridge_lexical_context() {
|
||||
let mut report = make_merge_item("Report", 25.0, 40.0);
|
||||
report.y = 30.0;
|
||||
let mut year = make_merge_item("2026", 69.0, 24.0);
|
||||
year.y = 30.0;
|
||||
let mut folio = make_merge_item("42", 97.0, 12.0);
|
||||
folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![report, year, folio]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_folio_delimiters_are_removed_with_the_number() {
|
||||
let mut left = make_merge_item("-", 270.0, 6.0);
|
||||
left.y = 30.0;
|
||||
let mut number = make_merge_item("42", 280.0, 12.0);
|
||||
number.y = 30.0;
|
||||
let mut right = make_merge_item("-", 296.0, 6.0);
|
||||
right.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![left, number, right]);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_delimiters_inside_substantive_text_are_preserved() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Result", 240.0, 36.0),
|
||||
make_merge_item("-", 280.0, 6.0),
|
||||
make_merge_item("42", 290.0, 12.0),
|
||||
make_merge_item("-", 306.0, 6.0),
|
||||
make_merge_item("approved", 316.0, 48.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 30.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Result-42-approved");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn changing_year_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, year) in [(1, "2020"), (2, "2021"), (3, "2022"), (4, "2023")] {
|
||||
let mut year = make_merge_item(year, 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert_eq!(lines[0].text(), "2020 Annual report");
|
||||
assert_eq!(lines[3].text(), "2023 Annual report");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_repeated_margin_numbers_do_not_meet_the_folio_evidence_floor() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
if page <= 2 {
|
||||
let value = if page == 1 { "2" } else { "4" };
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
} else {
|
||||
let mut body = make_merge_item("Body text", 72.0, 54.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
}
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "2 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "4 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_document_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (10, "10"), (19, "19"), (28, "28")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "1 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "28 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_blank_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 20);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().any(|item| item.text == "4"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefiltered_contextual_number_survives_layout_partitioning() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 1);
|
||||
let partitioned_number: Vec<TextItem> = filtered
|
||||
.into_iter()
|
||||
.filter(|item| item.text == "730")
|
||||
.collect();
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
partitioned_number,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "730");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_partition_does_not_define_columns() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..20 {
|
||||
let y = 90.0 - row as f32 * 4.0;
|
||||
let mut left = make_merge_item(&(row + 1).to_string(), 50.0, 20.0);
|
||||
left.y = y;
|
||||
let mut right = make_merge_item(&(row + 101).to_string(), 350.0, 20.0);
|
||||
right.y = y;
|
||||
items.extend([left, right]);
|
||||
}
|
||||
assert_eq!(detect_columns(&items, 1, false).len(), 2);
|
||||
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 20);
|
||||
assert!(lines.iter().all(|line| line.items.len() == 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separated_content_is_not_treated_as_a_spread_folio_pair() {
|
||||
let mut value = make_merge_item("12", 100.0, 12.0);
|
||||
value.y = 30.0;
|
||||
let mut label = make_merge_item("Total", 116.0, 30.0);
|
||||
label.y = 30.0;
|
||||
let mut unrelated_number = make_merge_item("13", 300.0, 12.0);
|
||||
unrelated_number.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![value, label, unrelated_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12 Total");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_folio_uses_the_full_page_edge_band() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 80.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 80.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folio_on_facing_page_spread_is_removed() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut left_folio = make_merge_item("326", 35.0, 17.0);
|
||||
left_folio.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 61.0, 120.0);
|
||||
footer.y = 30.0;
|
||||
let mut right_folio = make_merge_item("327", 1148.0, 17.0);
|
||||
right_folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, left_folio, footer, right_folio]);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| !line.text().contains("326") && !line.text().contains("327")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folios_alternating_across_pages_are_removed() {
|
||||
let headers = [
|
||||
"Letter to shareholders",
|
||||
"Corporate governance report",
|
||||
"Business environment overview",
|
||||
"Consolidated financial statements",
|
||||
];
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=8 {
|
||||
let mut body = make_merge_item("Body text", 50.0, 500.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
|
||||
let mut folio = make_merge_item(&(page + 22).to_string(), 0.0, 14.0);
|
||||
folio.page = page;
|
||||
folio.y = 780.0;
|
||||
if page % 2 == 0 {
|
||||
folio.x = 50.0;
|
||||
items.push(folio);
|
||||
} else {
|
||||
let mut header = make_merge_item(headers[(page / 2) as usize], 350.0, 180.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
folio.x = 536.0;
|
||||
items.extend([header, folio]);
|
||||
}
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 8);
|
||||
|
||||
assert!(filtered.iter().all(|item| {
|
||||
!matches!(
|
||||
item.text.as_str(),
|
||||
"23" | "24" | "25" | "26" | "27" | "28" | "29" | "30"
|
||||
)
|
||||
}));
|
||||
assert!(headers
|
||||
.iter()
|
||||
.all(|header| filtered.iter().any(|item| item.text == *header)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_isolated_neighbor_does_not_remove_contextual_number() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 450.0, 70.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 526.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated = make_merge_item("2", 50.0, 7.0);
|
||||
isolated.page = 2;
|
||||
isolated.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, label, contextual, body_two, isolated], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().all(|item| item.text != "2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_content_span_does_not_establish_adjacent_page_edges() {
|
||||
let mut body_one = make_merge_item("Body text", 100.0, 120.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 170.0, 60.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 235.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated_two = make_merge_item("2", 100.0, 7.0);
|
||||
isolated_two.page = 2;
|
||||
isolated_two.y = 780.0;
|
||||
|
||||
let mut body_four = body_one.clone();
|
||||
body_four.page = 4;
|
||||
let mut isolated_four = make_merge_item("4", 100.0, 7.0);
|
||||
isolated_four.page = 4;
|
||||
isolated_four.y = 780.0;
|
||||
|
||||
let filtered = filter_markdown_page_numbers(
|
||||
vec![
|
||||
body_one,
|
||||
label,
|
||||
contextual,
|
||||
body_two,
|
||||
isolated_two,
|
||||
body_four,
|
||||
isolated_four,
|
||||
],
|
||||
4,
|
||||
);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_edge_number_on_an_adjacent_page_is_not_folio_evidence() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut isolated = make_merge_item("42", 50.0, 14.0);
|
||||
isolated.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut contextual = make_merge_item("43", 50.0, 14.0);
|
||||
contextual.page = 2;
|
||||
contextual.y = 780.0;
|
||||
let mut label = make_merge_item("cases reviewed", 70.0, 90.0);
|
||||
label.page = 2;
|
||||
label.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, isolated, body_two, contextual, label], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "43"));
|
||||
assert!(filtered.iter().any(|item| item.text == "cases reviewed"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_number_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
let mut year = make_merge_item("2026", 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines.iter().all(|line| line.text() == "2026 Annual report"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_prefix_does_not_remove_substantive_text_during_layout() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("explains", 73.0, 44.0),
|
||||
make_merge_item("the result", 121.0, 55.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42 explains the result");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_page_number_prefix_with_substantive_text_is_preserved_during_layout() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value, chapter) in [(1, "42", "Chapter 1"), (2, "43", "Chapter 2")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
let mut suffix = make_merge_item(chapter, 73.0, 58.0);
|
||||
suffix.page = page;
|
||||
suffix.y = 50.0;
|
||||
items.extend([label, page_number, suffix]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "Page 42 Chapter 1");
|
||||
assert_eq!(lines[1].text(), "Page 43 Chapter 2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_phrase_in_the_page_body_is_preserved() {
|
||||
let items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
];
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_numeric_context_near_page_edge_is_preserved() {
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
|
||||
let chapter_lines = group_into_lines(vec![chapter, chapter_number]);
|
||||
assert_eq!(chapter_lines.len(), 1);
|
||||
assert_eq!(chapter_lines[0].text(), "Chapter 1");
|
||||
|
||||
let mut year = make_merge_item("2026", 100.0, 24.0);
|
||||
year.y = 760.0;
|
||||
let mut report = make_merge_item("Report", 130.0, 36.0);
|
||||
report.y = 760.0;
|
||||
|
||||
let report_lines = group_into_lines(vec![year, report]);
|
||||
assert_eq!(report_lines.len(), 1);
|
||||
assert_eq!(report_lines[0].text(), "2026 Report");
|
||||
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
let mut edition_year = make_merge_item("2026", 163.0, 24.0);
|
||||
edition_year.y = 760.0;
|
||||
|
||||
let chained_lines = group_into_lines(vec![chapter, chapter_number, edition_year]);
|
||||
assert_eq!(chained_lines.len(), 1);
|
||||
assert_eq!(chained_lines[0].text(), "Chapter 1 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bold_italic_detection() {
|
||||
// Test bold detection
|
||||
|
||||
@@ -1,591 +0,0 @@
|
||||
//! Region-graph evidence for page reading order.
|
||||
//!
|
||||
//! Whole-page column histograms fail when images or spanning captions occupy
|
||||
//! only part of a page. This module turns image geometry and repeated row
|
||||
//! gutters into a small directed acyclic graph: content above a local column
|
||||
//! band, the left flow, the right flow, and content below it. The graph is
|
||||
//! deliberately evidence-gated; ordinary pages keep the established layout
|
||||
//! path.
|
||||
|
||||
use crate::text_utils::{effective_width, is_cjk_char, is_rtl_text};
|
||||
use crate::types::TextItem;
|
||||
|
||||
const MIN_IMAGE_WIDTH: f32 = 60.0;
|
||||
const MIN_IMAGE_HEIGHT: f32 = 40.0;
|
||||
const MIN_ROW_GUTTER: f32 = 8.0;
|
||||
const SPLIT_CLUSTER_TOLERANCE: f32 = 20.0;
|
||||
const MIN_ALIGNED_ROWS: usize = 4;
|
||||
|
||||
pub(crate) type ImageRegion = (f32, f32, f32, f32);
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub(crate) struct ColumnFlowBand {
|
||||
pub(crate) split_x: f32,
|
||||
pub(crate) y_bottom: f32,
|
||||
pub(crate) y_top: f32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum RegionKind {
|
||||
FullWidth,
|
||||
Column,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RegionNode {
|
||||
pub(crate) kind: RegionKind,
|
||||
pub(crate) items: Vec<TextItem>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Row<'a> {
|
||||
y: f32,
|
||||
items: Vec<&'a TextItem>,
|
||||
}
|
||||
|
||||
fn page_x_bounds(items: &[TextItem], images: &[ImageRegion]) -> Option<(f32, f32)> {
|
||||
let text_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let text_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + effective_width(item))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let image_min = images
|
||||
.iter()
|
||||
.map(|region| region.0.min(region.2))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_max = images
|
||||
.iter()
|
||||
.map(|region| region.0.max(region.2))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let x_min = text_min.min(image_min);
|
||||
let x_max = text_max.max(image_max);
|
||||
(x_min.is_finite() && x_max.is_finite() && x_max > x_min).then_some((x_min, x_max))
|
||||
}
|
||||
|
||||
fn group_rows(items: &[TextItem]) -> Vec<Row<'_>> {
|
||||
const Y_TOLERANCE: f32 = 3.0;
|
||||
let mut sorted: Vec<&TextItem> = items.iter().collect();
|
||||
sorted.sort_by(|left, right| right.y.total_cmp(&left.y));
|
||||
let mut rows: Vec<Row<'_>> = Vec::new();
|
||||
for item in sorted {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| (row.y - item.y).abs() <= Y_TOLERANCE)
|
||||
{
|
||||
row.items.push(item);
|
||||
row.y = row.items.iter().map(|member| member.y).sum::<f32>() / row.items.len() as f32;
|
||||
} else {
|
||||
rows.push(Row {
|
||||
y: item.y,
|
||||
items: vec![item],
|
||||
});
|
||||
}
|
||||
}
|
||||
for row in &mut rows {
|
||||
row.items.sort_by(|left, right| left.x.total_cmp(&right.x));
|
||||
}
|
||||
rows
|
||||
}
|
||||
|
||||
fn side_is_prose(items: &[&TextItem]) -> bool {
|
||||
let text = items
|
||||
.iter()
|
||||
.map(|item| item.text.trim())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let alphabetic_count = text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphabetic())
|
||||
.count();
|
||||
let cjk_count = text
|
||||
.chars()
|
||||
.filter(|character| is_cjk_char(*character))
|
||||
.count();
|
||||
(text.split_whitespace().count() >= 3 || cjk_count >= 10) && alphabetic_count >= 10
|
||||
}
|
||||
|
||||
fn aligned_row_split(row: &Row<'_>, x_min: f32, x_max: f32) -> Option<f32> {
|
||||
if row.items.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let page_width = x_max - x_min;
|
||||
let center_low = x_min + page_width * 0.25;
|
||||
let center_high = x_min + page_width * 0.75;
|
||||
row.items
|
||||
.windows(2)
|
||||
.filter_map(|pair| {
|
||||
let left_end = pair[0].x + effective_width(pair[0]);
|
||||
let right_start = pair[1].x;
|
||||
let gap = right_start - left_end;
|
||||
let split_x = (left_end + right_start) / 2.0;
|
||||
if gap < MIN_ROW_GUTTER || split_x < center_low || split_x > center_high {
|
||||
return None;
|
||||
}
|
||||
let left: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 < split_x)
|
||||
.collect();
|
||||
let right: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 >= split_x)
|
||||
.collect();
|
||||
(side_is_prose(&left) && side_is_prose(&right)).then_some((split_x, gap))
|
||||
})
|
||||
.max_by(|left, right| left.1.total_cmp(&right.1))
|
||||
.map(|candidate| candidate.0)
|
||||
}
|
||||
|
||||
fn local_flow_below_full_width_image(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
let full_width_images: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let width = (x1 - x0).abs();
|
||||
let height = (y1 - y0).abs();
|
||||
width >= page_width * 0.65 && height >= 60.0
|
||||
})
|
||||
.collect();
|
||||
// A local column flow below an image is only unambiguous for a single,
|
||||
// nearly square hero/figure. Wide report banners and full-page artwork
|
||||
// frequently sit above unrelated page furniture whose aligned labels can
|
||||
// mimic prose columns.
|
||||
if full_width_images.len() != 1 {
|
||||
return None;
|
||||
}
|
||||
let (image_x0, _, image_x1, _) = full_width_images[0];
|
||||
let anchor_width = (image_x1 - image_x0).abs();
|
||||
let anchor_height = (full_width_images[0].3 - full_width_images[0].1).abs();
|
||||
if anchor_width < page_width * 0.85
|
||||
|| anchor_height < anchor_width * 0.85
|
||||
|| anchor_height > anchor_width * 1.2
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let image_bottom = full_width_images
|
||||
.iter()
|
||||
.map(|&(_, y0, _, y1)| y0.min(y1))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if !image_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let below: Vec<TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| item.y < image_bottom && item.y >= image_bottom - 220.0)
|
||||
.cloned()
|
||||
.collect();
|
||||
let candidates: Vec<(f32, f32)> = group_rows(&below)
|
||||
.into_iter()
|
||||
.filter_map(|row| aligned_row_split(&row, x_min, x_max).map(|split| (split, row.y)))
|
||||
.collect();
|
||||
if candidates.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut clusters: Vec<Vec<(f32, f32)>> = Vec::new();
|
||||
for candidate in candidates {
|
||||
if let Some(cluster) = clusters.iter_mut().find(|cluster| {
|
||||
let mean = cluster.iter().map(|entry| entry.0).sum::<f32>() / cluster.len() as f32;
|
||||
(mean - candidate.0).abs() <= SPLIT_CLUSTER_TOLERANCE
|
||||
}) {
|
||||
cluster.push(candidate);
|
||||
} else {
|
||||
clusters.push(vec![candidate]);
|
||||
}
|
||||
}
|
||||
let dominant = clusters.into_iter().max_by_key(Vec::len)?;
|
||||
if dominant.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
let split_x = dominant.iter().map(|entry| entry.0).sum::<f32>() / dominant.len() as f32;
|
||||
let y_top = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
let image_gap = image_bottom - y_top;
|
||||
if !(60.0..=120.0).contains(&image_gap) {
|
||||
return None;
|
||||
}
|
||||
let y_bottom = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if y_top - y_bottom > 130.0 {
|
||||
return None;
|
||||
}
|
||||
log::debug!(
|
||||
"page {}: full-width image flow images={} aligned_rows={} split={:.1} page=[{:.1}..{:.1}] image_bottom={:.1} y=[{:.1}..{:.1}] full_width={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
images.len(),
|
||||
dominant.len(),
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
image_bottom,
|
||||
y_bottom,
|
||||
y_top,
|
||||
full_width_images
|
||||
);
|
||||
Some(ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
})
|
||||
}
|
||||
|
||||
fn paired_column_images(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
split_x: f32,
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
if split_x < x_min + page_width * 0.4 || split_x > x_min + page_width * 0.6 {
|
||||
return None;
|
||||
}
|
||||
let qualifying: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let image_left = x0.min(x1);
|
||||
let image_right = x0.max(x1);
|
||||
let confined_to_one_column = image_right <= split_x || image_left >= split_x;
|
||||
confined_to_one_column
|
||||
&& (x1 - x0).abs() >= MIN_IMAGE_WIDTH
|
||||
&& (y1 - y0).abs() >= MIN_IMAGE_HEIGHT
|
||||
})
|
||||
.collect();
|
||||
let wide_images: Vec<ImageRegion> = qualifying
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(x0, _, x1, _)| (x1 - x0).abs() >= page_width * 0.35)
|
||||
.collect();
|
||||
if qualifying.len() < 3 || wide_images.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
let has_left = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 < split_x);
|
||||
let has_right = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 >= split_x);
|
||||
if !has_left || !has_right {
|
||||
return None;
|
||||
}
|
||||
// A meaningful image-backed column flow spans multiple vertical panels.
|
||||
// Three same-row header/logo images can otherwise satisfy the image count
|
||||
// and send an ordinary asymmetric page through sequential column order.
|
||||
let image_y_min = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.min(region.3))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_y_max = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let has_vertical_stack = wide_images.iter().enumerate().any(|(index, left)| {
|
||||
wide_images.iter().skip(index + 1).any(|right| {
|
||||
let same_side =
|
||||
((left.0 + left.2) / 2.0 < split_x) == ((right.0 + right.2) / 2.0 < split_x);
|
||||
let left_center = (left.1 + left.3) / 2.0;
|
||||
let right_center = (right.1 + right.3) / 2.0;
|
||||
let left_height = (left.3 - left.1).abs();
|
||||
let right_height = (right.3 - right.1).abs();
|
||||
let vertical_gap = if left.1.max(left.3) < right.1.min(right.3) {
|
||||
right.1.min(right.3) - left.1.max(left.3)
|
||||
} else if right.1.max(right.3) < left.1.min(left.3) {
|
||||
left.1.min(left.3) - right.1.max(right.3)
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
same_side
|
||||
&& (left_center - right_center).abs() >= left_height.min(right_height) * 0.5
|
||||
&& vertical_gap <= left_height.max(right_height) * 0.5
|
||||
})
|
||||
});
|
||||
if image_y_max - image_y_min < page_width * 0.45 || !has_vertical_stack {
|
||||
return None;
|
||||
}
|
||||
let y_top = qualifying
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
// Only column-confined text proves the lower extent of the flow. A
|
||||
// spanning heading or caption below the columns must become the trailing
|
||||
// full-width node rather than stretching the column band to the page foot.
|
||||
let y_bottom = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
let item_right = item.x + effective_width(item);
|
||||
item.y <= y_top && (item_right <= split_x || item.x >= split_x)
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if !y_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
let distinct_rows = |right: bool| {
|
||||
let mut ys: Vec<f32> = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
item.y <= y_top && (item.x + effective_width(item) / 2.0 >= split_x) == right
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.collect();
|
||||
ys.sort_by(|left, right| left.total_cmp(right));
|
||||
ys.dedup_by(|left, right| (*left - *right).abs() <= 3.0);
|
||||
ys.len()
|
||||
};
|
||||
let left_rows = distinct_rows(false);
|
||||
let right_rows = distinct_rows(true);
|
||||
let line_balance = left_rows.min(right_rows) as f32 / left_rows.max(right_rows).max(1) as f32;
|
||||
(left_rows >= 5 && right_rows >= 5 && line_balance < 0.55).then(|| {
|
||||
log::debug!(
|
||||
"page {}: paired-image flow qualifying_images={} rows={}/{} split={:.1} page=[{:.1}..{:.1}] y=[{:.1}..{:.1}] images={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
qualifying.len(),
|
||||
left_rows,
|
||||
right_rows,
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
y_bottom,
|
||||
y_top,
|
||||
qualifying
|
||||
);
|
||||
ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn infer_image_anchored_flow(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
detected_split: Option<f32>,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
if items.is_empty() || images.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let (x_min, x_max) = page_x_bounds(items, images)?;
|
||||
detected_split
|
||||
.and_then(|split_x| paired_column_images(items, images, split_x, x_min, x_max))
|
||||
.or_else(|| local_flow_below_full_width_image(items, images, x_min, x_max))
|
||||
}
|
||||
|
||||
/// Partition a page into the topological order `above -> left -> right -> below`.
|
||||
/// These edges encode the reading-order DAG; empty nodes are omitted.
|
||||
pub(crate) fn build_region_graph(items: Vec<TextItem>, band: ColumnFlowBand) -> Vec<RegionNode> {
|
||||
let mut above = Vec::new();
|
||||
let mut left = Vec::new();
|
||||
let mut right = Vec::new();
|
||||
let mut below = Vec::new();
|
||||
for item in items {
|
||||
if item.y > band.y_top {
|
||||
above.push(item);
|
||||
} else if item.y < band.y_bottom {
|
||||
below.push(item);
|
||||
} else if item.x + effective_width(&item) / 2.0 < band.split_x {
|
||||
left.push(item);
|
||||
} else {
|
||||
right.push(item);
|
||||
}
|
||||
}
|
||||
let rtl = is_rtl_text(left.iter().chain(right.iter()).map(|item| &item.text));
|
||||
let mut ordered = vec![(RegionKind::FullWidth, above)];
|
||||
if rtl {
|
||||
ordered.push((RegionKind::Column, right));
|
||||
ordered.push((RegionKind::Column, left));
|
||||
} else {
|
||||
ordered.push((RegionKind::Column, left));
|
||||
ordered.push((RegionKind::Column, right));
|
||||
}
|
||||
ordered.push((RegionKind::FullWidth, below));
|
||||
ordered
|
||||
.into_iter()
|
||||
.filter_map(|(kind, items)| (!items.is_empty()).then_some(RegionNode { kind, items }))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.into(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 11.0,
|
||||
font: "F1".into(),
|
||||
font_size: 11.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_local_two_column_flow() {
|
||||
let mut items = vec![
|
||||
item("A full width caption", 55.0, 230.0, 430.0),
|
||||
item("A trailing full width heading", 55.0, 80.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
let band = infer_image_anchored_flow(&items, &images, None).unwrap();
|
||||
assert!((band.split_x - 272.5).abs() < 2.0);
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph.len(), 4);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "A trailing full width heading");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_cjk_column_flow() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("左栏这是没有空格的正文内容", 55.0, y, 210.0));
|
||||
items.push(item("右栏这是没有空格的正文内容", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_images_anchor_unbalanced_column_flows() {
|
||||
let mut items = vec![
|
||||
item("running header", 55.0, 700.0, 430.0),
|
||||
item("trailing full width caption", 55.0, 300.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
500.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 5..12 {
|
||||
items.push(item(
|
||||
"right continuation prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 530.0, 255.0, 680.0),
|
||||
(55.0, 380.0, 255.0, 530.0),
|
||||
(280.0, 560.0, 490.0, 680.0),
|
||||
];
|
||||
let band = infer_image_anchored_flow(&items, &images, Some(270.0)).unwrap();
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "trailing full width caption");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rtl_region_graph_reads_right_column_first() {
|
||||
let items = vec![
|
||||
item("A long English report header", 55.0, 250.0, 430.0),
|
||||
item("نص العمود الأيسر", 55.0, 150.0, 180.0),
|
||||
item("نص العمود الأيمن", 300.0, 150.0, 180.0),
|
||||
];
|
||||
let graph = build_region_graph(
|
||||
items,
|
||||
ColumnFlowBand {
|
||||
split_x: 270.0,
|
||||
y_bottom: 100.0,
|
||||
y_top: 200.0,
|
||||
},
|
||||
);
|
||||
|
||||
assert_eq!(graph.len(), 3);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert!(graph[1].items[0].x > graph[2].items[0].x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_header_logos_do_not_anchor_page_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 0..30 {
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 720.0, 205.0, 770.0),
|
||||
(60.0, 718.0, 210.0, 768.0),
|
||||
(280.0, 720.0, 450.0, 770.0),
|
||||
];
|
||||
assert!(infer_image_anchored_flow(&items, &images, Some(270.0)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_banner_does_not_anchor_local_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
let y = 270.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 310.0, 490.0, 550.0)];
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_none());
|
||||
}
|
||||
}
|
||||
+17
-385
@@ -44,16 +44,6 @@ const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
/// Strikeout decorations are text-sized. Diagram connectors, signature
|
||||
/// lines, and chart rules often cross glyphs too, but extend well beyond the
|
||||
/// text they happen to intersect.
|
||||
const STRIKE_OWNER_PAD_EM: f32 = 0.75;
|
||||
const STRIKE_OWNER_MIN_PAD: f32 = 4.0;
|
||||
const STRIKE_ROW_Y_TOLERANCE_EM: f32 = 0.15;
|
||||
const STRIKE_ROW_Y_TOLERANCE_MIN: f32 = 5.0;
|
||||
const GRAPHIC_CONNECTION_EPS: f32 = 2.0;
|
||||
const GRAPHIC_CONNECTOR_MAX_THICKNESS: f32 = 4.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
@@ -397,206 +387,6 @@ fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
fn is_bare_list_marker(text: &str) -> bool {
|
||||
matches!(
|
||||
text.trim(),
|
||||
"•" | "◦" | "▪" | "▫" | "‣" | "⁃" | "●" | "○" | "■" | "□" | "-" | "*"
|
||||
)
|
||||
}
|
||||
|
||||
fn same_strike_row(left: &TextItem, right: &TextItem) -> bool {
|
||||
let font_size = left.font_size.max(right.font_size);
|
||||
let tolerance = (font_size * STRIKE_ROW_Y_TOLERANCE_EM).max(STRIKE_ROW_Y_TOLERANCE_MIN);
|
||||
(left.y - right.y).abs() <= tolerance
|
||||
}
|
||||
|
||||
fn is_inline_script(rule: &Rule, candidate: &TextItem, parent: &TextItem) -> bool {
|
||||
if !is_underline_candidate(candidate)
|
||||
|| is_bare_list_marker(&candidate.text)
|
||||
|| candidate.font_size <= 0.0
|
||||
|| candidate.font_size >= parent.font_size * 0.75
|
||||
|| candidate.text.len() > 4
|
||||
|| !candidate.text.chars().all(|c| c.is_ascii_digit())
|
||||
|| (candidate.y - parent.y).abs() > 5.0
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_ends_with_letter = parent.text.chars().last().is_some_and(char::is_alphabetic);
|
||||
if !parent_ends_with_letter {
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_right = parent.x + parent.width;
|
||||
let gap = candidate.x - parent_right;
|
||||
if gap >= parent.font_size * 0.2 || gap <= -parent.font_size * 0.3 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = rule.x2.min(candidate.x + candidate.width) - rule.x1.max(candidate.x);
|
||||
overlap >= candidate.width * MIN_X_OVERLAP
|
||||
}
|
||||
|
||||
/// Return the items owned by a snug mid-glyph rule.
|
||||
///
|
||||
/// Real strikeout decorations track the width of the deleted text, including
|
||||
/// runs split by font/style changes and adjacent numeric super/subscripts.
|
||||
/// Non-text graphics can cross the same vertical window, but arrow shafts,
|
||||
/// signature lines, fraction bars, and chart rules extend materially beyond
|
||||
/// the intersected glyphs. Requiring the rule to stay within a small em-sized
|
||||
/// pad of a contiguous matched row separates those cases without relying on
|
||||
/// document-specific fonts or coordinates.
|
||||
///
|
||||
/// Ownership is computed once per rule. This keeps the strikeout pass at the
|
||||
/// same rule-by-item scale as underline detection instead of rescanning the
|
||||
/// whole page for every matching item.
|
||||
fn snug_strike_owner_indices(rule: &Rule, items: &[TextItem]) -> Vec<usize> {
|
||||
let mut struck_indices: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
(is_underline_candidate(candidate)
|
||||
&& !is_bare_list_marker(&candidate.text)
|
||||
&& rule_strikes_item(rule, candidate))
|
||||
.then_some(index)
|
||||
})
|
||||
.collect();
|
||||
if struck_indices.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
struck_indices.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in struck_indices {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| same_strike_row(&items[row[0]], &items[index]))
|
||||
{
|
||||
row.push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
let mut owned_indices = Vec::new();
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
|
||||
// Underline detection runs before the extractor's script-merging
|
||||
// pass. Include the same tightly adjacent numeric script shape here
|
||||
// when the rule spans it, so the owner width and semantic mark both
|
||||
// survive that later merge.
|
||||
let scripts: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
let parent_pos =
|
||||
row.partition_point(|&row_index| items[row_index].x <= candidate.x);
|
||||
let parent_index = parent_pos.checked_sub(1).map(|pos| row[pos])?;
|
||||
is_inline_script(rule, candidate, &items[parent_index]).then_some(index)
|
||||
})
|
||||
.collect();
|
||||
row.extend(scripts);
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
row.dedup();
|
||||
|
||||
let x1 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x2 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x + items[index].width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let max_font_size = row
|
||||
.iter()
|
||||
.map(|&index| items[index].font_size)
|
||||
.fold(0.0, f32::max);
|
||||
let pad = (max_font_size * STRIKE_OWNER_PAD_EM).max(STRIKE_OWNER_MIN_PAD);
|
||||
|
||||
if rule.x1 < x1 - pad || rule.x2 > x2 + pad {
|
||||
continue;
|
||||
}
|
||||
|
||||
let contiguous = row.windows(2).all(|pair| {
|
||||
let gap = items[pair[1]].x - (items[pair[0]].x + items[pair[0]].width);
|
||||
gap <= (max_font_size * 2.0).max(12.0)
|
||||
});
|
||||
if contiguous {
|
||||
owned_indices.extend(row);
|
||||
}
|
||||
}
|
||||
|
||||
owned_indices.sort_unstable();
|
||||
owned_indices.dedup();
|
||||
owned_indices
|
||||
}
|
||||
|
||||
/// Diagram and table rules participate in larger path geometry. A vertical
|
||||
/// or diagonal segment meeting the candidate rule is strong evidence that
|
||||
/// the horizontal segment is a connector, border, arrow, or symbol rather
|
||||
/// than an isolated text decoration.
|
||||
fn has_connected_nonhorizontal_segment(rule: &Rule, lines: &[UnderlineLine], page: u32) -> bool {
|
||||
lines.iter().any(|line| {
|
||||
if line.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let dx = line.x2 - line.x1;
|
||||
let dy = line.y2 - line.y1;
|
||||
if dy.abs() <= MAX_RULE_THICKNESS {
|
||||
return false;
|
||||
}
|
||||
|
||||
let y_min = line.y1.min(line.y2) - GRAPHIC_CONNECTION_EPS;
|
||||
let y_max = line.y1.max(line.y2) + GRAPHIC_CONNECTION_EPS;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let t = (rule.y - line.y1) / dy;
|
||||
if !(-0.05..=1.05).contains(&t) {
|
||||
return false;
|
||||
}
|
||||
let intersection_x = line.x1 + t * dx;
|
||||
intersection_x >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& intersection_x <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Filled diagrams often build connectors from intersecting thin rectangles
|
||||
/// instead of stroked path segments. Treat only narrow, vertically elongated
|
||||
/// rectangles as connector geometry; broad fills can legitimately sit behind
|
||||
/// struck text and must not veto its decoration.
|
||||
fn has_connected_nonhorizontal_rect(rule: &Rule, rects: &[PdfRect], page: u32) -> bool {
|
||||
rects.iter().any(|rect| {
|
||||
if rect.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let (x1, x2) = if rect.width >= 0.0 {
|
||||
(rect.x, rect.x + rect.width)
|
||||
} else {
|
||||
(rect.x + rect.width, rect.x)
|
||||
};
|
||||
let (y1, y2) = if rect.height >= 0.0 {
|
||||
(rect.y, rect.y + rect.height)
|
||||
} else {
|
||||
(rect.y + rect.height, rect.y)
|
||||
};
|
||||
let width = x2 - x1;
|
||||
let height = y2 - y1;
|
||||
|
||||
width > 0.0
|
||||
&& width <= GRAPHIC_CONNECTOR_MAX_THICKNESS
|
||||
&& height > width * 2.0
|
||||
&& rule.y >= y1 - GRAPHIC_CONNECTION_EPS
|
||||
&& rule.y <= y2 + GRAPHIC_CONNECTION_EPS
|
||||
&& x2 >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& x1 <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
@@ -650,43 +440,27 @@ pub(crate) fn mark_underlined_items(
|
||||
.map(|(i, _)| i)
|
||||
.collect();
|
||||
|
||||
let mut strikeout_items = vec![false; items.len()];
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx)
|
||||
|| has_connected_nonhorizontal_segment(rule, lines, page)
|
||||
|| has_connected_nonhorizontal_rect(rule, rects, page)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
for item_idx in snug_strike_owner_indices(rule, items) {
|
||||
strikeout_items[item_idx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
let underlined_items: HashSet<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
is_underline_candidate(item)
|
||||
&& rules.iter().enumerate().any(|(rule_idx, rule)| {
|
||||
!tabular_rules.contains(&rule_idx)
|
||||
&& !fraction_rules.contains(&rule_idx)
|
||||
&& rule_matches_item(rule, item)
|
||||
})
|
||||
})
|
||||
.map(|(item_idx, _)| item_idx)
|
||||
.collect();
|
||||
|
||||
for (item_idx, item) in items.iter_mut().enumerate() {
|
||||
for item in items.iter_mut() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if strikeout_items[item_idx] {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if underlined_items.contains(&item_idx) {
|
||||
item.is_underline = true;
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
// The fraction guard only gates UNDERLINE marking — a rule that
|
||||
// reads as a fraction bar from below can still legitimately
|
||||
// strike through a line above it.
|
||||
if !fraction_rules.contains(&rule_idx) && rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -806,148 +580,6 @@ mod tests {
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_connector_crossing_text_is_not_a_strikeout() {
|
||||
let mut items = vec![item("diagram label", 160.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 280.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_rule_ending_inside_short_label_is_not_a_strikeout() {
|
||||
let mut items = vec![item("T 18", 300.0, 500.0, 20.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 315.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_diagram_segment_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let lines = vec![
|
||||
hline(99.0, 113.0, 503.0),
|
||||
UnderlineLine {
|
||||
x1: 106.0,
|
||||
y1: 496.0,
|
||||
x2: 109.0,
|
||||
y2: 510.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_filled_rect_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 14.0),
|
||||
PdfRect {
|
||||
x: 106.0,
|
||||
y: 496.0,
|
||||
width: 2.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn broad_fill_behind_text_does_not_block_strikeout() {
|
||||
let mut items = vec![item("deleted", 100.0, 500.0, 40.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 42.0),
|
||||
PdfRect {
|
||||
x: 90.0,
|
||||
y: 490.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_bullet_is_not_a_strikeout() {
|
||||
for marker in ["•", "-", "*"] {
|
||||
let mut items = vec![item(marker, 100.0, 500.0, 6.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 107.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout, "marker {marker:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_marks_adjacent_split_runs_as_strikeout() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 500.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_groups_split_runs_with_baseline_drift() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 498.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_superscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 503.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_subscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 497.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
|
||||
@@ -8,9 +8,8 @@ use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, build_type3_scales, compute_string_width_ts,
|
||||
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
FontStyleCache,
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache, FontStyleCache,
|
||||
};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
@@ -163,11 +162,10 @@ fn extract_form_xobject_text_inner(
|
||||
|
||||
// Get fonts from the Form's Resources
|
||||
let form_fonts = get_form_fonts(doc, &stream.dict);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
|
||||
|
||||
// Build font width info for the form
|
||||
let font_widths = build_font_widths(doc, &form_fonts);
|
||||
let type3_scales = build_type3_scales(doc, &form_fonts);
|
||||
|
||||
// Build font base names and ToUnicode refs for the form
|
||||
let mut font_base_names: HashMap<String, String> = HashMap::new();
|
||||
@@ -415,8 +413,7 @@ fn extract_form_xobject_text_inner(
|
||||
&font_widths,
|
||||
) {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = if let Some(font_info) = font_widths.get(¤t_font) {
|
||||
if let Some(raw_bytes) = get_operand_bytes(&op.operands[0]) {
|
||||
@@ -575,8 +572,7 @@ fn extract_form_xobject_text_inner(
|
||||
}
|
||||
if !sub_items.is_empty() {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let base_font = font_base_names
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
|
||||
+3
-40
@@ -4566,13 +4566,9 @@ pub fn glyph_to_char(name: &str) -> Option<char> {
|
||||
}
|
||||
}
|
||||
|
||||
// Try to parse uniXXXX format.
|
||||
// Use `get` rather than a byte-length check + slice: `name` can contain
|
||||
// non-ASCII bytes (e.g. U+FFFD from lossy UTF-8 decoding of an attacker
|
||||
// controlled /Differences name), so byte index 7 may not be a char
|
||||
// boundary and `&name[3..7]` would panic.
|
||||
if let Some(hex) = name.strip_prefix("uni").and_then(|rest| rest.get(..4)) {
|
||||
if let Ok(code) = u32::from_str_radix(hex, 16) {
|
||||
// Try to parse uniXXXX format
|
||||
if name.starts_with("uni") && name.len() >= 7 {
|
||||
if let Ok(code) = u32::from_str_radix(&name[3..7], 16) {
|
||||
// Strip PUA F000 offset: uniF0XX → U+00XX (Windows Symbol encoding convention)
|
||||
let code = if (0xF000..=0xF0FF).contains(&code) {
|
||||
code - 0xF000
|
||||
@@ -4592,36 +4588,3 @@ pub fn glyph_to_char(name: &str) -> Option<char> {
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn uni_hex_parsing() {
|
||||
assert_eq!(glyph_to_char("uni0041"), Some('A'));
|
||||
assert_eq!(glyph_to_char("uni00e9"), Some('\u{00e9}'));
|
||||
// PUA F0xx symbol-encoding offset is stripped.
|
||||
assert_eq!(glyph_to_char("uniF041"), Some('A'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u_hex_parsing() {
|
||||
assert_eq!(glyph_to_char("u0041"), Some('A'));
|
||||
assert_eq!(glyph_to_char("u1F600"), Some('\u{1F600}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_ascii_uni_name_does_not_panic() {
|
||||
// A crafted /Differences name like `/uni#80#80#80#80` decodes via
|
||||
// from_utf8_lossy into "uni" followed by four U+FFFD replacements.
|
||||
// Byte index 7 lands mid-character, so a naive `&name[3..7]` slice
|
||||
// would panic. It must be handled gracefully instead.
|
||||
let crafted = format!("uni{0}{0}{0}{0}", '\u{FFFD}');
|
||||
assert_eq!(glyph_to_char(&crafted), None);
|
||||
|
||||
// Assorted non-ASCII bytes right after the "uni" prefix.
|
||||
assert_eq!(glyph_to_char("uni\u{FFFD}bc"), None);
|
||||
assert_eq!(glyph_to_char("uni\u{00e9}00"), None);
|
||||
}
|
||||
}
|
||||
|
||||
+37
-646
@@ -50,11 +50,11 @@ pub use detector::{
|
||||
};
|
||||
pub use extractor::{
|
||||
extract_text, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
extract_text_with_positions_pages, extract_text_with_positions_pages_with_password,
|
||||
extract_text_with_positions_pages,
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, MarkdownProfile,
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -68,40 +68,6 @@ use text_quality::{
|
||||
};
|
||||
use tounicode::FontCMaps;
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
struct ProcessingTimer(std::time::Instant);
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
struct ProcessingTimer;
|
||||
|
||||
impl ProcessingTimer {
|
||||
fn start() -> Self {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
Self(std::time::Instant::now())
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
Self
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(&self) -> u64 {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
self.0.elapsed().as_millis() as u64
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
// The wasm32-unknown-unknown standard library has no clock.
|
||||
// Browser bindings measure with JavaScript's host clock.
|
||||
0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reason emitted when the extracted text layer appears garbled due to
|
||||
/// broken font decoding or mojibake.
|
||||
pub const OCR_REASON_SUSPECTED_GARBLED_TEXT: &str = "suspected_garbled_text";
|
||||
@@ -284,7 +250,7 @@ pub fn process_pdf_with_options<P: AsRef<Path>>(
|
||||
path: P,
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = ProcessingTimer::start();
|
||||
let start = std::time::Instant::now();
|
||||
validate_pdf_file(&path)?;
|
||||
|
||||
// Load the document once — shared by detection AND extraction.
|
||||
@@ -311,7 +277,7 @@ pub fn process_pdf_mem_with_options(
|
||||
buffer: &[u8],
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = ProcessingTimer::start();
|
||||
let start = std::time::Instant::now();
|
||||
validate_pdf_bytes(buffer)?;
|
||||
|
||||
let (doc, page_count) =
|
||||
@@ -462,46 +428,16 @@ pub fn extract_pages_markdown_mem(
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
// unselected page cannot make a valid requested page fail, but errors on a
|
||||
// requested page retain the normal extraction semantics.
|
||||
let required_pages: Option<HashSet<u32>> = pages.map(|pages| {
|
||||
pages
|
||||
.iter()
|
||||
.filter_map(|page| page.checked_add(1))
|
||||
.collect()
|
||||
});
|
||||
// Extract ALL pages to get accurate, document-wide font stats.
|
||||
let ((all_items, all_rects, all_lines), page_thresholds, gid_pages) =
|
||||
if let Some(required_pages) = required_pages.as_ref() {
|
||||
extractor::extract_positioned_text_for_document_analysis(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
required_pages,
|
||||
)?
|
||||
} else {
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?
|
||||
};
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||
let text_quality = analyze_text_quality(&all_items);
|
||||
|
||||
// Resolve page numbers with full-document context before partitioning.
|
||||
// Per-page Markdown receives the original items plus these decisions so
|
||||
// table detection can retain legitimate numeric cells.
|
||||
let (filtered_items, removed_page_number_pages, page_number_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
|
||||
// Tables need the original numeric cells; columns use folio-cleaned
|
||||
// evidence so removed page numbers cannot create false layout metadata.
|
||||
let chart_regions = markdown::chart_regions_by_page(&all_items, &all_rects, &all_lines);
|
||||
let complexity = compute_layout_complexity_with_chart_regions(
|
||||
&all_items,
|
||||
&filtered_items,
|
||||
&all_rects,
|
||||
&all_lines,
|
||||
&chart_regions,
|
||||
);
|
||||
// Compute layout complexity from full document (near-zero cost).
|
||||
let complexity = compute_layout_complexity(&all_items, &all_rects, &all_lines);
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -516,7 +452,6 @@ pub fn extract_pages_markdown_mem(
|
||||
let mut results = Vec::with_capacity(pages_slice.len());
|
||||
let mut pages_needing_ocr = Vec::new();
|
||||
let mut ocr_reasons_by_page = BTreeMap::new();
|
||||
let lopdf_pages = doc.get_pages();
|
||||
|
||||
for &page_0idx in pages_slice {
|
||||
// Out-of-range pages → empty + needs_ocr
|
||||
@@ -533,13 +468,12 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
let page_1idx = page_0idx + 1;
|
||||
|
||||
// Partition items, removal decisions, and rects for this page only.
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
// Filter items/rects for this page only
|
||||
let page_items: Vec<TextItem> = all_items
|
||||
.iter()
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
.filter(|i| i.page == page_1idx)
|
||||
.cloned()
|
||||
.collect();
|
||||
|
||||
let page_rects: Vec<PdfRect> = all_rects
|
||||
.iter()
|
||||
@@ -550,25 +484,6 @@ pub fn extract_pages_markdown_mem(
|
||||
let has_gid = gid_pages.contains(&page_1idx);
|
||||
let has_text_quality_issue = text_quality.pages_needing_ocr.contains(&page_1idx);
|
||||
|
||||
// A page can extract cleanly (no decoding issues, non-empty text)
|
||||
// while still being fundamentally a scan: a full-page raster with
|
||||
// a little genuine native text drawn over it (a header, a stamp, a
|
||||
// cover-sheet annotation). Text-quality signals alone can't see
|
||||
// that — consult the same "large background image" signal
|
||||
// classify_pdf/detect_pdf_type already uses, so the two APIs can't
|
||||
// silently disagree on whether a page needs OCR. See #227.
|
||||
// Also covers vector-outlined text (glyphs drawn as paths, not
|
||||
// shown via a text-showing operator): a hybrid page with real
|
||||
// embedded-font body text elsewhere would otherwise still extract
|
||||
// non-empty, non-garbled markdown and miss OCR routing entirely.
|
||||
// detect_from_document's Mixed-type per-page routing always sends
|
||||
// these pages to OCR; mirror that here too. Both signals share one
|
||||
// analyze_page_content pass — see page_ocr_signals's doc comment.
|
||||
let (has_template_image, has_vector_text) = lopdf_pages
|
||||
.get(&page_1idx)
|
||||
.map(|&page_id| detector::page_ocr_signals(&doc, page_id))
|
||||
.unwrap_or((false, false));
|
||||
|
||||
// Build markdown with document-wide font stats
|
||||
let options = MarkdownOptions {
|
||||
base_font_size: Some(font_stats.most_common_size),
|
||||
@@ -585,15 +500,9 @@ pub fn extract_pages_markdown_mem(
|
||||
options,
|
||||
&page_rects,
|
||||
&[],
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_page_number_pages),
|
||||
prefiltered_page_number_mask: Some(&page_number_removal_mask),
|
||||
precomputed_chart_regions: Some(&chart_regions),
|
||||
},
|
||||
&page_thresholds,
|
||||
None,
|
||||
&[],
|
||||
)
|
||||
};
|
||||
|
||||
@@ -606,20 +515,10 @@ pub fn extract_pages_markdown_mem(
|
||||
OCR_REASON_SUSPECTED_GARBLED_TEXT,
|
||||
);
|
||||
}
|
||||
if has_template_image {
|
||||
add_ocr_reason(&mut ocr_reasons_by_page, page_1idx, OCR_REASON_SCANNED);
|
||||
}
|
||||
if has_vector_text {
|
||||
add_ocr_reason(&mut ocr_reasons_by_page, page_1idx, OCR_REASON_VECTOR_TEXT);
|
||||
}
|
||||
let ocr_reason = page_ocr_reason(&ocr_reasons_by_page, page_1idx);
|
||||
|
||||
let needs_ocr = ocr_reason.is_some()
|
||||
|| md.trim().is_empty()
|
||||
|| has_gid
|
||||
|| is_garbage_text(&md)
|
||||
|| has_template_image
|
||||
|| has_vector_text;
|
||||
let needs_ocr =
|
||||
ocr_reason.is_some() || md.trim().is_empty() || has_gid || is_garbage_text(&md);
|
||||
|
||||
if needs_ocr {
|
||||
pages_needing_ocr.push(page_1idx);
|
||||
@@ -657,80 +556,6 @@ pub fn extract_pages_markdown<P: AsRef<Path>>(
|
||||
extract_pages_markdown_mem(&buffer, pages)
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Structure-tree element extraction (tagged PDFs)
|
||||
// =========================================================================
|
||||
|
||||
/// One structure-tree element reference from a tagged PDF, resolved to a
|
||||
/// page and Marked Content ID.
|
||||
///
|
||||
/// Join `(page, mcid)` against [`TextItem::page`] / [`TextItem::mcid`] from
|
||||
/// [`extract_text_with_positions`] to attach semantic roles (heading levels,
|
||||
/// paragraphs, table cells, …) to extracted text.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct StructureElement {
|
||||
/// 1-indexed page number (matches [`TextItem::page`]).
|
||||
pub page: u32,
|
||||
/// Marked Content ID from the page's content stream (matches
|
||||
/// [`TextItem::mcid`]).
|
||||
pub mcid: i64,
|
||||
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", …).
|
||||
/// Custom tags are resolved through the document's `/RoleMap`; tags
|
||||
/// with no standard mapping are returned verbatim.
|
||||
pub role: String,
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from a tagged PDF in memory.
|
||||
///
|
||||
/// Parses `/StructTreeRoot` (when present) and returns one entry per
|
||||
/// marked-content reference, resolved to its 1-indexed page, MCID, and
|
||||
/// structure type name. Returns an empty list when the PDF is not tagged.
|
||||
///
|
||||
/// Pass `Some(&[...])` with 1-indexed page numbers (matching
|
||||
/// [`TextItem::page`]) to restrict output to those pages; pass `None` for
|
||||
/// the whole document. Entries are sorted by `(page, mcid)`.
|
||||
pub fn extract_structure_elements_mem(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<Vec<StructureElement>, PdfError> {
|
||||
validate_pdf_bytes(buffer)?;
|
||||
let (doc, _page_count) = load_document_from_mem(buffer)?;
|
||||
let Some(tree) = structure_tree::StructTree::from_doc(&doc) else {
|
||||
return Ok(Vec::new());
|
||||
};
|
||||
let page_ids = doc.get_pages();
|
||||
let roles = tree.mcid_to_roles(&page_ids);
|
||||
|
||||
let page_filter: Option<HashSet<u32>> = pages.map(|p| p.iter().copied().collect());
|
||||
let mut elements: Vec<StructureElement> = roles
|
||||
.into_iter()
|
||||
.filter(|(page, _)| page_filter.as_ref().is_none_or(|f| f.contains(page)))
|
||||
.flat_map(|(page, mcids)| {
|
||||
mcids.into_iter().map(move |(mcid, role)| StructureElement {
|
||||
page,
|
||||
mcid,
|
||||
role: role.name().to_string(),
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
elements.sort_unstable_by_key(|e| (e.page, e.mcid));
|
||||
Ok(elements)
|
||||
}
|
||||
|
||||
/// Path-based wrapper for [`extract_structure_elements_mem`].
|
||||
///
|
||||
/// Reads the PDF from disk and extracts structure-tree element references.
|
||||
/// Pass `None` for `pages` to return the whole document, or `Some(&[...])`
|
||||
/// to restrict to specific 1-indexed pages.
|
||||
pub fn extract_structure_elements<P: AsRef<Path>>(
|
||||
path: P,
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<Vec<StructureElement>, PdfError> {
|
||||
validate_pdf_file(&path)?;
|
||||
let buffer = std::fs::read(path.as_ref())?;
|
||||
extract_structure_elements_mem(&buffer, pages)
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Region-based text extraction (for hybrid OCR pipelines)
|
||||
// =========================================================================
|
||||
@@ -1202,12 +1027,7 @@ pub fn extract_tables_in_regions_mem(
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
let detected = tables::detect_tables_with_page_width(
|
||||
&matched,
|
||||
base_font_size,
|
||||
false,
|
||||
items.map_or(1.0, |items| tables::content_width(items)),
|
||||
);
|
||||
let detected = tables::detect_tables(&matched, base_font_size, false);
|
||||
if let Some(candidate) = detected
|
||||
.iter()
|
||||
.find_map(|t| evaluate(TableCandidateSource::Heuristic, t))
|
||||
@@ -3627,7 +3447,6 @@ fn repair_pdf_container_candidates(buf: &[u8]) -> Vec<Vec<u8>> {
|
||||
let mut candidates = Vec::new();
|
||||
|
||||
add_repair_candidate(&mut candidates, append_missing_eof_marker(buf), buf);
|
||||
add_repair_candidate(&mut candidates, recover_startxref_pointer(buf), buf);
|
||||
|
||||
let stripped = strip_leading_pdf_container_bytes(buf);
|
||||
if let Some(stripped_buf) = stripped.as_deref() {
|
||||
@@ -3637,112 +3456,11 @@ fn repair_pdf_container_candidates(buf: &[u8]) -> Vec<Vec<u8>> {
|
||||
append_missing_eof_marker(stripped_buf),
|
||||
buf,
|
||||
);
|
||||
add_repair_candidate(
|
||||
&mut candidates,
|
||||
recover_startxref_pointer(stripped_buf),
|
||||
buf,
|
||||
);
|
||||
}
|
||||
|
||||
candidates
|
||||
}
|
||||
|
||||
/// Some PDF writers emit a `startxref` pointer that doesn't actually point
|
||||
/// at the cross-reference table — a single corrupted byte in the offset is
|
||||
/// enough. lopdf trusts that pointer outright and fails to load rather than
|
||||
/// searching for the real table, unlike pypdf/pdfium which both recover by
|
||||
/// locating it directly. This finds the real (classic, non-stream) `xref`
|
||||
/// table by scanning for the keyword — validating that a plausible
|
||||
/// subsection header follows, not just any standalone "xref" token, since
|
||||
/// this crate processes untrusted input and a coincidental match inside
|
||||
/// unrelated stream/string content must not get "repaired" against a bogus
|
||||
/// offset (lopdf would then load successfully against garbage instead of
|
||||
/// returning a clean error) — and appends a corrected trailing
|
||||
/// `startxref`/`%%EOF` block. lopdf's own `get_xref_start` always uses the
|
||||
/// *last* `%%EOF` in the final 512 bytes of the buffer, so ours
|
||||
/// transparently supersedes the broken one without needing to touch
|
||||
/// anything already in the file.
|
||||
///
|
||||
/// Doesn't cover cross-reference *streams* (`N 0 obj << /Type /XRef ...`,
|
||||
/// used by some PDF 1.5+ writers instead of a classic table) — recovering
|
||||
/// those needs the containing object's number, not just a byte offset.
|
||||
fn recover_startxref_pointer(buf: &[u8]) -> Option<Vec<u8>> {
|
||||
let xref_pos = find_last_valid_xref_table_start(buf)?;
|
||||
|
||||
let mut repaired = Vec::with_capacity(buf.len() + 32);
|
||||
repaired.extend_from_slice(buf);
|
||||
if !repaired.ends_with(b"\n") {
|
||||
repaired.push(b'\n');
|
||||
}
|
||||
repaired.extend_from_slice(format!("startxref\n{xref_pos}\n%%EOF\n").as_bytes());
|
||||
Some(repaired)
|
||||
}
|
||||
|
||||
/// Finds the last standalone `xref` token in `buf` that is immediately
|
||||
/// followed by a plausible classic cross-reference subsection header
|
||||
/// (`<start-id> <count>`, e.g. "0 6") — the shape every real classic xref
|
||||
/// table starts with. A single reverse byte scan: O(n) even on a
|
||||
/// pathological buffer with many non-matching or non-standalone "xref"
|
||||
/// occurrences, unlike repeatedly re-searching a shrinking prefix.
|
||||
fn find_last_valid_xref_table_start(buf: &[u8]) -> Option<usize> {
|
||||
const KEYWORD: &[u8] = b"xref";
|
||||
if buf.len() < KEYWORD.len() {
|
||||
return None;
|
||||
}
|
||||
let mut pos = buf.len() - KEYWORD.len();
|
||||
loop {
|
||||
if &buf[pos..pos + KEYWORD.len()] == KEYWORD {
|
||||
let before_ok = pos == 0 || buf[pos - 1].is_ascii_whitespace();
|
||||
let after_ok = buf
|
||||
.get(pos + KEYWORD.len())
|
||||
.is_none_or(|c| c.is_ascii_whitespace());
|
||||
if before_ok && after_ok && looks_like_xref_subsection_header(buf, pos + KEYWORD.len())
|
||||
{
|
||||
return Some(pos);
|
||||
}
|
||||
}
|
||||
if pos == 0 {
|
||||
return None;
|
||||
}
|
||||
pos -= 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// Checks that `buf[pos..]` starts (after whitespace) with two
|
||||
/// whitespace-separated runs of ASCII digits — `<start-id> <count>`, the
|
||||
/// first subsection header of a classic PDF cross-reference table.
|
||||
fn looks_like_xref_subsection_header(buf: &[u8], pos: usize) -> bool {
|
||||
fn skip_ws(buf: &[u8], mut pos: usize) -> usize {
|
||||
while buf.get(pos).is_some_and(u8::is_ascii_whitespace) {
|
||||
pos += 1;
|
||||
}
|
||||
pos
|
||||
}
|
||||
fn skip_digits(buf: &[u8], mut pos: usize) -> usize {
|
||||
while buf.get(pos).is_some_and(u8::is_ascii_digit) {
|
||||
pos += 1;
|
||||
}
|
||||
pos
|
||||
}
|
||||
|
||||
let pos = skip_ws(buf, pos);
|
||||
let after_first_digits = skip_digits(buf, pos);
|
||||
if after_first_digits == pos {
|
||||
return false; // no start-id
|
||||
}
|
||||
let sep = skip_ws(buf, after_first_digits);
|
||||
if sep == after_first_digits {
|
||||
return false; // start-id and count must be whitespace-separated
|
||||
}
|
||||
let after_count = skip_digits(buf, sep);
|
||||
if after_count == sep {
|
||||
return false; // no count
|
||||
}
|
||||
// The count run must end at whitespace/buffer-end, not run into trailing
|
||||
// garbage (e.g. a coincidental "xref\n0 6garbage" in stream content).
|
||||
buf.get(after_count).is_none_or(u8::is_ascii_whitespace)
|
||||
}
|
||||
|
||||
fn add_repair_candidate(
|
||||
candidates: &mut Vec<Vec<u8>>,
|
||||
candidate: Option<Vec<u8>>,
|
||||
@@ -3808,7 +3526,7 @@ fn process_document(
|
||||
doc: Document,
|
||||
page_count: u32,
|
||||
options: PdfOptions,
|
||||
start: ProcessingTimer,
|
||||
start: std::time::Instant,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
// Step 1 — Detection (cheap: scans content streams for text operators)
|
||||
let detection = detector::detect_from_document(&doc, page_count, &options.detection)?;
|
||||
@@ -3824,7 +3542,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3840,7 +3558,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3853,10 +3571,7 @@ fn process_document(
|
||||
// Step 2 — Extraction (reuses the already-loaded document)
|
||||
let extracted = {
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
// Most page-filtered requests extract only the selected pages. Gather
|
||||
// other pages only when a selected contextual folio needs cross-page
|
||||
// evidence; failures on those context-only pages are non-fatal.
|
||||
let result = extractor::extract_positioned_text_with_folio_context(
|
||||
let result = extractor::extract_positioned_text_from_doc(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3867,19 +3582,9 @@ fn process_document(
|
||||
// This unlocks OCR text layers behind scanned images.
|
||||
if pdf_type == PdfType::Mixed {
|
||||
if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) {
|
||||
let sample: String = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&item.page))
|
||||
})
|
||||
.take(200)
|
||||
.map(|item| item.text.as_str())
|
||||
.collect();
|
||||
let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect();
|
||||
if is_garbage_text(&sample) || sample.trim().is_empty() {
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3889,7 +3594,7 @@ fn process_document(
|
||||
}
|
||||
} else {
|
||||
// Normal extraction failed — try invisible as fallback
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3953,13 +3658,6 @@ fn process_document(
|
||||
let mut garbage_pages: std::collections::HashSet<u32> =
|
||||
std::collections::HashSet::new();
|
||||
for &pg in &ocr_set {
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|filter| !filter.contains(&pg))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let page_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.page == pg)
|
||||
@@ -3999,45 +3697,9 @@ fn process_document(
|
||||
}
|
||||
};
|
||||
|
||||
let selected_page = |page: u32| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&page))
|
||||
};
|
||||
let rects: Vec<_> = rects
|
||||
.into_iter()
|
||||
.filter(|rect| selected_page(rect.page))
|
||||
.collect();
|
||||
let lines: Vec<_> = lines
|
||||
.into_iter()
|
||||
.filter(|line| selected_page(line.page))
|
||||
.collect();
|
||||
let gid_encoded_pages: HashSet<_> = gid_encoded_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
let FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
} = select_items_with_document_folio_context(
|
||||
items,
|
||||
page_count,
|
||||
options.page_filter.as_ref(),
|
||||
);
|
||||
|
||||
let text_quality = analyze_text_quality(&items);
|
||||
merge_ocr_reasons(&mut ocr_reasons_by_page, text_quality.reasons_by_page);
|
||||
let chart_regions = markdown::chart_regions_by_page(&items, &rects, &lines);
|
||||
let layout = compute_layout_complexity_with_chart_regions(
|
||||
&items,
|
||||
&layout_items,
|
||||
&rects,
|
||||
&lines,
|
||||
&chart_regions,
|
||||
);
|
||||
let layout = compute_layout_complexity(&items, &rects, &lines);
|
||||
|
||||
let md = if options.mode == ProcessMode::Analyze {
|
||||
None
|
||||
@@ -4047,15 +3709,9 @@ fn process_document(
|
||||
options.markdown,
|
||||
&rects,
|
||||
&lines,
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: struct_roles.as_ref(),
|
||||
struct_tables: &struct_tables,
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(removal_mask.as_slice()),
|
||||
precomputed_chart_regions: Some(&chart_regions),
|
||||
},
|
||||
&page_thresholds,
|
||||
struct_roles.as_ref(),
|
||||
&struct_tables,
|
||||
))
|
||||
};
|
||||
|
||||
@@ -4168,7 +3824,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: {
|
||||
// Detector reasons (scanned / no_text / vector_text / garbled) merged
|
||||
@@ -5814,70 +5470,11 @@ mod looks_like_partial_table_tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct FolioFilteredItems {
|
||||
items: Vec<types::TextItem>,
|
||||
layout_items: Vec<types::TextItem>,
|
||||
removal_mask: Vec<bool>,
|
||||
removed_pages: HashSet<u32>,
|
||||
}
|
||||
|
||||
/// Resolve folios with complete document context, then select the caller's
|
||||
/// requested pages without losing those decisions.
|
||||
fn select_items_with_document_folio_context(
|
||||
all_items: Vec<types::TextItem>,
|
||||
page_count: u32,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> FolioFilteredItems {
|
||||
let (all_layout_items, all_removed_pages, all_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
let selected_page = |page: u32| page_filter.is_none_or(|filter| filter.contains(&page));
|
||||
|
||||
let (items, removal_mask) = all_items
|
||||
.into_iter()
|
||||
.zip(all_removal_mask)
|
||||
.filter(|(item, _)| selected_page(item.page))
|
||||
.unzip();
|
||||
let layout_items = all_layout_items
|
||||
.into_iter()
|
||||
.filter(|item| selected_page(item.page))
|
||||
.collect();
|
||||
let removed_pages = all_removed_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
|
||||
FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
}
|
||||
}
|
||||
|
||||
/// Analyse extracted items and rects for layout complexity.
|
||||
#[cfg(test)]
|
||||
fn compute_layout_complexity(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
) -> LayoutComplexity {
|
||||
let page_chart_regions = markdown::chart_regions_by_page(items, rects, lines);
|
||||
compute_layout_complexity_with_chart_regions(
|
||||
items,
|
||||
column_items,
|
||||
rects,
|
||||
lines,
|
||||
&page_chart_regions,
|
||||
)
|
||||
}
|
||||
|
||||
fn compute_layout_complexity_with_chart_regions(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
page_chart_regions: &markdown::PageChartRegions,
|
||||
) -> LayoutComplexity {
|
||||
use markdown::analysis::calculate_font_stats_from_items;
|
||||
|
||||
@@ -5897,12 +5494,7 @@ fn compute_layout_complexity_with_chart_regions(
|
||||
|
||||
// Check for side-by-side layout
|
||||
let owned_items: Vec<types::TextItem> = page_items.iter().map(|i| (*i).clone()).collect();
|
||||
let page_content_width = tables::content_width(&owned_items);
|
||||
let bands = markdown::split_side_by_side(&owned_items);
|
||||
let chart_regions = page_chart_regions
|
||||
.get(&page)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or_default();
|
||||
|
||||
let band_ranges: Vec<(f32, f32)> = if bands.is_empty() {
|
||||
// Single region — use sentinel range that includes everything
|
||||
@@ -5917,8 +5509,7 @@ fn compute_layout_complexity_with_chart_regions(
|
||||
let band_items: Vec<types::TextItem> = owned_items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
(x_lo == f32::MIN || (item.x >= x_lo - margin && item.x < x_hi + margin))
|
||||
&& !markdown::item_is_in_chart_region(item, chart_regions)
|
||||
x_lo == f32::MIN || (item.x >= x_lo - margin && item.x < x_hi + margin)
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
@@ -5953,12 +5544,7 @@ fn compute_layout_complexity_with_chart_regions(
|
||||
break;
|
||||
}
|
||||
// Heuristic fallback for borderless tables
|
||||
let heuristic_tables = tables::detect_tables_with_page_width(
|
||||
&band_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
let heuristic_tables = tables::detect_tables(&band_items, base_size, false);
|
||||
if has_data_table(&heuristic_tables) {
|
||||
found_table = true;
|
||||
break;
|
||||
@@ -5970,20 +5556,8 @@ fn compute_layout_complexity_with_chart_regions(
|
||||
}
|
||||
|
||||
let mut pages_with_columns: Vec<u32> = Vec::new();
|
||||
for &page in &seen_pages {
|
||||
let chart_regions = page_chart_regions
|
||||
.get(&page)
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or_default();
|
||||
let page_column_items: Vec<types::TextItem> = column_items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
item.page == page && !markdown::item_is_in_chart_region(item, chart_regions)
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
let cols =
|
||||
extractor::detect_columns(&page_column_items, page, pages_with_tables.contains(&page));
|
||||
for page in seen_pages {
|
||||
let cols = extractor::detect_columns(items, page, pages_with_tables.contains(&page));
|
||||
if cols.len() >= 2 {
|
||||
pages_with_columns.push(page);
|
||||
}
|
||||
@@ -6195,128 +5769,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sparse_folios_leave_no_layout_evidence() {
|
||||
let items = vec![
|
||||
test_item("1", 25.0, 20.0, 12.0, 10.0),
|
||||
test_item("2", 520.0, 60.0, 12.0, 10.0),
|
||||
];
|
||||
let (filtered, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(items.clone(), 1);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let filtered = compute_layout_complexity(&items, &filtered, &[], &[]);
|
||||
|
||||
assert!(!filtered.is_complex);
|
||||
assert!(filtered.pages_with_tables.is_empty());
|
||||
assert!(filtered.pages_with_columns.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dense_chart_panel_is_not_reported_as_a_table() {
|
||||
let mut items: Vec<TextItem> = (0..8)
|
||||
.flat_map(|row| {
|
||||
(0..6).map(move |column| {
|
||||
test_item(
|
||||
&format!("{}", row * 10 + column),
|
||||
105.0 + column as f32 * 35.0,
|
||||
525.0 - row as f32 * 15.0,
|
||||
24.0,
|
||||
10.0,
|
||||
)
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
for row in 0..6 {
|
||||
items.push(test_item(
|
||||
"Left column prose continues here",
|
||||
80.0,
|
||||
320.0 - row as f32 * 15.0,
|
||||
160.0,
|
||||
10.0,
|
||||
));
|
||||
items.push(test_item(
|
||||
"Right column prose continues here",
|
||||
300.0,
|
||||
320.0 - row as f32 * 15.0,
|
||||
160.0,
|
||||
10.0,
|
||||
));
|
||||
}
|
||||
let mut lines: Vec<PdfLine> = (0..30)
|
||||
.map(|column| PdfLine {
|
||||
x1: 100.0 + column as f32 * 8.0,
|
||||
y1: 400.0,
|
||||
x2: 100.0 + column as f32 * 8.0,
|
||||
y2: 550.0,
|
||||
page: 1,
|
||||
})
|
||||
.collect();
|
||||
lines.extend((0..6).map(|row| PdfLine {
|
||||
x1: 100.0,
|
||||
y1: 400.0 + row as f32 * 30.0,
|
||||
x2: 332.0,
|
||||
y2: 400.0 + row as f32 * 30.0,
|
||||
page: 1,
|
||||
}));
|
||||
let rects = vec![PdfRect {
|
||||
x: 80.0,
|
||||
y: 350.0,
|
||||
width: 280.0,
|
||||
height: 240.0,
|
||||
page: 1,
|
||||
}];
|
||||
|
||||
let complexity = compute_layout_complexity(&items, &items, &rects, &lines);
|
||||
|
||||
assert!(complexity.pages_with_tables.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_selection_keeps_document_wide_folio_layout_decisions() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
for row in 0..8 {
|
||||
let y = 20.0 + row as f32 * 8.0;
|
||||
let mut folio = test_item(&(row * 10 + page).to_string(), 25.0, y, 12.0, 10.0);
|
||||
folio.page = page;
|
||||
let mut footer =
|
||||
test_item(&format!("Footer row {row} summary"), 43.0, y, 470.0, 10.0);
|
||||
footer.page = page;
|
||||
let mut body = test_item(&format!("Body{row}"), 530.0, y, 55.0, 10.0);
|
||||
body.page = page;
|
||||
items.extend([folio, footer, body]);
|
||||
}
|
||||
}
|
||||
|
||||
let page_one_items: Vec<_> = items
|
||||
.iter()
|
||||
.filter(|item| item.page == 1)
|
||||
.cloned()
|
||||
.collect();
|
||||
let (page_local_layout, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(page_one_items.clone(), 4);
|
||||
let page_local = compute_layout_complexity(&page_one_items, &page_local_layout, &[], &[]);
|
||||
assert!(
|
||||
page_local.pages_with_columns.contains(&1),
|
||||
"fixture must reproduce page-local folio column evidence"
|
||||
);
|
||||
|
||||
let selected =
|
||||
select_items_with_document_folio_context(items, 4, Some(&HashSet::from([1])));
|
||||
assert_eq!(
|
||||
selected
|
||||
.removal_mask
|
||||
.iter()
|
||||
.filter(|remove| **remove)
|
||||
.count(),
|
||||
8
|
||||
);
|
||||
let document_wide =
|
||||
compute_layout_complexity(&selected.items, &selected.layout_items, &[], &[]);
|
||||
assert!(!document_wide.pages_with_columns.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_detect_encoding_issues_fffd() {
|
||||
assert!(detect_encoding_issues(
|
||||
@@ -7275,65 +6727,4 @@ mod tests {
|
||||
// Pre-filled cell was not touched.
|
||||
assert_eq!(cells[1].text, "Pre-filled");
|
||||
}
|
||||
|
||||
// -- recover_startxref_pointer / find_last_valid_xref_table_start ------
|
||||
//
|
||||
// Direct unit tests on the byte-level scan, addressing review feedback
|
||||
// on #230: a coincidental standalone "xref" token that isn't actually
|
||||
// followed by a subsection header (start-id + count) must not be
|
||||
// treated as a real table — accepting it would let lopdf "succeed"
|
||||
// against a bogus offset and silently return garbled/empty content
|
||||
// instead of a clean error.
|
||||
|
||||
#[test]
|
||||
fn find_xref_rejects_standalone_token_without_subsection_header() {
|
||||
// "xref" appears as a real standalone word, but nothing that looks
|
||||
// like "<start-id> <count>" follows it.
|
||||
let buf = b"Please refer to the xref appendix for details.";
|
||||
assert_eq!(find_last_valid_xref_table_start(buf), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_xref_accepts_real_classic_table_header() {
|
||||
let buf = b"garbage\nxref\n0 6\n0000000000 65535 f \n%%EOF";
|
||||
let pos = find_last_valid_xref_table_start(buf).expect("should find the real table");
|
||||
assert_eq!(&buf[pos..pos + 4], b"xref");
|
||||
assert_eq!(&buf[pos..], b"xref\n0 6\n0000000000 65535 f \n%%EOF");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_xref_skips_coincidental_match_and_finds_real_table_before_it() {
|
||||
// A coincidental "xref" (no subsection header) appears *after* the
|
||||
// real table in the buffer — the scan must not stop at the first
|
||||
// (rightmost) standalone token it finds; it must keep looking
|
||||
// backward until one actually validates.
|
||||
let buf = b"xref\n0 3\n0000000000 65535 f \ntrailer\nsee the xref\n";
|
||||
let pos = find_last_valid_xref_table_start(buf).expect("should find the real table");
|
||||
assert_eq!(pos, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_xref_rejects_substring_of_startxref() {
|
||||
// "xref" is a substring of "startxref" but isn't a standalone
|
||||
// token there (not preceded by whitespace) — must not match, even
|
||||
// though a number immediately follows it.
|
||||
let buf = b"startxref\n1234\n%%EOF";
|
||||
assert_eq!(find_last_valid_xref_table_start(buf), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_xref_rejects_count_run_with_trailing_garbage() {
|
||||
// "xref\n0 6garbage" has the right shape (digits, whitespace,
|
||||
// digits) but the count run doesn't end at whitespace/EOF — it
|
||||
// runs straight into non-digit garbage, so this must not be
|
||||
// accepted as a real subsection header.
|
||||
let buf = b"xref\n0 6garbage\n%%EOF";
|
||||
assert_eq!(find_last_valid_xref_table_start(buf), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recover_startxref_pointer_returns_none_without_a_valid_table() {
|
||||
let buf = b"Please refer to the xref appendix for details.";
|
||||
assert!(recover_startxref_pointer(buf).is_none());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -171,74 +171,6 @@ pub(crate) fn is_toc_marker_heading(text: &str) -> bool {
|
||||
/// equation and absent from name-plus-number headings. A bare trailing colon
|
||||
/// is NOT a fragment signal either: real headings frequently end with colons
|
||||
/// ("Procedure:", "Steps for Using the Microscope:").
|
||||
/// True when the line opens with a section number ("3.", "2.1.4", "IV)").
|
||||
///
|
||||
/// Mirrors the acceptance of `heading::parse_numbering` rather than the
|
||||
/// stricter `convert::starts_with_section_number`, which deliberately
|
||||
/// requires two components because it bypasses isolation checks. Here a
|
||||
/// single "1." counts: numbering is independent evidence of a heading, and
|
||||
/// `heading.rs` applies its numbered-prefix allowance *after* consulting
|
||||
/// `is_heading_fragment`, so without this exemption a numbered
|
||||
/// sentence-case heading would be vetoed before that allowance can run.
|
||||
fn starts_with_numbering_prefix(t: &str) -> bool {
|
||||
let Some(first) = t.split_whitespace().next() else {
|
||||
return false;
|
||||
};
|
||||
let has_delimiter = first.ends_with(['.', ')', ':']);
|
||||
let token = first.trim_end_matches(['.', ')', ':']);
|
||||
if token.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let parts: Vec<&str> = token.split('.').collect();
|
||||
let decimal = parts
|
||||
.iter()
|
||||
.all(|p| !p.is_empty() && p.len() <= 3 && p.chars().all(|c| c.is_ascii_digit()));
|
||||
if decimal {
|
||||
// "1." / "2.1." carry a delimiter; "2.3 Title" is written without
|
||||
// one, so a multi-component number is accepted bare. A bare single
|
||||
// number ("3 apples") is not — that is ordinary prose.
|
||||
return has_delimiter || parts.len() >= 2;
|
||||
}
|
||||
// Roman numerals go through the heading parser's own grammar so the two
|
||||
// agree: uppercase I/V/X/L/C only, at most 8 characters. A looser rule
|
||||
// here would exempt markers the parser rejects — "iv)" or "d)" from an
|
||||
// alphabetical list — letting an ordinary list item bypass the veto and
|
||||
// reach heading promotion.
|
||||
//
|
||||
// A delimiter is also required: a bare leading "I" is the pronoun far
|
||||
// more often than a section number.
|
||||
has_delimiter && crate::markdown::heading::roman_value(token).is_some()
|
||||
}
|
||||
|
||||
/// True when the line reads as a title rather than a sentence: every
|
||||
/// content word (ignoring minor words) starts uppercase. Used to spare real
|
||||
/// headings from the dangling-verb veto — "Bond Yields" is a section title,
|
||||
/// "the method yields" is a stranded clause, and only the casing tells them
|
||||
/// apart.
|
||||
fn looks_title_case(t: &str) -> bool {
|
||||
const MINOR: &[&str] = &[
|
||||
"a", "an", "the", "of", "and", "or", "for", "to", "in", "on", "at", "by", "with", "from",
|
||||
"as", "is", "are", "that", "than", "into",
|
||||
];
|
||||
let mut content = 0usize;
|
||||
let mut capitalized = 0usize;
|
||||
for w in t.split_whitespace() {
|
||||
let cleaned: String = w.chars().filter(|c| c.is_alphabetic()).collect();
|
||||
if cleaned.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if MINOR.contains(&cleaned.to_lowercase().as_str()) {
|
||||
continue;
|
||||
}
|
||||
content += 1;
|
||||
if cleaned.chars().next().is_some_and(char::is_uppercase) {
|
||||
capitalized += 1;
|
||||
}
|
||||
}
|
||||
// A single content word ("Yields") is a title by default.
|
||||
content == 0 || capitalized == content
|
||||
}
|
||||
|
||||
pub(crate) fn is_heading_fragment(text: &str) -> bool {
|
||||
let t = text.trim_end();
|
||||
|
||||
@@ -312,133 +244,9 @@ pub(crate) fn is_heading_fragment(text: &str) -> bool {
|
||||
if t.ends_with(':') && t.split_whitespace().any(is_equation_number) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Dangling clause: a stranded sentence lead-in ends on a relational
|
||||
// verb with no terminal punctuation — "Note that the exact error equals"
|
||||
// left ahead of its formula when a phantom table dissolved.
|
||||
//
|
||||
// Gated on the line reading as prose rather than a title. Case is the
|
||||
// discriminator the trailing word alone cannot provide: a heading is
|
||||
// title case ("Bond Yields", "The Method Yields") while a stranded
|
||||
// lead-in is sentence case ("the method yields"). Without this gate the
|
||||
// veto eats real headings — "Bond Yields", "Crop Yields" and any wrapped
|
||||
// title-case heading the preprocessor failed to merge.
|
||||
if !t.ends_with(['.', '!', '?', ':', ';', ')', ']'])
|
||||
&& !looks_title_case(t)
|
||||
&& !starts_with_numbering_prefix(t)
|
||||
{
|
||||
if let Some(last) = t.split_whitespace().next_back() {
|
||||
let word: String = last
|
||||
.trim_matches(|c: char| !c.is_alphanumeric())
|
||||
.to_lowercase();
|
||||
// Relational verbs only, and only those with no common noun
|
||||
// sense. "yields" was dropped for exactly that reason: "Bond
|
||||
// Yields" is a real section title. Function words, copulas and
|
||||
// auxiliaries were measured and rejected outright — a heading
|
||||
// that wraps across lines ends on those, and suppressing them
|
||||
// destroyed real IRS Publication 17 headings.
|
||||
const DANGLING_TAIL: &[&str] =
|
||||
&["equals", "denotes", "implies", "satisfies", "signifies"];
|
||||
if DANGLING_TAIL.contains(&word.as_str()) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod fragment_heading_tests {
|
||||
use super::is_heading_fragment;
|
||||
|
||||
#[test]
|
||||
fn dangling_tail_marks_stranded_clause() {
|
||||
// opendataloader 01030000000144: left behind when a phantom table
|
||||
// dissolved, ahead of its formula on the next line.
|
||||
assert!(is_heading_fragment("Note that the exact error equals"));
|
||||
assert!(is_heading_fragment("The remainder term satisfies"));
|
||||
assert!(is_heading_fragment("we conclude that the sum equals"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn real_headings_survive() {
|
||||
assert!(!is_heading_fragment("Introduction"));
|
||||
assert!(!is_heading_fragment("Error Analysis"));
|
||||
assert!(!is_heading_fragment("Materials and Methods"));
|
||||
assert!(!is_heading_fragment("Results"));
|
||||
assert!(!is_heading_fragment("3.2 Richardson Extrapolation"));
|
||||
assert!(!is_heading_fragment("Discussion and Conclusions"));
|
||||
// Terminal punctuation means the clause is complete.
|
||||
assert!(!is_heading_fragment("What is a Derivative?"));
|
||||
assert!(!is_heading_fragment("Procedure:"));
|
||||
assert!(!is_heading_fragment("Note that this is important."));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn title_case_headings_ending_in_a_verb_survive() {
|
||||
// "yields" is also a plural noun; these are real section titles.
|
||||
assert!(!is_heading_fragment("Bond Yields"));
|
||||
assert!(!is_heading_fragment("Crop Yields"));
|
||||
assert!(!is_heading_fragment("Dividend Yields"));
|
||||
assert!(!is_heading_fragment("Yields"));
|
||||
// A wrapped title-case heading whose first line ends on a listed
|
||||
// verb must survive even if the preprocessor failed to merge it.
|
||||
assert!(!is_heading_fragment("The Theorem Implies"));
|
||||
assert!(!is_heading_fragment("What This Denotes"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numbered_sentence_case_headings_survive() {
|
||||
// heading.rs consults is_heading_fragment BEFORE applying its
|
||||
// numbered-prefix allowance, so the veto must not pre-empt it.
|
||||
assert!(!is_heading_fragment("1. What the model implies"));
|
||||
assert!(!is_heading_fragment("2.3 How the estimator satisfies"));
|
||||
assert!(!is_heading_fragment("IV) What this denotes"));
|
||||
// Without numbering the same wording is still a stranded clause.
|
||||
assert!(is_heading_fragment("What the model implies"));
|
||||
// A bare leading number or pronoun is prose, not numbering.
|
||||
assert!(is_heading_fragment("3 apples and what that implies"));
|
||||
assert!(is_heading_fragment("I think the model implies"));
|
||||
// Markers heading::parse_numbering rejects must not be exempted
|
||||
// either, or an ordinary list item bypasses the veto: lowercase
|
||||
// roman, alphabetical markers, and over-long tokens.
|
||||
assert!(is_heading_fragment("iv) the estimator satisfies"));
|
||||
assert!(is_heading_fragment("d) the value implies"));
|
||||
// Unsupported character (M is outside the parser's I/V/X/L/C set).
|
||||
assert!(is_heading_fragment("MMMM. the value implies"));
|
||||
// Over-long token: nine valid characters, so this exercises the
|
||||
// 8-character bound rather than the character set.
|
||||
assert!(is_heading_fragment("IIIIIIIII. the value implies"));
|
||||
// Eight is still within the bound and stays exempt.
|
||||
assert!(!is_heading_fragment("IIIIIIII. What this implies"));
|
||||
// Uppercase roman within the parser's grammar is still exempt.
|
||||
assert!(!is_heading_fragment("IV. What this denotes"));
|
||||
assert!(!is_heading_fragment("XII) What this implies"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wrapped_headings_are_not_fragments() {
|
||||
// A heading that wraps across lines ends on a function word. These
|
||||
// are real headings from IRS Publication 17 and must survive.
|
||||
assert!(!is_heading_fragment("Casualty and"));
|
||||
assert!(!is_heading_fragment("Rule 10. You Must Be at"));
|
||||
assert!(!is_heading_fragment("Higher Standard Deduction for"));
|
||||
assert!(!is_heading_fragment("Qualifying Child of"));
|
||||
assert!(!is_heading_fragment("When Can I Withdraw or"));
|
||||
// Copulas and auxiliaries also end real wrapped headings.
|
||||
assert!(!is_heading_fragment("Rule 15. Your AGI Must Be"));
|
||||
assert!(!is_heading_fragment("What Medical Expenses Are"));
|
||||
assert!(!is_heading_fragment("Rule 13. You Must Have"));
|
||||
assert!(!is_heading_fragment("When Can a Roth IRA Be"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dangling_check_is_case_insensitive() {
|
||||
// All-caps is not sentence case, so the veto must not fire there.
|
||||
assert!(!is_heading_fragment("THE REMAINDER EQUALS"));
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute the Y-gap threshold for paragraph break detection.
|
||||
///
|
||||
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
|
||||
|
||||
@@ -127,9 +127,7 @@ fn visual_style(line: &TextLine) -> Option<VisualStyle> {
|
||||
})
|
||||
}
|
||||
|
||||
/// Shared with `analysis::starts_with_numbering_prefix` so the veto
|
||||
/// exemption and the heading parser agree on what a roman numeral is.
|
||||
pub(super) fn roman_value(token: &str) -> Option<u32> {
|
||||
fn roman_value(token: &str) -> Option<u32> {
|
||||
if token.is_empty() || token.len() > 8 {
|
||||
return None;
|
||||
}
|
||||
|
||||
+39
-258
@@ -69,7 +69,7 @@ fn is_chart_adjacent_label(item: &TextItem, region: (f32, f32, f32, f32)) -> boo
|
||||
|| (mostly_inside_chart_width && close_to_chart_edge && category_sized))
|
||||
}
|
||||
|
||||
pub(crate) fn item_is_in_chart_region(item: &TextItem, regions: &[(f32, f32, f32, f32)]) -> bool {
|
||||
fn item_is_in_chart_region(item: &TextItem, regions: &[(f32, f32, f32, f32)]) -> bool {
|
||||
regions.iter().any(|&(x0, y0, x1, y1)| {
|
||||
let cx = item.x + item.width / 2.0;
|
||||
let within_padded_x = cx >= x0 - CHART_REGION_PAD && cx <= x1 + CHART_REGION_PAD;
|
||||
@@ -92,72 +92,6 @@ fn items_outside_chart_regions(
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn merge_chart_regions(
|
||||
regions: impl IntoIterator<Item = (f32, f32, f32, f32)>,
|
||||
) -> Vec<(f32, f32, f32, f32)> {
|
||||
const MERGE_TOLERANCE: f32 = 3.0;
|
||||
|
||||
let mut merged: Vec<(f32, f32, f32, f32)> = Vec::new();
|
||||
for (x0, y0, x1, y1) in regions {
|
||||
let mut current = (x0.min(x1), y0.min(y1), x0.max(x1), y0.max(y1));
|
||||
let mut index = 0;
|
||||
while index < merged.len() {
|
||||
let candidate = merged[index];
|
||||
let overlaps = current.2 + MERGE_TOLERANCE >= candidate.0
|
||||
&& candidate.2 + MERGE_TOLERANCE >= current.0
|
||||
&& current.3 + MERGE_TOLERANCE >= candidate.1
|
||||
&& candidate.3 + MERGE_TOLERANCE >= current.1;
|
||||
if overlaps {
|
||||
current = (
|
||||
current.0.min(candidate.0),
|
||||
current.1.min(candidate.1),
|
||||
current.2.max(candidate.2),
|
||||
current.3.max(candidate.3),
|
||||
);
|
||||
merged.swap_remove(index);
|
||||
} else {
|
||||
index += 1;
|
||||
}
|
||||
}
|
||||
merged.push(current);
|
||||
}
|
||||
merged
|
||||
}
|
||||
|
||||
pub(crate) type PageChartRegions = HashMap<u32, Vec<(f32, f32, f32, f32)>>;
|
||||
|
||||
/// Compute the chart masks used by both layout analysis and Markdown output.
|
||||
///
|
||||
/// Keeping the rect-backed and dense-line heuristics behind one entry point
|
||||
/// ensures metadata and extraction cannot drift when either detector changes.
|
||||
pub(crate) fn chart_regions_by_page(
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[PdfLine],
|
||||
) -> PageChartRegions {
|
||||
let mut page_items: HashMap<u32, Vec<TextItem>> = HashMap::new();
|
||||
for item in items.iter().filter(|item| {
|
||||
matches!(
|
||||
&item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
)
|
||||
}) {
|
||||
page_items.entry(item.page).or_default().push(item.clone());
|
||||
}
|
||||
|
||||
page_items
|
||||
.into_iter()
|
||||
.filter_map(|(page, items)| {
|
||||
let rect_regions = crate::tables::detect_chart_regions(&items, rects, page);
|
||||
let line_regions = crate::tables::detect_dense_line_chart_regions(lines, rects, page)
|
||||
.into_iter()
|
||||
.filter(|®ion| chart_region_separates_prose_columns(&items, region));
|
||||
let regions = merge_chart_regions(rect_regions.into_iter().chain(line_regions));
|
||||
(!regions.is_empty()).then_some((page, regions))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Detect side-by-side table layout by finding a significant X-position gap.
|
||||
///
|
||||
/// Returns X-band boundaries `[(x_min, split_x), (split_x, x_max)]` when a
|
||||
@@ -395,15 +329,6 @@ fn chart_spans_prose_split(region: (f32, f32, f32, f32), split_x: f32) -> bool {
|
||||
split_x - left >= MIN_CHART_WIDTH_PER_SIDE && right - split_x >= MIN_CHART_WIDTH_PER_SIDE
|
||||
}
|
||||
|
||||
pub(crate) fn chart_region_separates_prose_columns(
|
||||
items: &[TextItem],
|
||||
region: (f32, f32, f32, f32),
|
||||
) -> bool {
|
||||
let outside = items_outside_chart_regions(items, &[region]);
|
||||
chart_page_prose_column_split(&outside)
|
||||
.is_some_and(|split_x| chart_spans_prose_split(region, split_x))
|
||||
}
|
||||
|
||||
/// True when adjacent physical rows form an unterminated, lowercase prose
|
||||
/// continuation in the same projected column.
|
||||
fn is_cross_row_prose_continuation(previous: &str, current: &str) -> bool {
|
||||
@@ -1050,58 +975,18 @@ pub fn to_markdown_from_items_with_rects(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
) -> String {
|
||||
let document_page_count = items.iter().map(|item| item.page).max().unwrap_or(0);
|
||||
to_markdown_from_items_with_rects_and_page_count(items, options, rects, document_page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to Markdown with an authoritative PDF page count.
|
||||
///
|
||||
/// Use this overload when the owning PDF is available so trailing blank or
|
||||
/// unextracted pages are included in document-level header and folio coverage.
|
||||
/// Item-only callers can continue using [`to_markdown_from_items_with_rects`],
|
||||
/// which falls back to the highest observed item page.
|
||||
pub fn to_markdown_from_items_with_rects_and_page_count(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
document_page_count: u32,
|
||||
) -> String {
|
||||
to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
options,
|
||||
rects,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages: None,
|
||||
prefiltered_page_number_mask: None,
|
||||
precomputed_chart_regions: None,
|
||||
},
|
||||
&HashMap::new(),
|
||||
None,
|
||||
&[],
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) struct MarkdownDocumentContext<'a> {
|
||||
pub(crate) page_thresholds: &'a HashMap<u32, f32>,
|
||||
pub(crate) struct_roles:
|
||||
Option<&'a HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
pub(crate) struct_tables: &'a [crate::structure_tree::StructTable],
|
||||
pub(crate) page_count: u32,
|
||||
/// Pages where an upstream document-level pass removed folios. This keeps
|
||||
/// table-continuation classification consistent after masked items drop.
|
||||
pub(crate) prefiltered_page_number_pages: Option<&'a HashSet<u32>>,
|
||||
/// Document-level removal decisions aligned with this call's input items.
|
||||
/// Table detection consumes the original items; the mask is applied only
|
||||
/// after table claims have been established.
|
||||
pub(crate) prefiltered_page_number_mask: Option<&'a [bool]>,
|
||||
/// Optional chart masks shared with layout analysis so the geometry is
|
||||
/// detected once and interpreted identically by both pipelines.
|
||||
pub(crate) precomputed_chart_regions: Option<&'a PageChartRegions>,
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown, using rectangles and line segments for table detection.
|
||||
///
|
||||
/// Line-based detection runs first (strongest structural evidence), then rect-based,
|
||||
@@ -1111,52 +996,28 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
pdf_lines: &[crate::types::PdfLine],
|
||||
context: MarkdownDocumentContext<'_>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
struct_roles: Option<&HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
struct_tables: &[crate::structure_tree::StructTable],
|
||||
) -> String {
|
||||
use crate::tables::{
|
||||
content_width, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, detect_tables_with_page_width, try_build_rect_guided_table,
|
||||
detect_tables, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, try_build_rect_guided_table,
|
||||
};
|
||||
use crate::types::ItemType;
|
||||
|
||||
let MarkdownDocumentContext {
|
||||
page_thresholds,
|
||||
struct_roles,
|
||||
struct_tables,
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages,
|
||||
prefiltered_page_number_mask,
|
||||
precomputed_chart_regions,
|
||||
} = context;
|
||||
|
||||
if items.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Table detection must retain the original collection because short
|
||||
// numeric table cells can be indistinguishable from folios until
|
||||
// structural context is available. A precomputed mask carries the
|
||||
// document-wide decision without removing items before table claims.
|
||||
debug_assert!(prefiltered_page_number_mask.is_none_or(|mask| mask.len() == items.len()));
|
||||
let has_precomputed_page_number_mask = prefiltered_page_number_mask.is_some();
|
||||
let removed_page_number_pages = prefiltered_page_number_pages.cloned().unwrap_or_default();
|
||||
|
||||
// Separate images and links from text items
|
||||
let mut images: Vec<TextItem> = Vec::new();
|
||||
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
let mut links: Vec<TextItem> = Vec::new();
|
||||
let mut text_items: Vec<TextItem> = Vec::new();
|
||||
let mut text_item_page_number_mask: Vec<bool> = Vec::new();
|
||||
|
||||
for (input_index, item) in items.into_iter().enumerate() {
|
||||
for item in items {
|
||||
match &item.item_type {
|
||||
ItemType::Image => {
|
||||
page_image_regions.entry(item.page).or_default().push((
|
||||
item.x,
|
||||
item.y,
|
||||
item.x + item.width,
|
||||
item.y + item.height,
|
||||
));
|
||||
if options.include_images {
|
||||
images.push(item);
|
||||
}
|
||||
@@ -1167,12 +1028,6 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
ItemType::Text | ItemType::FormField => {
|
||||
text_item_page_number_mask.push(
|
||||
prefiltered_page_number_mask
|
||||
.and_then(|mask| mask.get(input_index))
|
||||
.copied()
|
||||
.unwrap_or(false),
|
||||
);
|
||||
text_items.push(item);
|
||||
}
|
||||
}
|
||||
@@ -1199,12 +1054,21 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Chart regions per page: their text must not steer column detection
|
||||
// during line grouping (it fills the gutter and fuses two-column lines).
|
||||
let page_chart_map = precomputed_chart_regions
|
||||
.cloned()
|
||||
.unwrap_or_else(|| chart_regions_by_page(&text_items, rects, pdf_lines));
|
||||
let mut page_chart_map: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
for &page in page_groups.keys() {
|
||||
let page_items_ref: Vec<TextItem> = page_groups[&page]
|
||||
.iter()
|
||||
.map(|(_, item)| (*item).clone())
|
||||
.collect();
|
||||
let regions = crate::tables::detect_chart_regions(&page_items_ref, rects, page);
|
||||
if !regions.is_empty() {
|
||||
page_chart_map.insert(page, regions);
|
||||
}
|
||||
}
|
||||
|
||||
let mut pages: Vec<u32> = page_groups.keys().copied().collect();
|
||||
pages.sort();
|
||||
let page_count = pages.last().copied().unwrap_or(0) + 1;
|
||||
|
||||
// Track band splits per page so we can split non-table items later
|
||||
let mut page_band_splits: HashMap<u32, Vec<(f32, f32)>> = HashMap::new();
|
||||
@@ -1216,7 +1080,6 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
for page in pages {
|
||||
let group = page_groups.get(&page).unwrap();
|
||||
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
|
||||
let page_content_width = content_width(&page_items);
|
||||
|
||||
// Chart-bar regions: bar charts drawn as filled rects read as cell
|
||||
// rects or aligned text and get gridded into phantom tables. Their
|
||||
@@ -1258,10 +1121,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
});
|
||||
let chart_prose_columns = chart_prose_split.is_some();
|
||||
|
||||
// Check for side-by-side table layout using the original items. Sparse
|
||||
// numeric cells need table context before they can be distinguished
|
||||
// safely from folios; cleaned evidence is reserved for column and
|
||||
// final non-table layout decisions.
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// A rect table crossing a proposed split boundary means the "gutter"
|
||||
// is really the gap between ruled and borderless table columns —
|
||||
@@ -1501,12 +1361,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// table can share the prose anchors. Reject only candidates
|
||||
// whose cells prove they are parallel prose fragments.
|
||||
let reject_parallel_prose = chart_prose_columns && !was_split;
|
||||
let tables = detect_tables_with_page_width(
|
||||
subset_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
let tables = detect_tables(subset_items, base_size, false);
|
||||
for table in tables {
|
||||
if reject_parallel_prose && is_parallel_prose_table(&table) {
|
||||
log::debug!(
|
||||
@@ -1687,12 +1542,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// and reject chart-page prose candidates individually below.
|
||||
let skip_body_font =
|
||||
merged_retry_skips_body_font(detected_columns, !chart_regions.is_empty());
|
||||
let heuristic_tables = detect_tables_with_page_width(
|
||||
&chart_free,
|
||||
base_size,
|
||||
skip_body_font,
|
||||
page_content_width,
|
||||
);
|
||||
let heuristic_tables = detect_tables(&chart_free, base_size, skip_body_font);
|
||||
for table in &heuristic_tables {
|
||||
if !chart_regions.is_empty() && is_parallel_prose_table(table) {
|
||||
log::debug!(
|
||||
@@ -1777,20 +1627,16 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
};
|
||||
|
||||
// Filter out table items and process the rest
|
||||
let non_table_items: Vec<(usize, TextItem)> = text_items
|
||||
let non_table_items: Vec<TextItem> = text_items
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.filter(|(idx, _)| !table_items.contains(idx))
|
||||
.map(|(_, item)| item)
|
||||
.collect();
|
||||
|
||||
// Find pages that are table-only (no remaining non-table text)
|
||||
let table_only_pages: HashSet<u32> = {
|
||||
let mut pages_with_text: HashSet<u32> =
|
||||
non_table_items.iter().map(|(_, item)| item.page).collect();
|
||||
// Preserve the pre-filter continuation classification: a page that
|
||||
// originally also contained a folio does not become table-only merely
|
||||
// because an upstream document-level pass removed it.
|
||||
pages_with_text.extend(removed_page_number_pages);
|
||||
let pages_with_text: HashSet<u32> = non_table_items.iter().map(|i| i.page).collect();
|
||||
page_tables
|
||||
.keys()
|
||||
.filter(|p| !pages_with_text.contains(p))
|
||||
@@ -1805,30 +1651,15 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// column detection on pages where table column gaps would be misidentified.
|
||||
let table_page_set: HashSet<u32> = page_tables.keys().copied().collect();
|
||||
|
||||
let non_table_items = if has_precomputed_page_number_mask {
|
||||
non_table_items
|
||||
.into_iter()
|
||||
.filter(|(index, _)| !text_item_page_number_mask[*index])
|
||||
.map(|(_, item)| item)
|
||||
.collect()
|
||||
} else {
|
||||
crate::extractor::filter_markdown_page_numbers_with_removed_pages(
|
||||
non_table_items.into_iter().map(|(_, item)| item).collect(),
|
||||
document_page_count,
|
||||
)
|
||||
.0
|
||||
};
|
||||
|
||||
// Split non-table items by band boundaries before line grouping so that
|
||||
// items from different side-by-side zones (e.g. left/right month columns
|
||||
// in a calendar) don't merge into the same line.
|
||||
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
non_table_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
)
|
||||
} else {
|
||||
// Separate items into physical-band pages, chart/prose pages, and
|
||||
@@ -1851,14 +1682,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
// Process unsplit pages normally
|
||||
let mut all_lines =
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
);
|
||||
// Process each split page's bands independently, then interleave
|
||||
// by Y position so paired zones (e.g. left/right months) appear together.
|
||||
let mut split_pages: Vec<u32> = split_page_items.keys().copied().collect();
|
||||
@@ -1876,7 +1705,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !band_items.is_empty() {
|
||||
page_lines.extend(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
band_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1910,7 +1739,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !column_items.is_empty() {
|
||||
zone_lines.extend(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
column_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1951,7 +1780,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
item.y >= low || item_is_in_chart_region(item, chart_regions)
|
||||
});
|
||||
all_lines.extend(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
chart_zone,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1970,7 +1799,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Strip repeated headers/footers before conversion
|
||||
let lines = if options.strip_headers_footers {
|
||||
preprocess::strip_repeated_lines(lines, document_page_count)
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
} else {
|
||||
lines
|
||||
};
|
||||
@@ -2083,54 +1912,6 @@ mod tests {
|
||||
it
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn precomputed_folio_mask_preserves_numeric_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for column in 0..2 {
|
||||
let mut item = make_item_w(
|
||||
110.0 + column as f32 * 100.0,
|
||||
30.0 + row as f32 * 20.0,
|
||||
20.0,
|
||||
1,
|
||||
);
|
||||
item.text = (row * 2 + column + 1).to_string();
|
||||
items.push(item);
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + column as f32 * 100.0,
|
||||
y: 20.0 + row as f32 * 20.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Simulate document-level folio decisions that would remove every
|
||||
// short numeric item if applied before structural table detection.
|
||||
let removal_mask = vec![true; items.len()];
|
||||
let removed_pages = HashSet::from([1]);
|
||||
let markdown = to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
MarkdownOptions::default(),
|
||||
&rects,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: 1,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(&removal_mask),
|
||||
precomputed_chart_regions: None,
|
||||
},
|
||||
);
|
||||
|
||||
assert!(markdown.contains("|1|2|"), "{markdown}");
|
||||
assert!(markdown.contains("|7|8|"), "{markdown}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn early_layout_excludes_chart_items_before_column_detection() {
|
||||
let mut items = Vec::new();
|
||||
|
||||
+64
-29
@@ -3,7 +3,6 @@
|
||||
use regex::Regex;
|
||||
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
use crate::text_utils::is_page_number_line;
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
@@ -146,7 +145,7 @@ fn fix_hyphenation(text: &str) -> String {
|
||||
result
|
||||
}
|
||||
|
||||
/// Remove isolated page-number expressions from Markdown.
|
||||
/// Remove standalone page numbers (lines that are just 1-4 digit numbers)
|
||||
fn remove_page_numbers(text: &str) -> String {
|
||||
let mut result = Vec::new();
|
||||
let lines: Vec<&str> = text.lines().collect();
|
||||
@@ -184,6 +183,69 @@ fn remove_page_numbers(text: &str) -> String {
|
||||
result.join("\n")
|
||||
}
|
||||
|
||||
/// Check if a line looks like a page number
|
||||
fn is_page_number_line(trimmed: &str) -> bool {
|
||||
// Empty lines are not page numbers
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Pattern 1: Just a number (1-4 digits)
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Pattern 2: "Page X of Y" or "Page X" or "Page of" (placeholder)
|
||||
let lower = trimmed.to_lowercase();
|
||||
if let Some(rest) = lower.strip_prefix("page") {
|
||||
let rest = rest.trim();
|
||||
// "Page of" (empty page numbers)
|
||||
if rest == "of" || rest.starts_with("of ") {
|
||||
return true;
|
||||
}
|
||||
// "Page X" or "Page X of Y"
|
||||
if rest
|
||||
.chars()
|
||||
.next()
|
||||
.map(|c| c.is_ascii_digit())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Just "Page" followed by whitespace and maybe "of"
|
||||
if rest.is_empty()
|
||||
|| rest
|
||||
.split_whitespace()
|
||||
.all(|w| w == "of" || w.chars().all(|c| c.is_ascii_digit()))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 3: "X of Y" where X and Y are numbers
|
||||
if let Some(of_idx) = trimmed.find(" of ") {
|
||||
let before = trimmed[..of_idx].trim();
|
||||
let after = trimmed[of_idx + 4..].trim();
|
||||
if before.chars().all(|c| c.is_ascii_digit())
|
||||
&& after.chars().all(|c| c.is_ascii_digit())
|
||||
&& !before.is_empty()
|
||||
&& !after.is_empty()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 4: "- X -" centered page number
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if inner.chars().all(|c| c.is_ascii_digit()) && !inner.is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Convert URLs to markdown links
|
||||
fn format_urls(text: &str) -> String {
|
||||
use once_cell::sync::Lazy;
|
||||
@@ -427,14 +489,12 @@ mod tests {
|
||||
fn test_is_page_number_page_x() {
|
||||
assert!(is_page_number_line("Page 5"));
|
||||
assert!(is_page_number_line("page 12"));
|
||||
assert!(is_page_number_line("Page123"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_page_x_of_y() {
|
||||
assert!(is_page_number_line("Page 3 of 10"));
|
||||
assert!(is_page_number_line("page 1 of 5"));
|
||||
assert!(is_page_number_line("Page 3 of 10 Report header"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -464,13 +524,6 @@ mod tests {
|
||||
assert!(!is_page_number_line("Hello World"));
|
||||
assert!(!is_page_number_line("Chapter 1"));
|
||||
assert!(!is_page_number_line("Total: 500"));
|
||||
assert!(!is_page_number_line("PAGE0-PARA2-END-MARKER-0"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_labeled_running_header() {
|
||||
assert!(is_page_number_line("Page 42 Chapter 5"));
|
||||
assert!(is_page_number_line("Page 42 explains the result"));
|
||||
}
|
||||
|
||||
// --- remove_page_numbers ---
|
||||
@@ -498,24 +551,6 @@ mod tests {
|
||||
assert!(result.contains("42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_labeled_header_with_content() {
|
||||
let input = "Content\n\nPage 42 explains the result\n---\nEnd";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert!(!result.contains("Page 42 explains the result"));
|
||||
assert!(result.contains("Content"));
|
||||
assert!(result.contains("End"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_preserves_page_prefixed_content() {
|
||||
let input = "PAGE0-PARA2-START substantive report text PAGE0-PARA2-END-MARKER-0";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert_eq!(result, input);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_multiple_patterns() {
|
||||
let input = "\n1\n\nContent\n\n2\n\n---\nMore\n\n3\n";
|
||||
|
||||
@@ -271,12 +271,6 @@ pub struct PyTextItem {
|
||||
pub is_strikeout: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator, None
|
||||
/// when the text is not part of marked content. Join with the
|
||||
/// (page, mcid) pairs from extract_structure_elements to attach
|
||||
/// structure-tree roles (headings, paragraphs, ...) in tagged PDFs.
|
||||
#[pyo3(get)]
|
||||
pub mcid: Option<i64>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -292,32 +286,6 @@ impl PyTextItem {
|
||||
}
|
||||
}
|
||||
|
||||
/// One structure-tree element reference from a tagged PDF.
|
||||
#[pyclass(name = "StructureElement")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyStructureElement {
|
||||
/// 1-indexed page number (matches TextItem.page).
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Marked Content ID from the page's content stream (matches
|
||||
/// TextItem.mcid).
|
||||
#[pyo3(get)]
|
||||
pub mcid: i64,
|
||||
/// Standard structure type name ("H1".."H6", "P", "Table", "TD", ...).
|
||||
#[pyo3(get)]
|
||||
pub role: String,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyStructureElement {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"StructureElement(page={}, mcid={}, role='{}')",
|
||||
self.page, self.mcid, self.role
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -388,18 +356,6 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
mcid: item.mcid,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_structure_elements(elements: Vec<crate::StructureElement>) -> Vec<PyStructureElement> {
|
||||
elements
|
||||
.into_iter()
|
||||
.map(|e| PyStructureElement {
|
||||
page: e.page,
|
||||
mcid: e.mcid,
|
||||
role: e.role,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
@@ -657,48 +613,6 @@ fn extract_pages_markdown_bytes(
|
||||
Ok(to_py_pages_result(result))
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from a tagged PDF file.
|
||||
///
|
||||
/// Parses the document's structure tree (when present) and returns one
|
||||
/// entry per marked-content reference, resolved to its 1-indexed page,
|
||||
/// MCID, and structure type name ("H1".."H6", "P", "Table", ...). Returns
|
||||
/// an empty list when the PDF is not tagged.
|
||||
///
|
||||
/// Join (page, mcid) against the page/mcid attributes from
|
||||
/// [`extract_text_with_positions`] to attach heading levels and other
|
||||
/// semantic roles to extracted text.
|
||||
///
|
||||
/// Args:
|
||||
/// path: Path to the PDF file.
|
||||
/// pages: Optional list of 1-indexed pages (matching TextItem.page).
|
||||
/// When None (default), the whole document is returned.
|
||||
///
|
||||
/// Returns:
|
||||
/// List of StructureElement sorted by (page, mcid).
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn extract_structure_elements(
|
||||
path: &str,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<Vec<PyStructureElement>> {
|
||||
let elements = crate::extract_structure_elements(path, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(convert_structure_elements(elements))
|
||||
}
|
||||
|
||||
/// Extract structure-tree element references from tagged PDF bytes.
|
||||
///
|
||||
/// See [`extract_structure_elements`] for details.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn extract_structure_elements_bytes(
|
||||
data: &[u8],
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<Vec<PyStructureElement>> {
|
||||
let elements =
|
||||
crate::extract_structure_elements_mem(data, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(convert_structure_elements(elements))
|
||||
}
|
||||
|
||||
/// Python module definition.
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
@@ -706,7 +620,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPageOcrReasons>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyStructureElement>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
m.add_class::<PyPageRegionTexts>()?;
|
||||
m.add_class::<PyPageMarkdown>()?;
|
||||
@@ -721,8 +634,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_function(wrap_pyfunction!(extract_text_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_structure_elements, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_structure_elements_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
|
||||
|
||||
+38
-819
File diff suppressed because it is too large
Load Diff
+12
-998
File diff suppressed because it is too large
Load Diff
+2
-450
@@ -4,10 +4,10 @@
|
||||
//! gridlines. Many IRS forms and government PDFs use these instead of
|
||||
//! `re` (rectangle) operators.
|
||||
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::tables::Table;
|
||||
use crate::types::{PdfLine, PdfRect, TextItem};
|
||||
use crate::types::{PdfLine, TextItem};
|
||||
|
||||
use super::detect_rects::{assign_items_to_grid, snap_edges};
|
||||
|
||||
@@ -15,33 +15,11 @@ const RULE_Y_TOLERANCE: f32 = 2.0;
|
||||
const RULE_JOIN_GAP: f32 = 6.0;
|
||||
const RULE_SPAN_TOLERANCE: f32 = 8.0;
|
||||
const TEXT_ROW_TOLERANCE: f32 = 2.5;
|
||||
const DENSE_CHART_MIN_VERTICAL_EDGES: usize = 27;
|
||||
const DENSE_CHART_LABEL_PAD: f32 = 20.0;
|
||||
const DENSE_CHART_MAX_SHARED_PANEL_GRIDS: usize = 4;
|
||||
|
||||
type HorizontalRule = (f32, f32, f32); // (y, x_min, x_max)
|
||||
type VerticalRule = (f32, f32, f32); // (x, y_min, y_max)
|
||||
type AnchoredRow<'a> = (f32, Vec<(usize, &'a TextItem)>);
|
||||
|
||||
fn dense_chart_grids_are_co_located(
|
||||
left: (f32, f32, f32, f32),
|
||||
right: (f32, f32, f32, f32),
|
||||
) -> bool {
|
||||
let left_width = left.2 - left.0;
|
||||
let right_width = right.2 - right.0;
|
||||
let left_height = left.3 - left.1;
|
||||
let right_height = right.3 - right.1;
|
||||
let horizontal_overlap = (left.2.min(right.2) - left.0.max(right.0)).max(0.0);
|
||||
let vertical_overlap = (left.3.min(right.3) - left.1.max(right.1)).max(0.0);
|
||||
let horizontal_gap = (left.0.max(right.0) - left.2.min(right.2)).max(0.0);
|
||||
let vertical_gap = (left.1.max(right.1) - left.3.min(right.3)).max(0.0);
|
||||
|
||||
(vertical_overlap >= left_height.min(right_height) * 0.5
|
||||
&& horizontal_gap <= left_width.min(right_width) * 0.5)
|
||||
|| (horizontal_overlap >= left_width.min(right_width) * 0.5
|
||||
&& vertical_gap <= left_height.min(right_height) * 0.5)
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct TextAnchorTable {
|
||||
table: Table,
|
||||
@@ -1209,279 +1187,6 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
detect_tables_from_lines_inner(items, lines, page, true, true)
|
||||
}
|
||||
|
||||
/// Bounding boxes of chart panels backed by a very dense vector grid.
|
||||
///
|
||||
/// Tables support at most 25 columns, so a panel with at least 27 distinct,
|
||||
/// long vertical coordinates plus repeated horizontal rules is treated as
|
||||
/// chart geometry. When the grid is enclosed by a painted panel rectangle,
|
||||
/// the region expands to that rectangle so axis labels, legends, and source
|
||||
/// notes remain part of the figure instead of forming a heuristic table.
|
||||
pub(crate) fn detect_dense_line_chart_regions(
|
||||
lines: &[PdfLine],
|
||||
rects: &[PdfRect],
|
||||
page: u32,
|
||||
) -> Vec<(f32, f32, f32, f32)> {
|
||||
const ANGLE_TOLERANCE: f32 = 0.035;
|
||||
const MIN_GRID_LINE_LENGTH: f32 = 40.0;
|
||||
const EXTENT_TOLERANCE: f32 = 6.0;
|
||||
|
||||
let mut verticals = Vec::new();
|
||||
let mut horizontals = Vec::new();
|
||||
for line in lines.iter().filter(|line| line.page == page) {
|
||||
let dx = (line.x2 - line.x1).abs();
|
||||
let dy = (line.y2 - line.y1).abs();
|
||||
let length = dx.hypot(dy);
|
||||
if length < MIN_GRID_LINE_LENGTH {
|
||||
continue;
|
||||
}
|
||||
if dy > 0.01 && dx / dy <= ANGLE_TOLERANCE {
|
||||
verticals.push((
|
||||
(line.x1 + line.x2) / 2.0,
|
||||
line.y1.min(line.y2),
|
||||
line.y1.max(line.y2),
|
||||
));
|
||||
} else if dx > 0.01 && dy / dx <= ANGLE_TOLERANCE {
|
||||
horizontals.push((
|
||||
(line.y1 + line.y2) / 2.0,
|
||||
line.x1.min(line.x2),
|
||||
line.x1.max(line.x2),
|
||||
));
|
||||
}
|
||||
}
|
||||
if verticals.len() < DENSE_CHART_MIN_VERTICAL_EDGES || horizontals.len() < 3 {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Group similar vertical extents once. Neighboring buckets are consulted
|
||||
// below so coordinates that straddle a bucket boundary still form one
|
||||
// family, while each line participates in only a constant number of
|
||||
// candidates instead of being re-scanned for every vertical anchor.
|
||||
let extent_key = |value: f32| (value / EXTENT_TOLERANCE).round() as i32;
|
||||
let mut extent_buckets: HashMap<(i32, i32), Vec<VerticalRule>> = HashMap::new();
|
||||
for vertical in verticals {
|
||||
extent_buckets
|
||||
.entry((extent_key(vertical.1), extent_key(vertical.2)))
|
||||
.or_default()
|
||||
.push(vertical);
|
||||
}
|
||||
|
||||
let mut grid_regions = Vec::new();
|
||||
let extent_keys: Vec<(i32, i32)> = extent_buckets.keys().copied().collect();
|
||||
for key in extent_keys {
|
||||
let anchor_family = &extent_buckets[&key];
|
||||
let anchor_bottom = anchor_family.iter().map(|vertical| vertical.1).sum::<f32>()
|
||||
/ anchor_family.len() as f32;
|
||||
let anchor_top = anchor_family.iter().map(|vertical| vertical.2).sum::<f32>()
|
||||
/ anchor_family.len() as f32;
|
||||
|
||||
let mut family = Vec::new();
|
||||
for bottom_offset in -1..=1 {
|
||||
for top_offset in -1..=1 {
|
||||
if let Some(bucket) =
|
||||
extent_buckets.get(&(key.0 + bottom_offset, key.1 + top_offset))
|
||||
{
|
||||
family.extend(bucket.iter().copied().filter(|vertical| {
|
||||
(vertical.1 - anchor_bottom).abs() <= EXTENT_TOLERANCE
|
||||
&& (vertical.2 - anchor_top).abs() <= EXTENT_TOLERANCE
|
||||
}));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let xs = snap_edges(&family.iter().map(|&(x, _, _)| x).collect::<Vec<_>>(), 3.0);
|
||||
if xs.len() < DENSE_CHART_MIN_VERTICAL_EDGES {
|
||||
continue;
|
||||
}
|
||||
|
||||
let grid_bottom =
|
||||
family.iter().map(|vertical| vertical.1).sum::<f32>() / family.len() as f32;
|
||||
let grid_top = family.iter().map(|vertical| vertical.2).sum::<f32>() / family.len() as f32;
|
||||
if grid_top - grid_bottom < 60.0 {
|
||||
continue;
|
||||
}
|
||||
|
||||
// A horizontal rule must support the same contiguous dense run of
|
||||
// vertical coordinates. Splitting at sparse X gaps prevents a shared
|
||||
// rule from joining a chart to a neighboring ruled table. Keying by
|
||||
// the covered X-index range also lets multiple chart panels sharing
|
||||
// the same Y extents produce independent regions.
|
||||
let mut supported_spans: HashMap<(usize, usize), Vec<f32>> = HashMap::new();
|
||||
for &(y, line_left, line_right) in &horizontals {
|
||||
if y < grid_bottom - EXTENT_TOLERANCE || y > grid_top + EXTENT_TOLERANCE {
|
||||
continue;
|
||||
}
|
||||
let start = xs.partition_point(|&x| x < line_left - EXTENT_TOLERANCE);
|
||||
let end = xs.partition_point(|&x| x <= line_right + EXTENT_TOLERANCE);
|
||||
if end - start < DENSE_CHART_MIN_VERTICAL_EDGES {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut gaps: Vec<f32> = xs[start..end]
|
||||
.windows(2)
|
||||
.map(|pair| pair[1] - pair[0])
|
||||
.collect();
|
||||
gaps.sort_by(f32::total_cmp);
|
||||
let dense_gap = gaps[gaps.len() / 4];
|
||||
let run_break = (dense_gap * 3.0).max(12.0);
|
||||
let locally_dense_gap_limit = (dense_gap * 1.5).max(6.0);
|
||||
let locally_dense_gaps = gaps
|
||||
.iter()
|
||||
.filter(|&&gap| gap <= locally_dense_gap_limit)
|
||||
.count();
|
||||
|
||||
let mut run_start = start;
|
||||
let mut retained_dense_run = false;
|
||||
for index in start..end - 1 {
|
||||
if xs[index + 1] - xs[index] <= run_break {
|
||||
continue;
|
||||
}
|
||||
let run_end = index + 1;
|
||||
if run_end - run_start >= DENSE_CHART_MIN_VERTICAL_EDGES
|
||||
&& xs[run_end - 1] - xs[run_start] >= 120.0
|
||||
{
|
||||
supported_spans
|
||||
.entry((run_start, run_end))
|
||||
.or_default()
|
||||
.push(y);
|
||||
retained_dense_run = true;
|
||||
}
|
||||
run_start = run_end;
|
||||
}
|
||||
if end - run_start >= DENSE_CHART_MIN_VERTICAL_EDGES
|
||||
&& xs[end - 1] - xs[run_start] >= 120.0
|
||||
{
|
||||
supported_spans.entry((run_start, end)).or_default().push(y);
|
||||
retained_dense_run = true;
|
||||
}
|
||||
|
||||
// One or two wider category gaps may split an otherwise dense
|
||||
// chart into sub-threshold runs. Keep the full family only when
|
||||
// its total width remains close to the expected dense spacing;
|
||||
// a neighboring sparse table makes this ratio much larger.
|
||||
let span_width = xs[end - 1] - xs[start];
|
||||
let expected_dense_width = dense_gap * (end - start - 1) as f32;
|
||||
if !retained_dense_run
|
||||
&& span_width >= 120.0
|
||||
&& span_width <= expected_dense_width * 1.35
|
||||
&& gaps.len().saturating_sub(locally_dense_gaps) <= 2
|
||||
{
|
||||
supported_spans.entry((start, end)).or_default().push(y);
|
||||
}
|
||||
}
|
||||
|
||||
for ((start, end), ys) in supported_spans {
|
||||
if snap_edges(&ys, 3.0).len() >= 3 {
|
||||
grid_regions.push((xs[start], grid_bottom, xs[end - 1], grid_top));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Prefer the smallest qualifying region when a broad rule happens to
|
||||
// cover a denser nested panel, and retain every non-overlapping panel.
|
||||
grid_regions.sort_by(|left, right| {
|
||||
let left_area = (left.2 - left.0) * (left.3 - left.1);
|
||||
let right_area = (right.2 - right.0) * (right.3 - right.1);
|
||||
left_area.total_cmp(&right_area)
|
||||
});
|
||||
let mut selected_regions: Vec<(f32, f32, f32, f32)> = Vec::new();
|
||||
for region in grid_regions {
|
||||
let area = (region.2 - region.0) * (region.3 - region.1);
|
||||
let duplicates_existing = selected_regions.iter().any(|existing| {
|
||||
let overlap_width = (region.2.min(existing.2) - region.0.max(existing.0)).max(0.0);
|
||||
let overlap_height = (region.3.min(existing.3) - region.1.max(existing.1)).max(0.0);
|
||||
let overlap_area = overlap_width * overlap_height;
|
||||
let existing_area = (existing.2 - existing.0) * (existing.3 - existing.1);
|
||||
overlap_area >= area.min(existing_area) * 0.8
|
||||
});
|
||||
if !duplicates_existing {
|
||||
selected_regions.push(region);
|
||||
}
|
||||
}
|
||||
|
||||
let all_grid_regions = selected_regions.clone();
|
||||
let mut regions: Vec<_> = selected_regions
|
||||
.into_iter()
|
||||
.map(|grid_region| {
|
||||
let (grid_left, grid_bottom, grid_right, grid_top) = grid_region;
|
||||
let enclosing_panel = rects
|
||||
.iter()
|
||||
.filter(|rect| rect.page == page)
|
||||
.filter_map(|rect| {
|
||||
let (left, width) = if rect.width < 0.0 {
|
||||
(rect.x + rect.width, -rect.width)
|
||||
} else {
|
||||
(rect.x, rect.width)
|
||||
};
|
||||
let (bottom, height) = if rect.height < 0.0 {
|
||||
(rect.y + rect.height, -rect.height)
|
||||
} else {
|
||||
(rect.y, rect.height)
|
||||
};
|
||||
let right = left + width;
|
||||
let top = bottom + height;
|
||||
let enclosed_grids: Vec<_> = all_grid_regions
|
||||
.iter()
|
||||
.filter(|&&(other_left, other_bottom, other_right, other_top)| {
|
||||
left <= other_left + EXTENT_TOLERANCE
|
||||
&& right >= other_right - EXTENT_TOLERANCE
|
||||
&& bottom <= other_bottom + EXTENT_TOLERANCE
|
||||
&& top >= other_top - EXTENT_TOLERANCE
|
||||
})
|
||||
.copied()
|
||||
.collect();
|
||||
if enclosed_grids.len() > DENSE_CHART_MAX_SHARED_PANEL_GRIDS
|
||||
|| enclosed_grids.iter().any(|&other| {
|
||||
other != grid_region
|
||||
&& !dense_chart_grids_are_co_located(grid_region, other)
|
||||
})
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let enclosed_grid_bounds =
|
||||
enclosed_grids.into_iter().reduce(|bounds, other| {
|
||||
(
|
||||
bounds.0.min(other.0),
|
||||
bounds.1.min(other.1),
|
||||
bounds.2.max(other.2),
|
||||
bounds.3.max(other.3),
|
||||
)
|
||||
})?;
|
||||
let enclosed_width = enclosed_grid_bounds.2 - enclosed_grid_bounds.0;
|
||||
let enclosed_height = enclosed_grid_bounds.3 - enclosed_grid_bounds.1;
|
||||
(left <= grid_left + EXTENT_TOLERANCE
|
||||
&& right >= grid_right - EXTENT_TOLERANCE
|
||||
&& bottom <= grid_bottom + EXTENT_TOLERANCE
|
||||
&& top >= grid_top - EXTENT_TOLERANCE
|
||||
&& width <= enclosed_width * 2.0
|
||||
&& height <= enclosed_height * 4.0
|
||||
&& !(left < 5.0 && bottom < 5.0))
|
||||
.then_some(((left, bottom, right, top), width * height))
|
||||
})
|
||||
.min_by(|left, right| left.1.total_cmp(&right.1))
|
||||
.map(|(region, _)| region);
|
||||
|
||||
enclosing_panel.unwrap_or((
|
||||
grid_left - DENSE_CHART_LABEL_PAD,
|
||||
grid_bottom - DENSE_CHART_LABEL_PAD,
|
||||
grid_right + DENSE_CHART_LABEL_PAD,
|
||||
grid_top + DENSE_CHART_LABEL_PAD,
|
||||
))
|
||||
})
|
||||
.collect();
|
||||
regions.sort_by(|left, right| {
|
||||
left.0
|
||||
.total_cmp(&right.0)
|
||||
.then_with(|| left.1.total_cmp(&right.1))
|
||||
});
|
||||
regions.dedup_by(|left, right| {
|
||||
(left.0 - right.0).abs() <= EXTENT_TOLERANCE
|
||||
&& (left.1 - right.1).abs() <= EXTENT_TOLERANCE
|
||||
&& (left.2 - right.2).abs() <= EXTENT_TOLERANCE
|
||||
&& (left.3 - right.3).abs() <= EXTENT_TOLERANCE
|
||||
});
|
||||
regions
|
||||
}
|
||||
|
||||
/// Detect only tables whose cell grid is backed by explicit vector geometry.
|
||||
///
|
||||
/// Region-level TSR callers need physical cell boundaries for crop bboxes, so
|
||||
@@ -1916,159 +1621,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dense_vector_grid_expands_to_enclosing_chart_panel() {
|
||||
let mut lines: Vec<PdfLine> = (0..30)
|
||||
.map(|column| make_vline(100.0 + column as f32 * 8.0, 400.0, 550.0, 1))
|
||||
.collect();
|
||||
lines.extend((0..6).map(|row| make_hline(400.0 + row as f32 * 30.0, 100.0, 332.0, 1)));
|
||||
let rects = vec![PdfRect {
|
||||
x: 80.0,
|
||||
y: 350.0,
|
||||
width: 280.0,
|
||||
height: 240.0,
|
||||
page: 1,
|
||||
}];
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &rects, 1),
|
||||
vec![(80.0, 350.0, 360.0, 590.0)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn frameless_dense_vector_grid_includes_label_padding() {
|
||||
let mut lines: Vec<PdfLine> = (0..30)
|
||||
.map(|column| make_vline(100.0 + column as f32 * 8.0, 400.0, 550.0, 1))
|
||||
.collect();
|
||||
lines.extend((0..6).map(|row| make_hline(400.0 + row as f32 * 30.0, 100.0, 332.0, 1)));
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &[], 1),
|
||||
vec![(80.0, 380.0, 352.0, 570.0)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_dense_vector_panels_are_retained() {
|
||||
let mut lines = Vec::new();
|
||||
for panel_left in [60.0, 380.0] {
|
||||
lines.extend(
|
||||
(0..30).map(|column| make_vline(panel_left + column as f32 * 8.0, 400.0, 550.0, 1)),
|
||||
);
|
||||
lines.extend((0..6).map(|row| {
|
||||
make_hline(400.0 + row as f32 * 30.0, panel_left, panel_left + 232.0, 1)
|
||||
}));
|
||||
}
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &[], 1),
|
||||
vec![(40.0, 380.0, 312.0, 570.0), (360.0, 380.0, 632.0, 570.0),]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_dense_vector_panels_use_shared_enclosing_panel() {
|
||||
let mut lines = Vec::new();
|
||||
for panel_left in [60.0, 380.0] {
|
||||
lines.extend(
|
||||
(0..30).map(|column| make_vline(panel_left + column as f32 * 8.0, 400.0, 550.0, 1)),
|
||||
);
|
||||
lines.extend((0..6).map(|row| {
|
||||
make_hline(400.0 + row as f32 * 30.0, panel_left, panel_left + 232.0, 1)
|
||||
}));
|
||||
}
|
||||
let rects = vec![PdfRect {
|
||||
x: 40.0,
|
||||
y: 350.0,
|
||||
width: 592.0,
|
||||
height: 240.0,
|
||||
page: 1,
|
||||
}];
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &rects, 1),
|
||||
vec![(40.0, 350.0, 632.0, 590.0)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shared_rules_do_not_join_dense_chart_to_adjacent_table() {
|
||||
let mut lines: Vec<PdfLine> = (0..30)
|
||||
.map(|column| make_vline(60.0 + column as f32 * 8.0, 400.0, 550.0, 1))
|
||||
.collect();
|
||||
|
||||
lines.extend(
|
||||
[330.0, 390.0, 450.0, 510.0, 570.0, 630.0]
|
||||
.into_iter()
|
||||
.map(|x| make_vline(x, 400.0, 550.0, 1)),
|
||||
);
|
||||
lines.extend((0..6).map(|row| make_hline(400.0 + row as f32 * 30.0, 60.0, 630.0, 1)));
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &[], 1),
|
||||
vec![(40.0, 380.0, 312.0, 570.0)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn uneven_dense_spacing_keeps_the_complete_chart_region() {
|
||||
let mut xs: Vec<f32> = (0..15).map(|column| 60.0 + column as f32 * 8.0).collect();
|
||||
xs.extend((0..15).map(|column| 212.0 + column as f32 * 8.0));
|
||||
let mut lines: Vec<PdfLine> = xs.iter().map(|&x| make_vline(x, 400.0, 550.0, 1)).collect();
|
||||
lines.extend((0..6).map(|row| make_hline(400.0 + row as f32 * 30.0, 60.0, 324.0, 1)));
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &[], 1),
|
||||
vec![(40.0, 380.0, 344.0, 570.0)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subthreshold_dense_run_does_not_absorb_adjacent_sparse_grid() {
|
||||
let mut xs: Vec<f32> = (0..21).map(|column| 60.0 + column as f32 * 8.0).collect();
|
||||
xs.extend((0..6).map(|column| 248.0 + column as f32 * 18.0));
|
||||
let mut lines: Vec<PdfLine> = xs.iter().map(|&x| make_vline(x, 400.0, 550.0, 1)).collect();
|
||||
lines.extend((0..6).map(|row| make_hline(400.0 + row as f32 * 30.0, 60.0, 338.0, 1)));
|
||||
|
||||
assert!(detect_dense_line_chart_regions(&lines, &[], 1).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn broad_frame_does_not_merge_distant_dense_grids() {
|
||||
let mut lines = Vec::new();
|
||||
for panel_left in [60.0, 700.0] {
|
||||
lines.extend(
|
||||
(0..30).map(|column| make_vline(panel_left + column as f32 * 8.0, 400.0, 550.0, 1)),
|
||||
);
|
||||
lines.extend((0..6).map(|row| {
|
||||
make_hline(400.0 + row as f32 * 30.0, panel_left, panel_left + 232.0, 1)
|
||||
}));
|
||||
}
|
||||
let rects = vec![PdfRect {
|
||||
x: 40.0,
|
||||
y: 350.0,
|
||||
width: 912.0,
|
||||
height: 240.0,
|
||||
page: 1,
|
||||
}];
|
||||
|
||||
assert_eq!(
|
||||
detect_dense_line_chart_regions(&lines, &rects, 1),
|
||||
vec![(40.0, 380.0, 312.0, 570.0), (680.0, 380.0, 952.0, 570.0)]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn supported_width_vector_table_is_not_a_dense_chart() {
|
||||
let mut lines: Vec<PdfLine> = (0..26)
|
||||
.map(|column| make_vline(100.0 + column as f32 * 10.0, 400.0, 550.0, 1))
|
||||
.collect();
|
||||
lines.extend((0..6).map(|row| make_hline(400.0 + row as f32 * 30.0, 100.0, 350.0, 1)));
|
||||
|
||||
assert!(detect_dense_line_chart_regions(&lines, &[], 1).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_basic_grid_detection() {
|
||||
// 3x2 grid with horizontal lines at y=500, 480, 460 and vertical at x=100, 200, 300
|
||||
|
||||
+10
-454
@@ -1996,249 +1996,17 @@ fn without_dominant_page_backgrounds(rects: &[(f32, f32, f32, f32)]) -> Vec<(f32
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Repeated rows of touching cell rectangles are stronger table evidence
|
||||
/// than the bar-length variation used by the chart detector.
|
||||
/// Detect a table from cell-background rects that failed grid detection.
|
||||
///
|
||||
/// Ruled tables with wrapped labels naturally have variable row heights, and
|
||||
/// numeric-heavy cells can otherwise resemble horizontal or vertical bars.
|
||||
/// Require several rows to repeat a shared edge schema before overriding the
|
||||
/// chart hypothesis so sparse plots and independent bars remain unaffected.
|
||||
fn is_repeated_cell_grid(group_rects: &[(f32, f32, f32, f32)]) -> bool {
|
||||
type RowGroup = (f32, f32, Vec<(f32, f32)>);
|
||||
|
||||
const ROW_EDGE_TOLERANCE: f32 = 3.0;
|
||||
const MIN_GRID_ROWS: usize = 4;
|
||||
const MIN_CELLS_PER_ROW: usize = 3;
|
||||
|
||||
if group_rects.len() < MIN_GRID_ROWS * MIN_CELLS_PER_ROW {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut row_groups: Vec<RowGroup> = Vec::new();
|
||||
for &(x, y, width, height) in group_rects {
|
||||
if width < 5.0 || height < 5.0 {
|
||||
continue;
|
||||
}
|
||||
let top = y + height;
|
||||
if let Some((_, _, cells)) = row_groups.iter_mut().find(|(bottom, row_top, _)| {
|
||||
(y - *bottom).abs() <= ROW_EDGE_TOLERANCE
|
||||
&& (top - *row_top).abs() <= ROW_EDGE_TOLERANCE
|
||||
}) {
|
||||
cells.push((x, x + width));
|
||||
} else {
|
||||
row_groups.push((y, top, vec![(x, x + width)]));
|
||||
}
|
||||
}
|
||||
|
||||
let mut row_schemas = Vec::new();
|
||||
for (_, _, mut cells) in row_groups {
|
||||
if cells.len() < MIN_CELLS_PER_ROW {
|
||||
continue;
|
||||
}
|
||||
let mut widths: Vec<f32> = cells.iter().map(|&(left, right)| right - left).collect();
|
||||
widths.sort_by(f32::total_cmp);
|
||||
let median_width = widths[widths.len() / 2];
|
||||
cells.retain(|&(left, right)| right - left <= median_width * 2.5);
|
||||
cells.sort_by(|left, right| {
|
||||
left.0
|
||||
.total_cmp(&right.0)
|
||||
.then_with(|| left.1.total_cmp(&right.1))
|
||||
});
|
||||
cells.dedup_by(|left, right| {
|
||||
(left.0 - right.0).abs() <= ROW_EDGE_TOLERANCE
|
||||
&& (left.1 - right.1).abs() <= ROW_EDGE_TOLERANCE
|
||||
});
|
||||
if cells.len() < MIN_CELLS_PER_ROW
|
||||
|| cells
|
||||
.windows(2)
|
||||
.any(|pair| pair[1].0 > pair[0].1 + ROW_EDGE_TOLERANCE)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let edges: Vec<f32> = cells
|
||||
.iter()
|
||||
.flat_map(|&(left, right)| [left, right])
|
||||
.collect();
|
||||
let schema = snap_edges(&edges, ROW_EDGE_TOLERANCE);
|
||||
if schema.len() > MIN_CELLS_PER_ROW {
|
||||
row_schemas.push(schema);
|
||||
}
|
||||
}
|
||||
if row_schemas.len() < MIN_GRID_ROWS {
|
||||
return false;
|
||||
}
|
||||
|
||||
let reference = row_schemas
|
||||
.iter()
|
||||
.max_by_key(|schema| schema.len())
|
||||
.expect("grid rows are non-empty");
|
||||
row_schemas
|
||||
.iter()
|
||||
.filter(|schema| {
|
||||
let comparable_edges = reference.len().min(schema.len());
|
||||
let matched_edges = schema
|
||||
.iter()
|
||||
.filter(|edge| {
|
||||
reference
|
||||
.iter()
|
||||
.any(|reference_edge| (*edge - *reference_edge).abs() <= ROW_EDGE_TOLERANCE)
|
||||
})
|
||||
.count();
|
||||
matched_edges > MIN_CELLS_PER_ROW && matched_edges * 4 >= comparable_edges * 3
|
||||
})
|
||||
.count()
|
||||
>= MIN_GRID_ROWS
|
||||
}
|
||||
|
||||
fn repeated_cell_grid_overrides_bar_hypothesis(group_rects: &[(f32, f32, f32, f32)]) -> bool {
|
||||
is_repeated_cell_grid(group_rects)
|
||||
&& without_dominant_page_backgrounds(group_rects).len() == group_rects.len()
|
||||
}
|
||||
|
||||
/// Detect horizontal segmented stacks from aligned rows of touching rects.
|
||||
///
|
||||
/// Category rows must have visible gutters and data-varying internal segment
|
||||
/// boundaries, unlike the stable boundaries of a ruled table.
|
||||
struct SegmentedBarGeometry {
|
||||
bounds: (f32, f32, f32, f32),
|
||||
row_bands: Vec<(f32, f32)>,
|
||||
}
|
||||
|
||||
fn segmented_stacked_bar_geometry(
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
) -> Option<SegmentedBarGeometry> {
|
||||
type BarRow = (f32, f32, Vec<(f32, f32)>);
|
||||
|
||||
const EDGE_TOLERANCE: f32 = 3.0;
|
||||
const MIN_ROWS: usize = 4;
|
||||
const MIN_SEGMENTS: usize = 3;
|
||||
|
||||
let mut rows: Vec<BarRow> = Vec::new();
|
||||
for &(x, y, width, height) in group_rects {
|
||||
if width < 5.0 || height < 5.0 {
|
||||
continue;
|
||||
}
|
||||
let top = y + height;
|
||||
if let Some((_, _, segments)) = rows.iter_mut().find(|(bottom, row_top, _)| {
|
||||
(y - *bottom).abs() <= EDGE_TOLERANCE && (top - *row_top).abs() <= EDGE_TOLERANCE
|
||||
}) {
|
||||
segments.push((x, x + width));
|
||||
} else {
|
||||
rows.push((y, top, vec![(x, x + width)]));
|
||||
}
|
||||
}
|
||||
|
||||
rows.retain_mut(|(_, _, segments)| {
|
||||
segments.sort_by(|left, right| left.0.total_cmp(&right.0));
|
||||
segments.len() >= MIN_SEGMENTS
|
||||
&& segments
|
||||
.windows(2)
|
||||
.all(|pair| (pair[1].0 - pair[0].1).abs() <= EDGE_TOLERANCE)
|
||||
});
|
||||
if rows.len() < MIN_ROWS {
|
||||
return None;
|
||||
}
|
||||
rows.sort_by(|left, right| left.0.total_cmp(&right.0));
|
||||
|
||||
// Table rows normally share borders. Horizontal stacked bars instead
|
||||
// leave a visible gutter between category rows.
|
||||
if rows.windows(2).any(|pair| {
|
||||
let shorter_height = (pair[0].1 - pair[0].0).min(pair[1].1 - pair[1].0);
|
||||
pair[1].0 - pair[0].1 < (shorter_height * 0.25).max(2.0)
|
||||
}) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// At least two rows must move an internal segment boundary. Stable
|
||||
// boundaries across every row are stronger evidence for a ruled table.
|
||||
let reference_edges: Vec<f32> = rows[0]
|
||||
.2
|
||||
.iter()
|
||||
.take(rows[0].2.len() - 1)
|
||||
.map(|segment| segment.1)
|
||||
.collect();
|
||||
let drifting_rows = rows
|
||||
.iter()
|
||||
.skip(1)
|
||||
.filter(|(_, _, segments)| {
|
||||
let edges: Vec<f32> = segments
|
||||
.iter()
|
||||
.take(segments.len() - 1)
|
||||
.map(|segment| segment.1)
|
||||
.collect();
|
||||
edges.len() == reference_edges.len()
|
||||
&& edges
|
||||
.iter()
|
||||
.zip(&reference_edges)
|
||||
.any(|(edge, reference)| (edge - reference).abs() > EDGE_TOLERANCE)
|
||||
})
|
||||
.count();
|
||||
if drifting_rows < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let left = rows
|
||||
.iter()
|
||||
.flat_map(|row| &row.2)
|
||||
.map(|segment| segment.0)
|
||||
.reduce(f32::min)?;
|
||||
let right = rows
|
||||
.iter()
|
||||
.flat_map(|row| &row.2)
|
||||
.map(|segment| segment.1)
|
||||
.reduce(f32::max)?;
|
||||
let bottom = rows.iter().map(|row| row.0).reduce(f32::min)?;
|
||||
let top = rows.iter().map(|row| row.1).reduce(f32::max)?;
|
||||
let row_bands = rows.iter().map(|row| (row.0, row.1)).collect();
|
||||
Some(SegmentedBarGeometry {
|
||||
bounds: (left, bottom, right, top),
|
||||
row_bands,
|
||||
})
|
||||
}
|
||||
|
||||
/// Category labels beside multiple bar rows are independent chart evidence:
|
||||
/// numeric table text stays inside its cells, regardless of whether the table
|
||||
/// has an outer border or extra padding.
|
||||
fn has_external_segmented_bar_labels(
|
||||
items: &[TextItem],
|
||||
page: u32,
|
||||
geometry: &SegmentedBarGeometry,
|
||||
) -> bool {
|
||||
const LABEL_EDGE_TOLERANCE: f32 = 3.0;
|
||||
const LABEL_CLAIM_PAD: f32 = 20.0;
|
||||
|
||||
let (content_left, _, content_right, _) = geometry.bounds;
|
||||
let labeled_rows = geometry
|
||||
.row_bands
|
||||
.iter()
|
||||
.filter(|&&(row_bottom, row_top)| {
|
||||
items.iter().any(|item| {
|
||||
if item.page != page || item.text.trim().is_empty() {
|
||||
return false;
|
||||
}
|
||||
let item_left = item.x.min(item.x + item.width);
|
||||
let item_right = item.x.max(item.x + item.width);
|
||||
let item_center_x = (item_left + item_right) / 2.0;
|
||||
let item_center_y = item.y + item.height / 2.0;
|
||||
let beside_stack = (item_center_x <= content_left + LABEL_EDGE_TOLERANCE
|
||||
&& item_center_x >= content_left - LABEL_CLAIM_PAD
|
||||
&& item_left < content_left)
|
||||
|| (item_center_x >= content_right - LABEL_EDGE_TOLERANCE
|
||||
&& item_center_x <= content_right + LABEL_CLAIM_PAD
|
||||
&& item_right > content_right);
|
||||
beside_stack
|
||||
&& item_center_y >= row_bottom - LABEL_EDGE_TOLERANCE
|
||||
&& item_center_y <= row_top + LABEL_EDGE_TOLERANCE
|
||||
})
|
||||
})
|
||||
.count();
|
||||
|
||||
labeled_rows >= 2 && labeled_rows * 2 >= geometry.row_bands.len()
|
||||
}
|
||||
|
||||
/// Recognize filled vertical or horizontal bars whose geometry and labels are
|
||||
/// data-driven rather than uniform table cells.
|
||||
fn has_chart_bar_signature(
|
||||
/// Uses rect Y-edges for row boundaries and text X-position clustering for
|
||||
/// columns. Handles tables with cell backgrounds that don't form a clean
|
||||
/// X-edge grid (variable column widths, decorative fills).
|
||||
/// Chart-bar signature: ≥3 rects sharing an aligned bottom edge (the axis),
|
||||
/// with similar widths (bars) but strongly varying heights (data-driven),
|
||||
/// holding at most a single numeric data label each. Bar charts drawn as
|
||||
/// filled rects otherwise read as cell rects and grid their axis labels
|
||||
/// into a phantom table. The mirrored check catches horizontal bar charts.
|
||||
fn is_chart_bar_cluster(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
@@ -2345,30 +2113,6 @@ fn has_chart_bar_signature(
|
||||
|| bar_family(|r| r.1, |r| r.3, |r| r.2, |r| r.0)
|
||||
}
|
||||
|
||||
fn is_chart_bar_cluster(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> bool {
|
||||
let has_bar_signature = has_chart_bar_signature(items, group_rects, page);
|
||||
|
||||
// A segmented horizontal chart can share most of its edges across rows.
|
||||
// Row-aligned category labels outside the stack distinguish it from a
|
||||
// numeric table without depending on whether either shape has a frame.
|
||||
if has_bar_signature {
|
||||
if let Some(geometry) = segmented_stacked_bar_geometry(group_rects) {
|
||||
if has_external_segmented_bar_labels(items, page, &geometry) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if repeated_cell_grid_overrides_bar_hypothesis(group_rects) {
|
||||
return false;
|
||||
}
|
||||
|
||||
has_bar_signature
|
||||
}
|
||||
|
||||
fn detect_row_stripe_table_from_cell_rects(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
@@ -3339,194 +3083,6 @@ mod tests {
|
||||
assert!(detect_chart_regions(&items, &rects, 1).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn variable_height_ruled_grid_overrides_bar_hypothesis() {
|
||||
let edge_sets = [
|
||||
[80.0, 140.0, 200.0, 260.0, 320.0, 380.0, 440.0, 500.0, 560.0],
|
||||
[80.0, 140.0, 210.0, 260.0, 320.0, 380.0, 450.0, 500.0, 560.0],
|
||||
];
|
||||
let heights = [20.0, 34.0, 26.0, 42.0, 20.0, 34.0];
|
||||
let edge_variants = [0, 0, 0, 0, 1, 1];
|
||||
let mut rects = Vec::new();
|
||||
let mut y = 650.0;
|
||||
for (row, height) in heights.into_iter().enumerate() {
|
||||
let edges = edge_sets[edge_variants[row]];
|
||||
rects.extend(
|
||||
edges
|
||||
.windows(2)
|
||||
.map(|edge| (edge[0], y, edge[1] - edge[0], height)),
|
||||
);
|
||||
y -= height;
|
||||
}
|
||||
|
||||
assert!(is_repeated_cell_grid(&rects));
|
||||
assert!(has_chart_bar_signature(&[], &rects, 1));
|
||||
assert!(repeated_cell_grid_overrides_bar_hypothesis(&rects));
|
||||
assert!(segmented_stacked_bar_geometry(&rects).is_none());
|
||||
assert!(!is_chart_bar_cluster(&[], &rects, 1));
|
||||
|
||||
let mut with_page_fills =
|
||||
vec![(0.0, 0.0, 600.0, 800.0); DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS];
|
||||
with_page_fills.extend(rects);
|
||||
assert!(!repeated_cell_grid_overrides_bar_hypothesis(
|
||||
&with_page_fills
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn touching_segments_with_spaced_rows_remain_a_chart() {
|
||||
let row_edges = [
|
||||
[100.0, 140.0, 180.0, 220.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 228.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 214.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 232.0, 260.0],
|
||||
];
|
||||
let mut raw_rects = vec![(90.0, 530.0, 190.0, 100.0)];
|
||||
for (row, edges) in row_edges.into_iter().enumerate() {
|
||||
let y = 540.0 + row as f32 * 20.0;
|
||||
raw_rects.extend(
|
||||
edges
|
||||
.windows(2)
|
||||
.map(|edge| (edge[0], y, edge[1] - edge[0], 12.0)),
|
||||
);
|
||||
}
|
||||
let items: Vec<TextItem> = (0..4)
|
||||
.map(|row| make_item("Category", 62.0, 541.0 + row as f32 * 20.0, 9.0))
|
||||
.collect();
|
||||
|
||||
assert!(is_repeated_cell_grid(&raw_rects));
|
||||
assert!(has_chart_bar_signature(&items, &raw_rects, 1));
|
||||
let geometry = segmented_stacked_bar_geometry(&raw_rects).expect("segmented stack");
|
||||
assert!(has_external_segmented_bar_labels(&items, 1, &geometry));
|
||||
assert!(is_chart_bar_cluster(&items, &raw_rects, 1));
|
||||
|
||||
let numeric_items: Vec<TextItem> = (0..4)
|
||||
.map(|row| make_item("2024", 80.0, 541.0 + row as f32 * 18.0, 9.0))
|
||||
.collect();
|
||||
assert!(has_external_segmented_bar_labels(
|
||||
&numeric_items,
|
||||
1,
|
||||
&geometry
|
||||
));
|
||||
assert!(is_chart_bar_cluster(&numeric_items, &raw_rects, 1));
|
||||
|
||||
let edge_adjacent_items: Vec<TextItem> = (0..4)
|
||||
.map(|row| make_item("2024", 92.0, 541.0 + row as f32 * 18.0, 9.0))
|
||||
.collect();
|
||||
assert!(has_external_segmented_bar_labels(
|
||||
&edge_adjacent_items,
|
||||
1,
|
||||
&geometry
|
||||
));
|
||||
assert!(is_chart_bar_cluster(&edge_adjacent_items, &raw_rects, 1));
|
||||
|
||||
let far_items: Vec<TextItem> = (0..4)
|
||||
.map(|row| make_item("Category", 20.0, 541.0 + row as f32 * 18.0, 9.0))
|
||||
.collect();
|
||||
assert!(!has_external_segmented_bar_labels(&far_items, 1, &geometry));
|
||||
assert!(!is_chart_bar_cluster(&far_items, &raw_rects, 1));
|
||||
|
||||
let rects: Vec<PdfRect> = raw_rects
|
||||
.into_iter()
|
||||
.map(|(x, y, width, height)| PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
page: 1,
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(detect_chart_regions(&items, &rects, 1).len(), 1);
|
||||
let (tables, hints) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert!(tables.is_empty());
|
||||
assert!(hints.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn padded_numeric_grid_frame_remains_a_table() {
|
||||
let row_edges = [
|
||||
[100.0, 140.0, 180.0, 220.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 228.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 214.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 232.0, 260.0],
|
||||
];
|
||||
let mut raw_rects = vec![(96.0, 536.0, 168.0, 80.0)];
|
||||
let mut items = Vec::new();
|
||||
for (row, edges) in row_edges.into_iter().enumerate() {
|
||||
let y = 540.0 + row as f32 * 20.0;
|
||||
for edge in edges.windows(2) {
|
||||
raw_rects.push((edge[0], y, edge[1] - edge[0], 12.0));
|
||||
items.push(make_item("42", edge[0] + 8.0, y + 1.0, 9.0));
|
||||
}
|
||||
}
|
||||
|
||||
assert!(is_repeated_cell_grid(&raw_rects));
|
||||
assert!(has_chart_bar_signature(&items, &raw_rects, 1));
|
||||
let geometry = segmented_stacked_bar_geometry(&raw_rects).expect("segmented rows");
|
||||
assert!(!has_external_segmented_bar_labels(&items, 1, &geometry));
|
||||
assert!(!is_chart_bar_cluster(&items, &raw_rects, 1));
|
||||
|
||||
let flush_items: Vec<TextItem> = (0..4)
|
||||
.map(|row| make_item("1", 100.0, 541.0 + row as f32 * 20.0, 9.0))
|
||||
.collect();
|
||||
assert!(!has_external_segmented_bar_labels(
|
||||
&flush_items,
|
||||
1,
|
||||
&geometry
|
||||
));
|
||||
assert!(!is_chart_bar_cluster(&flush_items, &raw_rects, 1));
|
||||
|
||||
let rects: Vec<PdfRect> = raw_rects
|
||||
.into_iter()
|
||||
.map(|(x, y, width, height)| PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
page: 1,
|
||||
})
|
||||
.collect();
|
||||
assert!(detect_chart_regions(&items, &rects, 1).is_empty());
|
||||
assert!(!detect_tables_from_rects(&items, &rects, 1).0.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn frameless_segmented_chart_with_category_labels_remains_a_chart() {
|
||||
let row_edges = [
|
||||
[100.0, 140.0, 180.0, 220.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 228.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 214.0, 260.0],
|
||||
[100.0, 140.0, 180.0, 232.0, 260.0],
|
||||
];
|
||||
let mut raw_rects = Vec::new();
|
||||
let mut items = Vec::new();
|
||||
for (row, edges) in row_edges.into_iter().enumerate() {
|
||||
let y = 540.0 + row as f32 * 18.0;
|
||||
raw_rects.extend(
|
||||
edges
|
||||
.windows(2)
|
||||
.map(|edge| (edge[0], y, edge[1] - edge[0], 12.0)),
|
||||
);
|
||||
items.push(make_item("Category", 62.0, y + 1.0, 9.0));
|
||||
}
|
||||
|
||||
let geometry = segmented_stacked_bar_geometry(&raw_rects).expect("segmented stack");
|
||||
assert!(has_external_segmented_bar_labels(&items, 1, &geometry));
|
||||
assert!(is_chart_bar_cluster(&items, &raw_rects, 1));
|
||||
|
||||
let rects: Vec<PdfRect> = raw_rects
|
||||
.into_iter()
|
||||
.map(|(x, y, width, height)| PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
page: 1,
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(detect_chart_regions(&items, &rects, 1).len(), 1);
|
||||
}
|
||||
|
||||
// --- detect_stacked_box_table ---
|
||||
|
||||
/// N stacked boxes at x=100, w=300, h=22, top-to-bottom from y=600.
|
||||
|
||||
+1
-27
@@ -436,8 +436,7 @@ pub(crate) fn recover_header_row(
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
!item.is_strikeout
|
||||
&& item.font_size > small_font_threshold
|
||||
item.font_size > small_font_threshold
|
||||
&& item.y > first_row_y
|
||||
&& item.y <= first_row_y + row_gap_limit
|
||||
})
|
||||
@@ -805,31 +804,6 @@ mod tests {
|
||||
assert_eq!(table.rows.len(), rows_before);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_skips_strikeout_candidates() {
|
||||
let mut old_col1 = make_item("Old Col1", 100.0, 520.0, 12.0);
|
||||
old_col1.is_strikeout = true;
|
||||
let mut old_col2 = make_item("Old Col2", 200.0, 520.0, 12.0);
|
||||
old_col2.is_strikeout = true;
|
||||
let all_items = vec![
|
||||
old_col1,
|
||||
old_col2,
|
||||
make_item("A", 100.0, 500.0, 8.0),
|
||||
make_item("B", 200.0, 500.0, 8.0),
|
||||
];
|
||||
let mut table = Table {
|
||||
columns: vec![100.0, 200.0],
|
||||
rows: vec![500.0, 480.0],
|
||||
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||
item_indices: vec![2, 3],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
recover_header_row(&mut table, &all_items, 9.0);
|
||||
assert_eq!(table.rows.len(), 2);
|
||||
assert_eq!(table.cells[0], vec!["A", "B"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_too_far_above() {
|
||||
let all_items = vec![
|
||||
|
||||
+2
-6
@@ -12,13 +12,9 @@ mod grid;
|
||||
pub mod structured;
|
||||
|
||||
pub use detect_heuristic::detect_tables;
|
||||
pub(crate) use detect_heuristic::{
|
||||
content_width, detect_tables_with_page_width, is_table_of_contents,
|
||||
};
|
||||
pub(crate) use detect_heuristic::is_table_of_contents;
|
||||
pub use detect_lines::detect_tables_from_lines;
|
||||
pub(crate) use detect_lines::{
|
||||
detect_dense_line_chart_regions, detect_vector_grid_tables_from_lines,
|
||||
};
|
||||
pub(crate) use detect_lines::detect_vector_grid_tables_from_lines;
|
||||
pub(crate) use detect_rects::cluster_rects;
|
||||
pub use detect_rects::{detect_chart_regions, detect_tables_from_rects, RectHintRegion};
|
||||
pub use detect_struct::detect_tables_from_struct_tree;
|
||||
|
||||
@@ -7,83 +7,6 @@
|
||||
use crate::types::TextItem;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
/// Return whether text is an explicit page-number expression.
|
||||
///
|
||||
/// This strict form is suitable before layout, where removing one numeric item
|
||||
/// from substantive text such as `Page 42 explains the result` would lose data.
|
||||
pub(crate) fn is_explicit_page_number_expression(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let is_number = |value: &str| {
|
||||
!value.is_empty() && value.chars().all(|character| character.is_ascii_digit())
|
||||
};
|
||||
|
||||
if trimmed.len() <= 4 && is_number(trimmed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if is_number(inner) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
let lowercase = trimmed.to_ascii_lowercase();
|
||||
if let Some(rest) = lowercase.strip_prefix("page") {
|
||||
let words: Vec<&str> = rest.split_whitespace().collect();
|
||||
if words.len() >= 3 && is_number(words[0]) && words[1] == "of" && is_number(words[2]) {
|
||||
return true;
|
||||
}
|
||||
if words.len() >= 2 && words[0] == "of" && is_number(words[1]) {
|
||||
return true;
|
||||
}
|
||||
return match words.as_slice() {
|
||||
[] | ["of"] => true,
|
||||
[number] => is_number(number),
|
||||
["of", total] => is_number(total),
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
};
|
||||
}
|
||||
|
||||
let words: Vec<&str> = lowercase.split_whitespace().collect();
|
||||
match words.as_slice() {
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return whether a completed Markdown line looks like a page number or a
|
||||
/// labeled running header.
|
||||
///
|
||||
/// At this stage the complete line and surrounding breaks are available, so a
|
||||
/// leading `Page N` remains compatible with the existing header cleanup even
|
||||
/// when the PDF appends a chapter or document title.
|
||||
pub(crate) fn is_page_number_line(text: &str) -> bool {
|
||||
if is_explicit_page_number_expression(text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let lowercase = text.trim().to_ascii_lowercase();
|
||||
lowercase.strip_prefix("page").is_some_and(|rest| {
|
||||
let mut characters = rest.trim_start().chars().peekable();
|
||||
let mut has_page_number = false;
|
||||
while characters
|
||||
.peek()
|
||||
.is_some_and(|character| character.is_ascii_digit())
|
||||
{
|
||||
has_page_number = true;
|
||||
characters.next();
|
||||
}
|
||||
|
||||
has_page_number && characters.next().is_none_or(char::is_whitespace)
|
||||
})
|
||||
}
|
||||
|
||||
/// Check if a character is CJK (Chinese, Japanese, Korean).
|
||||
/// CJK languages don't use spaces between words, so word-boundary
|
||||
/// heuristics should not apply when CJK characters are involved.
|
||||
|
||||
+11
-163
@@ -4,17 +4,11 @@
|
||||
|
||||
use log::{debug, warn};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::borrow::Cow;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::glyph_names::glyph_to_char;
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
static BUILTIN_CMAPS: include_dir::Dir<'_> =
|
||||
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
|
||||
|
||||
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ToUnicodeCMap {
|
||||
@@ -594,7 +588,7 @@ fn hex_to_unicode_string(hex: &str) -> Option<String> {
|
||||
|
||||
let bytes: Option<Vec<u8>> = (0..hex.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(hex.get(i..i + 2)?, 16).ok())
|
||||
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
|
||||
.collect();
|
||||
let bytes = bytes?;
|
||||
|
||||
@@ -880,25 +874,6 @@ fn try_remap_subset_cmap(
|
||||
None => return (cmap, None),
|
||||
};
|
||||
|
||||
// Both repair paths below assume CIDs are glyph indices that a subsetter can
|
||||
// renumber, which is only true for CIDFontType2 (TrueType). For CIDFontType0
|
||||
// (CFF), CIDs are resolved through the CFF charset, so a valid CMap stays valid
|
||||
// after subsetting and renumbering it corrupts otherwise-correct text.
|
||||
// CIDToGIDMap is likewise CIDFontType2-only (PDF 32000-1:2008, 9.7.4.2), so this
|
||||
// also ignores a CIDToGIDMap that a malformed producer attached to a CFF font.
|
||||
// /Subtype may be an indirect reference, so resolve it through the document.
|
||||
// Only bail out when the descendant is *explicitly* something other than
|
||||
// CIDFontType2: a missing or unresolvable /Subtype keeps the previous
|
||||
// behaviour rather than silently disabling the repair.
|
||||
let subtype = cid_font_dict.get(b"Subtype").ok().and_then(|o| match o {
|
||||
Object::Reference(r) => doc.get_object(*r).ok().and_then(|o| o.as_name().ok()),
|
||||
other => other.as_name().ok(),
|
||||
});
|
||||
if subtype.is_some_and(|name| name != b"CIDFontType2") {
|
||||
debug!("Subset remap skipped for obj={obj_num}: descendant is not CIDFontType2");
|
||||
return (cmap, None);
|
||||
}
|
||||
|
||||
// If there's an explicit CIDToGIDMap, build a repaired CMap using it.
|
||||
if let Some(cid_to_gid) = get_cid_to_gid_map(cid_font_dict, doc) {
|
||||
if let Some(repaired) = build_cmap_with_cid_to_gid_map(&cmap, &cid_to_gid) {
|
||||
@@ -1174,7 +1149,9 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
|
||||
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
|
||||
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
|
||||
let data = read_builtin_cmap_file(&name)?;
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(name);
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -1182,14 +1159,13 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
cmap.code_byte_length = 2;
|
||||
debug!(
|
||||
"Built-in CMap {}: char_map={} ranges={}",
|
||||
name,
|
||||
path.display(),
|
||||
cmap.char_map.len(),
|
||||
cmap.ranges.len()
|
||||
);
|
||||
Some(cmap)
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
|
||||
let p = PathBuf::from(dir);
|
||||
@@ -1206,18 +1182,6 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let path = find_bcmaps_dir()?.join(name);
|
||||
std::fs::read(path).ok().map(Cow::Owned)
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let file = BUILTIN_CMAPS.get_file(name)?;
|
||||
Some(Cow::Borrowed(file.contents()))
|
||||
}
|
||||
|
||||
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
|
||||
let mut stream = BinaryCMapStream::new(data);
|
||||
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
|
||||
@@ -1528,7 +1492,9 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
|
||||
}
|
||||
|
||||
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
parse_binary_cmap_encoding(&data).ok()
|
||||
}
|
||||
|
||||
@@ -1813,7 +1779,9 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
|
||||
if !name.ends_with("UCS2") {
|
||||
return None;
|
||||
}
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -2606,24 +2574,6 @@ endcmap
|
||||
assert_eq!(cmap.lookup(0x0025), Some("B".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_hex_to_unicode_non_ascii_no_panic() {
|
||||
// A destination containing a multi-byte char makes the byte length even
|
||||
// while a byte offset can land inside a char. Slicing must not panic;
|
||||
// it should be rejected gracefully.
|
||||
assert_eq!(hex_to_unicode_string("XéY"), None);
|
||||
assert_eq!(hex_to_unicode_string("\u{fffd}0"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_non_ascii_destination_no_panic() {
|
||||
// Crafted /ToUnicode CMap: a non-hex, non-ASCII destination previously
|
||||
// triggered a char-boundary panic in hex_to_unicode_string.
|
||||
let cmap_content = "beginbfchar <0041> <XéY> endbfchar";
|
||||
// Must not panic; the malformed entry is simply skipped.
|
||||
let _ = ToUnicodeCMap::parse(cmap_content.as_bytes());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_1byte() {
|
||||
// This is the pattern that caused the CJK bug: codespace is <0000><FFFF>
|
||||
@@ -3196,106 +3146,4 @@ endbfrange
|
||||
"Remap must fire when CMap's CIDs are outside W array coverage"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_try_remap_skipped_for_cid_font_type0() {
|
||||
// Same W/CMap mismatch as the CIDFontType2 case above, but the descendant is
|
||||
// CIDFontType0 (CFF). There CIDs are resolved through the CFF charset, so the
|
||||
// ToUnicode CIDs stay valid after subsetting and must not be renumbered.
|
||||
// Real-world case: Japanese Adobe-Japan1 PDFs (e.g. National Diet Library
|
||||
// minutes) where remapping turned correct text into unrelated glyphs.
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<0200> <0220> <0410>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
let mut doc = Document::new();
|
||||
|
||||
// CIDToGIDMap is CIDFontType2-only, but a malformed producer can still emit
|
||||
// one on a CFF font. Use a real stream (not /Identity, which is treated as
|
||||
// "no map") so this also fails if the guard is moved back below the
|
||||
// CIDToGIDMap branch: cid 1 -> gid 0x0200, which the CMap resolves.
|
||||
let mut cid_to_gid = vec![0u8; 68];
|
||||
cid_to_gid[2] = 0x02;
|
||||
cid_to_gid[3] = 0x00;
|
||||
let cid_to_gid_id =
|
||||
doc.add_object(lopdf::Stream::new(lopdf::Dictionary::new(), cid_to_gid));
|
||||
|
||||
let mut cid_font = lopdf::Dictionary::new();
|
||||
cid_font.set("Subtype", lopdf::Object::Name(b"CIDFontType0".to_vec()));
|
||||
cid_font.set("CIDToGIDMap", lopdf::Object::Reference(cid_to_gid_id));
|
||||
cid_font.set(
|
||||
"W",
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(1),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 34]),
|
||||
]),
|
||||
);
|
||||
let cid_font_id = doc.add_object(cid_font);
|
||||
|
||||
let mut font_dict = lopdf::Dictionary::new();
|
||||
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
|
||||
font_dict.set(
|
||||
"DescendantFonts",
|
||||
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
|
||||
);
|
||||
|
||||
let (primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 789);
|
||||
assert!(
|
||||
remapped.is_none(),
|
||||
"Remap must be skipped for CIDFontType0 (CFF) descendants, including a \
|
||||
CIDToGIDMap a malformed producer attached to one"
|
||||
);
|
||||
// The original CMap must still resolve its own CIDs.
|
||||
assert_eq!(primary.lookup(0x0200), Some("\u{0410}".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_try_remap_resolves_indirect_subtype() {
|
||||
// /Subtype may be stored as an indirect reference. A genuine CIDFontType2
|
||||
// font must still get the repair, so the guard has to dereference it rather
|
||||
// than treat the unresolved value as "not CIDFontType2".
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<0200> <0220> <0410>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
let mut doc = Document::new();
|
||||
let subtype_id = doc.add_object(lopdf::Object::Name(b"CIDFontType2".to_vec()));
|
||||
|
||||
let mut cid_font = lopdf::Dictionary::new();
|
||||
cid_font.set("Subtype", lopdf::Object::Reference(subtype_id));
|
||||
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
|
||||
cid_font.set(
|
||||
"W",
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 34]),
|
||||
]),
|
||||
);
|
||||
let cid_font_id = doc.add_object(cid_font);
|
||||
|
||||
let mut font_dict = lopdf::Dictionary::new();
|
||||
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
|
||||
font_dict.set(
|
||||
"DescendantFonts",
|
||||
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
|
||||
);
|
||||
|
||||
let (_primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 790);
|
||||
assert!(
|
||||
remapped.is_some(),
|
||||
"An indirect /Subtype naming CIDFontType2 must still reach the remap"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+10
-113
@@ -148,17 +148,14 @@ impl TextLine {
|
||||
self.text_with_formatting(false, false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic/decorative markdown formatting.
|
||||
///
|
||||
/// `format_decorations` enables both geometrically detected source
|
||||
/// decorations: underline (`<u>`) and strikeout (`<s>`).
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_decorations: bool,
|
||||
format_underline: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_decorations {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -168,7 +165,6 @@ impl TextLine {
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
let mut current_strikeout = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -194,16 +190,13 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes. Source decorations are exclusive:
|
||||
// `<u>`/`<s>` content stays free of `**`/`*` markers — consumers
|
||||
// (and the eval harnesses this feeds) match tag content literally,
|
||||
// and mixed nesting breaks that. A struck-and-underlined item is
|
||||
// emitted as struck text because deletion is the stronger semantic
|
||||
// distinction in redline documents.
|
||||
let item_strikeout = format_decorations && item.is_strikeout;
|
||||
let item_underline = format_decorations && item.is_underline && !item_strikeout;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline && !item_strikeout;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline && !item_strikeout;
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -218,10 +211,6 @@ impl TextLine {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
if current_strikeout && !item_strikeout {
|
||||
result.push_str("</s>");
|
||||
current_strikeout = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -233,10 +222,6 @@ impl TextLine {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_strikeout && !current_strikeout {
|
||||
result.push_str("<s>");
|
||||
current_strikeout = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -259,9 +244,6 @@ impl TextLine {
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
if current_strikeout {
|
||||
result.push_str("</s>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
@@ -327,88 +309,3 @@ impl TextLine {
|
||||
|| space_already_exists)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod formatting_tests {
|
||||
use super::{ItemType, TextItem, TextLine};
|
||||
|
||||
fn item(text: &str, x: f32, width: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y: 100.0,
|
||||
width,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn line(items: Vec<TextItem>) -> TextLine {
|
||||
TextLine {
|
||||
items,
|
||||
y: 100.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_emits_semantic_strikeout() {
|
||||
let line = line(vec![item("deleted", 10.0, 42.0, true)]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_closes_strikeout_before_live_text() {
|
||||
let line = line(vec![
|
||||
item("keep", 10.0, 24.0, false),
|
||||
item("remove", 40.0, 42.0, true),
|
||||
item("keep", 88.0, 24.0, false),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"keep <s>remove</s> keep"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_coalesces_adjacent_struck_items() {
|
||||
let line = line(vec![
|
||||
item("deleted", 10.0, 42.0, true),
|
||||
item("words", 58.0, 30.0, true),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted words</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strikeout_takes_precedence_over_other_styles() {
|
||||
let mut decorated = item("deleted", 10.0, 42.0, true);
|
||||
decorated.is_bold = true;
|
||||
decorated.is_italic = true;
|
||||
decorated.is_underline = true;
|
||||
let line = line(vec![decorated]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
assert_eq!(line.text(), "deleted");
|
||||
}
|
||||
}
|
||||
|
||||
-68
@@ -1,68 +0,0 @@
|
||||
%PDF-1.3
|
||||
%“Œ‹ž ReportLab Generated PDF document (opensource)
|
||||
1 0 obj
|
||||
<<
|
||||
/F1 2 0 R
|
||||
>>
|
||||
endobj
|
||||
2 0 obj
|
||||
<<
|
||||
/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font
|
||||
>>
|
||||
endobj
|
||||
3 0 obj
|
||||
<<
|
||||
/Contents 7 0 R /MediaBox [ 0 0 612 792 ] /Parent 6 0 R /Resources <<
|
||||
/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ]
|
||||
>> /Rotate 0 /Trans <<
|
||||
|
||||
>>
|
||||
/Type /Page
|
||||
>>
|
||||
endobj
|
||||
4 0 obj
|
||||
<<
|
||||
/PageMode /UseNone /Pages 6 0 R /Type /Catalog
|
||||
>>
|
||||
endobj
|
||||
5 0 obj
|
||||
<<
|
||||
/Author (anonymous) /CreationDate (D:20260803112923+00'00') /Creator (anonymous) /Keywords () /ModDate (D:20260803112923+00'00') /Producer (ReportLab PDF Library - \(opensource\))
|
||||
/Subject (unspecified) /Title (untitled) /Trapped /False
|
||||
>>
|
||||
endobj
|
||||
6 0 obj
|
||||
<<
|
||||
/Count 1 /Kids [ 3 0 R ] /Type /Pages
|
||||
>>
|
||||
endobj
|
||||
7 0 obj
|
||||
<<
|
||||
/Filter [ /ASCII85Decode /FlateDecode ] /Length 202
|
||||
>>
|
||||
stream
|
||||
GarW05mr9@&;9NOME,dW.,B;'jAjYq0S4Z`*D9aMA;]$5J)A/$3lESen1?F)ZJsa4$4&N%-%cs)#qW5EVhhbPiRDrAV>MC%.spto@CU"ZdipR'TtFiMR_%m*Hm$N%qL7a"ckkp9T/s[N2"Og377mP*M^akb2XQZ@'l*qT(9bVtDb5+)S&Q.#%E)<]Ao`TSk2AE'/E\fn~>endstream
|
||||
endobj
|
||||
xref
|
||||
0 8
|
||||
0000000000 65535 f
|
||||
0000000061 00000 n
|
||||
0000000092 00000 n
|
||||
0000000199 00000 n
|
||||
0000000392 00000 n
|
||||
0000000460 00000 n
|
||||
0000000721 00000 n
|
||||
0000000780 00000 n
|
||||
trailer
|
||||
<<
|
||||
/ID
|
||||
[<6d7ea1213c5974c78613d5d2a08423b5><6d7ea1213c5974c78613d5d2a08423b5>]
|
||||
% ReportLab generated PDF document -- digest (opensource)
|
||||
|
||||
/Info 5 0 R
|
||||
/Root 4 0 R
|
||||
/Size 8
|
||||
>>
|
||||
startxref
|
||||
9072
|
||||
%%EOF
|
||||
-79
File diff suppressed because one or more lines are too long
BIN
Binary file not shown.
Binary file not shown.
-1239
File diff suppressed because it is too large
Load Diff
+4
-475
@@ -8,13 +8,12 @@ use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_mem_with_options, process_pdf_with_options, to_markdown,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, PdfError, PdfOptions,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
@@ -41,11 +40,10 @@ fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [{media_box}] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>"
|
||||
),
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
let content = "BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
@@ -81,186 +79,6 @@ fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_recurring_contextual_folio_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R 7 0 R 9 0 R] /Count 4 >>",
|
||||
);
|
||||
for page_index in 0..4 {
|
||||
let page_id = 3 + page_index * 2;
|
||||
let content_id = page_id + 1;
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
page_id,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 11 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
),
|
||||
);
|
||||
let page_number = page_index + 1;
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 1 0 0 1 25 30 Tm ({page_number}) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Body page {page_number}) Tj ET"
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
content_id,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
}
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
11,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_malformed_unselected_page() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R] /Count 2 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
let content = "BT /F1 12 Tf 1 0 0 1 25 30 Tm (1) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Selected page text) Tj 0 -16 Td (More selected text) Tj 0 -16 Td (Still selected text) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 6 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Length 3 >>\nstream\nBI \nendstream",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
make_text_pdf(
|
||||
"BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET",
|
||||
"0 0 612 792",
|
||||
)
|
||||
}
|
||||
|
||||
fn make_digit_run_repro_pdf() -> Vec<u8> {
|
||||
let content = r#"BT
|
||||
/F1 12 Tf
|
||||
1 0 0 1 72 780 Tm (A\)) Tj
|
||||
1 0 0 1 96 780 Tm (The) Tj
|
||||
1 0 0 1 126 780 Tm (total) Tj
|
||||
1 0 0 1 166 780 Tm (of) Tj
|
||||
1 0 0 1 186 780 Tm (730) Tj
|
||||
1 0 0 1 220 780 Tm (seats) Tj
|
||||
1 0 0 1 262 780 Tm (was) Tj
|
||||
1 0 0 1 296 780 Tm (approved.) Tj
|
||||
1 0 0 1 72 755 Tm (B\)) Tj
|
||||
1 0 0 1 96 755 Tm (let) Tj
|
||||
1 0 0 1 120 755 Tm (log) Tj
|
||||
1 0 0 1 150 755 Tm (2) Tj
|
||||
1 0 0 1 164 755 Tm (=) Tj
|
||||
1 0 0 1 180 755 Tm (a) Tj
|
||||
1 0 0 1 72 720 Tm (C\) Control: The total of 730 seats was approved. let log 2 = a) Tj
|
||||
ET"#;
|
||||
make_text_pdf(content, "0 0 595 842")
|
||||
}
|
||||
|
||||
fn truncate_eof_marker(mut pdf: Vec<u8>) -> Vec<u8> {
|
||||
assert!(pdf.ends_with(b"%%EOF"));
|
||||
pdf.pop();
|
||||
@@ -512,21 +330,6 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
assert_eq!(lines[0].text(), "First Second Third");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_digit_only_text_runs_are_preserved_in_markdown() {
|
||||
let pdf = make_digit_run_repro_pdf();
|
||||
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract positioned text");
|
||||
assert!(items.iter().any(|item| item.text == "730"));
|
||||
assert!(items.iter().any(|item| item.text == "2"));
|
||||
|
||||
let result = process_pdf_mem(&pdf).expect("convert PDF to markdown");
|
||||
assert_eq!(
|
||||
result.markdown.expect("markdown output").trim(),
|
||||
"A) The total of 730 seats was approved.\nB) let log 2 = a\nC) Control: The total of 730 seats was approved. let log 2 = a"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MarkdownOptions Tests
|
||||
// ============================================================================
|
||||
@@ -744,30 +547,6 @@ fn test_markdown_from_items_page_breaks() {
|
||||
assert!(md.contains("Content on second page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_page_count_overload_includes_trailing_blank_pages_in_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
items.push(make_text_item(value, 25.0, 30.0, 12.0, page));
|
||||
items.push(make_text_item(
|
||||
"Company report footer",
|
||||
41.0,
|
||||
30.0,
|
||||
12.0,
|
||||
page,
|
||||
));
|
||||
}
|
||||
let options = MarkdownOptions {
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = to_markdown_from_items_with_rects_and_page_count(items, options, &[], 20);
|
||||
|
||||
assert!(md.contains("1 Company report footer"));
|
||||
assert!(md.contains("4 Company report footer"));
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Markdown From Lines Tests
|
||||
// ============================================================================
|
||||
@@ -1310,18 +1089,6 @@ fn test_snapshot_2013_app2() {
|
||||
assert_snapshot("2013-app2");
|
||||
}
|
||||
|
||||
/// First two pages of Shannon's "A Mathematical Theory of Communication"
|
||||
/// (1998 dvips 5.58 → Distiller 3 retypesetting). Canonical legacy-TeX PDF:
|
||||
/// non-embedded base-14 fonts with no /Widths (exercises the built-in AFM
|
||||
/// metrics fallback), Type3 PK bitmap math fonts with FontMatrix
|
||||
/// [1 0 0 -1 0 0] (exercises visual-size scaling), a two-line embedded drop
|
||||
/// cap, indent-only paragraph breaks, and display math that must not be
|
||||
/// detected as tables or headings.
|
||||
#[test]
|
||||
fn test_snapshot_shannon_entropy() {
|
||||
assert_snapshot("shannon-entropy-p1-2");
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Pages Needing OCR Tests
|
||||
// ============================================================================
|
||||
@@ -1429,77 +1196,6 @@ fn test_firecrawl_tagged_pdf_struct_tree() {
|
||||
assert_eq!(fence_count % 2, 0, "Code fences should be balanced");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_tagged_pdf_text_items_carry_mcid() {
|
||||
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
|
||||
assert!(
|
||||
items.iter().any(|i| i.mcid.is_some()),
|
||||
"Tagged PDF text items should carry Marked Content IDs"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_structure_elements_tagged_pdf() {
|
||||
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
|
||||
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
|
||||
assert!(!elements.is_empty(), "Tagged PDF should yield elements");
|
||||
assert!(
|
||||
elements.iter().any(|e| e.role == "H1"),
|
||||
"Should surface H1 heading roles"
|
||||
);
|
||||
assert!(
|
||||
elements.iter().all(|e| !e.role.is_empty()),
|
||||
"Every element should carry a role name"
|
||||
);
|
||||
|
||||
// Sorted by (page, mcid) for deterministic output
|
||||
assert!(
|
||||
elements
|
||||
.windows(2)
|
||||
.all(|w| (w[0].page, w[0].mcid) <= (w[1].page, w[1].mcid)),
|
||||
"Elements should be sorted by (page, mcid)"
|
||||
);
|
||||
|
||||
// The advertised join: (page, mcid) pairs must line up with the
|
||||
// mcid-carrying TextItems from positioned extraction, and joining the
|
||||
// H1 entries must recover non-empty heading text.
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
|
||||
let h1_refs: std::collections::HashSet<(u32, i64)> = elements
|
||||
.iter()
|
||||
.filter(|e| e.role == "H1")
|
||||
.map(|e| (e.page, e.mcid))
|
||||
.collect();
|
||||
let h1_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.mcid.is_some_and(|mcid| h1_refs.contains(&(i.page, mcid))))
|
||||
.map(|i| i.text.as_str())
|
||||
.collect();
|
||||
assert!(
|
||||
!h1_text.trim().is_empty(),
|
||||
"Joining H1 structure elements to text items should recover heading text"
|
||||
);
|
||||
|
||||
// Page filter is 1-indexed (matching TextItem.page) and equals the
|
||||
// corresponding subset of the full document result.
|
||||
let page1 = pdf_inspector::extract_structure_elements_mem(&buf, Some(&[1])).unwrap();
|
||||
assert!(!page1.is_empty(), "Page 1 should have elements");
|
||||
assert!(page1.iter().all(|e| e.page == 1));
|
||||
let full_page1_count = elements.iter().filter(|e| e.page == 1).count();
|
||||
assert_eq!(page1.len(), full_page1_count);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_structure_elements_untagged_pdf_empty() {
|
||||
let buf = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
|
||||
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
|
||||
assert!(
|
||||
elements.is_empty(),
|
||||
"Untagged PDF should yield no structure elements, got {:?}",
|
||||
elements
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_identity_h_no_tounicode_suppresses_garbage() {
|
||||
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
|
||||
@@ -3219,57 +2915,6 @@ fn test_extract_pages_markdown_basic() {
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = extract_pages_markdown_mem(&pdf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 4);
|
||||
for (index, page) in result.pages.iter().enumerate() {
|
||||
assert!(page.markdown.contains("Company report footer"));
|
||||
assert!(
|
||||
!page
|
||||
.markdown
|
||||
.contains(&format!("{} Company report footer", index + 1)),
|
||||
"recurring contextual folio survived on page {}: {}",
|
||||
index + 1,
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_process_pdf_page_filter_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
|
||||
assert!(markdown.contains("Company report footer"));
|
||||
assert!(!markdown.contains("1 Company report footer"), "{markdown}");
|
||||
assert!(markdown.contains("Body page 1"));
|
||||
assert!(!markdown.contains("Body page 2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_selected_page_ignores_context_only_extraction_failure() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
let pages = extract_pages_markdown_mem(&pdf, Some(&[0])).unwrap();
|
||||
assert_eq!(pages.pages.len(), 1);
|
||||
assert!(pages.pages[0].markdown.contains("Selected page text"));
|
||||
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
assert!(markdown.contains("Selected page text"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_requested_page_extraction_failure_remains_fatal() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
assert!(extract_pages_markdown_mem(&pdf, Some(&[1])).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
@@ -3997,65 +3642,6 @@ fn encrypted_pdf_decrypts_with_correct_password() {
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression for the #231 review finding: `extract_pages_markdown`'s
|
||||
/// `has_template_image` check must be gated the same way
|
||||
/// `classify_pdf`/`detect_pdf_type` gates it (image_count <= 1, few text
|
||||
/// ops, low alphanumeric diversity) — not treated as sufficient on its
|
||||
/// own. The fixture is a real text page with substantial, richly varied
|
||||
/// body text (>=50 Tj ops) drawn over a full-bleed background image
|
||||
/// (e.g. letterhead/watermark). Before the fix, has_template_image alone
|
||||
/// forced needs_ocr=true and discarded the page's clean markdown; now the
|
||||
/// page must extract normally.
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_does_not_ocr_text_page_with_watermark_image() {
|
||||
let buf = std::fs::read("tests/fixtures/text_page_with_watermark_image.pdf").unwrap();
|
||||
|
||||
let ext = extract_pages_markdown_mem(&buf, None).expect("fixture should extract");
|
||||
let page = &ext.pages[0];
|
||||
assert!(
|
||||
!page.needs_ocr,
|
||||
"a text page with substantial real text should not be routed to OCR \
|
||||
just because it has a background image"
|
||||
);
|
||||
assert!(
|
||||
page.markdown.contains("watermark"),
|
||||
"expected the page's real body text to be preserved, got: {:?}",
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression for the #231 review finding: `extract_pages_markdown` never
|
||||
/// checked `has_vector_text` at all, even though `detect_from_document`'s
|
||||
/// Mixed-type per-page routing always sends vector-outlined-text pages to
|
||||
/// OCR (outlined glyphs can't be extracted as text). A page with massive
|
||||
/// path ops (outlined decorative text) plus a short genuine caption would
|
||||
/// extract that caption cleanly — non-empty, non-garbled — so the
|
||||
/// existing empty/garbage-text checks alone couldn't catch it.
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_ocrs_page_with_vector_outlined_text() {
|
||||
let buf = std::fs::read("tests/fixtures/vector_outlined_text_with_caption.pdf").unwrap();
|
||||
|
||||
let cls = pdf_inspector::detector::detect_pdf_type_mem(&buf).expect("fixture should classify");
|
||||
assert!(
|
||||
cls.pages_needing_ocr.contains(&1),
|
||||
"classify_pdf should flag page 1 as needing OCR (vector-outlined text), got: {:?}",
|
||||
cls.pages_needing_ocr
|
||||
);
|
||||
|
||||
let ext = extract_pages_markdown_mem(&buf, None).expect("fixture should extract");
|
||||
let page = &ext.pages[0];
|
||||
assert!(
|
||||
page.needs_ocr,
|
||||
"extract_pages_markdown must agree with classify_pdf that this page needs OCR"
|
||||
);
|
||||
assert!(
|
||||
page.markdown.is_empty(),
|
||||
"a page flagged needs_ocr must not return markdown as if extraction were \
|
||||
trustworthy, got: {:?}",
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pdf_options_debug_redacts_password() {
|
||||
let opts = PdfOptions::new().password("secret123");
|
||||
@@ -4066,60 +3652,3 @@ fn pdf_options_debug_redacts_password() {
|
||||
);
|
||||
assert!(dbg.contains("REDACTED"), "expected redaction marker: {dbg}");
|
||||
}
|
||||
|
||||
/// Regression for #228: a `startxref` pointer corrupted to point at the
|
||||
/// wrong byte offset (a single flipped digit — a real, common writer bug)
|
||||
/// must not make the whole file unprocessable. The real classic xref table
|
||||
/// is still present and findable by scanning for the `xref` keyword; both
|
||||
/// pypdf and pdfium recover the same way. Before this fix, every entry
|
||||
/// point raised "Invalid PDF structure" on a file whose object data was
|
||||
/// otherwise completely intact.
|
||||
#[test]
|
||||
fn test_process_pdf_recovers_corrupted_startxref_pointer() {
|
||||
let result = process_pdf_with_options(
|
||||
"tests/fixtures/broken_startxref_pointer.pdf",
|
||||
PdfOptions::new(),
|
||||
)
|
||||
.expect("a corrupted startxref pointer should be recoverable, like pypdf/pdfium");
|
||||
|
||||
assert_eq!(result.page_count, 1);
|
||||
let md = result.markdown.unwrap_or_default();
|
||||
assert!(
|
||||
md.contains("Order Detail Report by Account") && md.contains("WIDGET ASSEMBLY"),
|
||||
"recovered document should extract its real text, got: {md:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression for #227: `extract_pages_markdown`'s per-page `needs_ocr`
|
||||
/// must agree with `classify_pdf`/`detect_pdf_type` on the same page. The
|
||||
/// fixture is a full-page raster "scan" with a single line of genuine
|
||||
/// native text drawn over it (a header) — the native text extracts
|
||||
/// perfectly cleanly (no decoding issues, non-empty), so a needs_ocr
|
||||
/// computation based on text-quality signals alone says `false`, while
|
||||
/// detection correctly sees a dominant background image and says the page
|
||||
/// needs OCR. Both must now agree it needs OCR, and the markdown must not
|
||||
/// be returned as if the extraction were trustworthy.
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_agrees_with_classify_on_scan_with_native_header() {
|
||||
let buf = std::fs::read("tests/fixtures/scan_with_native_header_text.pdf").unwrap();
|
||||
|
||||
let cls = pdf_inspector::detector::detect_pdf_type_mem(&buf).expect("fixture should classify");
|
||||
assert!(
|
||||
cls.pages_needing_ocr.contains(&1),
|
||||
"classify_pdf should flag page 1 as needing OCR (image-dominated), got: {:?}",
|
||||
cls.pages_needing_ocr
|
||||
);
|
||||
|
||||
let ext = extract_pages_markdown_mem(&buf, None).expect("fixture should extract");
|
||||
let page = &ext.pages[0];
|
||||
assert!(
|
||||
page.needs_ocr,
|
||||
"extract_pages_markdown must agree with classify_pdf that this page needs OCR"
|
||||
);
|
||||
assert!(
|
||||
page.markdown.is_empty(),
|
||||
"a page flagged needs_ocr must not return markdown as if extraction were \
|
||||
trustworthy, got: {:?}",
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
|
||||
@@ -56,7 +56,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form **4070** Employee’s Report (Rev. July 1996)
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
|
||||
## of Tips to EmployerOMB No. 1545-0065
|
||||
|
||||
@@ -81,3 +81,4 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
@@ -1,43 +0,0 @@
|
||||
Reprinted with corrections from *The Bell System Technical Journal,* Vol. 27, pp. 379–423, 623–656, July, October, 1948.
|
||||
|
||||
## A Mathematical Theory of Communication
|
||||
|
||||
### By C. E. SHANNON
|
||||
|
||||
INTRODUCTION
|
||||
|
||||
HE recent development of various methods of modulation such as PCM and PPM which exchange
|
||||
|
||||
# Tbandwidth for signal-to-noise ratio has intensified the interest in a general theory of communication. A
|
||||
|
||||
basis for such a theory is contained in the important papers of Nyquist¹ and Hartley² on this subject. In the present paper we will extend the theory to include a number of new factors, in particular the effect of noise in the channel, and the savings possible due to the statistical structure of the original message and due to the nature of the final destination of the information. The fundamental problem of communication is that of reproducing at one point either exactly or ap- proximately a message selected at another point. Frequently the messages have *meaning*; that is they refer to or are correlated according to some system with certain physical or conceptual entities. These semantic aspects of communication are irrelevant to the engineering problem. The significant aspect is that the actual message is one *selected from a set* of possible messages. The system must be designed to operate for each possible selection, not just the one which will actually be chosen since this is unknown at the time of design. If the number of messages in the set is finite then this number or any monotonic function of this number can be regarded as a measure of the information produced when one message is chosen from the set, all choices being equally likely. As was pointed out by Hartley the most natural choice is the logarithmic function. Although this definition must be generalized considerably when we consider the influence of the statistics of the message and when we have a continuous range of messages, we will in all cases use an essentially logarithmic measure. The logarithmic measure is more convenient for various reasons:
|
||||
|
||||
1. It is practically more useful. Parameters of engineering importance such as time, bandwidth, number of relays, etc., tend to vary linearly with the logarithm of the number of possibilities. For example, adding one relay to a group doubles the number of possible states of the relays. It adds 1 to the base 2 logarithm of this number. Doubling the time roughly squares the number of possible messages, or doubles the logarithm, etc.
|
||||
2. It is nearer to our intuitive feeling as to the proper measure. This is closely related to (1) since we in- tuitively measures entities by linear comparison with common standards. One feels, for example, that two punched cards should have twice the capacity of one for information storage, and two identical channels twice the capacity of one for transmitting information.
|
||||
3. It is mathematically more suitable. Many of the limiting operations are simple in terms of the loga- rithm but would require clumsy restatement in terms of the number of possibilities. The choice of a logarithmic base corresponds to the choice of a unit for measuring information. If the
|
||||
base 2 is used the resulting units may be called binary digits, or more briefly *bits,* a word suggested by
|
||||
|
||||
J. W. Tukey. A device with two stable positions, such as a relay or a flip-flop circuit, can store one bit of information. *N* such devices can store*N* bits, since the total number of possible states is 2
|
||||
*N* and log₂2 *N* = *N*. If the base 10 is used the units may be called decimal digits. Since
|
||||
|
||||
log₂*M* = log₁₀*M*= log₁₀2 = 3:32 log₁₀*M*;
|
||||
|
||||
1 Nyquist, H., “Certain Factors Affecting Telegraph Speed,” *Bell System Technical Journal,* April 1924, p. 324; “Certain Topics in Telegraph Transmission Theory,” *A.I.E.E. Trans.,* v. 47, April 1928, p. 617. 2 Hartley, R. V. L., “Transmission of Information,” *Bell System Technical Journal,* July 1928, p. 535.
|
||||
|
||||
INFORMATION SOURCE TRANSMITTER RECEIVER DESTINATION
|
||||
|
||||
SIGNAL RECEIVED SIGNAL MESSAGE MESSAGE
|
||||
|
||||
NOISE SOURCE
|
||||
|
||||
Fig. 1 — Schematic diagram of a general communication system.
|
||||
|
||||
a decimal digit is about 3 13 bits. A digit wheel on a desk computing machine has ten stable positions and therefore has a storage capacity of one decimal digit. In analytical work where integration and differentiation are involved the base *e* is sometimes useful. The resulting units of information will be called natural units. Change from the base *a* to base *b* merely requires multiplication by log*ba*. By a communication system we will mean a system of the type indicated schematically in Fig. 1. It consists of essentially five parts:
|
||||
|
||||
1. An *information source* which produces a message or sequence of messages to be communicated to the receiving terminal. The message may be of various types: (a) A sequence of letters as in a telegraph of teletype system; (b) A single function of time *f* (*t*) as in radio or telephony; (c) A function of time and other variables as in black and white television — here the message may be thought of as a function *f* (*x*; *y*;*t*) of two space coordinates and time, the light intensity at point (*x*; *y*) and time *t* on a pickup tube plate; (d) Two or more functions of time, say *f* (*t*), *g*(*t*), *h*(*t*) — this is the case in “three- dimensional” sound transmission or if the system is intended to service several individual channels in multiplex; (e) Several functions of several variables — in color television the message consists of three functions *f* (*x*; *y*;*t*), *g*(*x*; *y*;*t*), *h*(*x*; *y*;*t*) defined in a three-dimensional continuum — we may also think of these three functions as components of a vector field defined in the region — similarly, several black and white television sources would produce “messages” consisting of a number of functions of three variables; (f) Various combinations also occur, for example in television with an associated audio channel.
|
||||
2. A *transmitter* which operates on the message in some way to produce a signal suitable for trans- mission over the channel. In telephony this operation consists merely of changing sound pressure into a proportional electrical current. In telegraphy we have an encoding operation which produces a sequence of dots, dashes and spaces on the channel corresponding to the message. In a multiplex PCM system the different speech functions must be sampled, compressed, quantized and encoded, and finally interleaved properly to construct the signal. Vocoder systems, television and frequency modulation are other examples of complex operations applied to the message to obtain the signal.
|
||||
3. The *channel* is merely the medium used to transmit the signal from transmitter to receiver. It may be a pair of wires, a coaxial cable, a band of radio frequencies, a beam of light, etc.
|
||||
4. The *receiver* ordinarily performs the inverse operation of that done by the transmitter, reconstructing the message from the signal.
|
||||
5. The *destination* is the person (or thing) for whom the message is intended. We wish to consider certain general problems involving communication systems. To do this it is first
|
||||
necessary to represent the various elements involved as mathematical entities, suitably idealized from their
|
||||
|
||||
@@ -203,79 +203,6 @@ class TestExtractTextWithPositions:
|
||||
assert len(items) > 0
|
||||
assert all(item.page == 1 for item in items)
|
||||
|
||||
def test_mcid(self):
|
||||
# Untagged fixture: mcid is None or int, never anything else
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert all(item.mcid is None or isinstance(item.mcid, int) for item in items)
|
||||
# Tagged fixture: marked content carries MCIDs
|
||||
tagged = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("firecrawl_docs_tagged.pdf")
|
||||
)
|
||||
assert any(item.mcid is not None for item in tagged)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_structure_elements / extract_structure_elements_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractStructureElements:
|
||||
def test_tagged_file(self):
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("firecrawl_docs_tagged.pdf")
|
||||
)
|
||||
assert len(elements) > 0
|
||||
assert all(isinstance(e.page, int) for e in elements)
|
||||
assert all(isinstance(e.mcid, int) for e in elements)
|
||||
assert all(isinstance(e.role, str) and len(e.role) > 0 for e in elements)
|
||||
assert any(e.role == "H1" for e in elements)
|
||||
|
||||
def test_join_with_text_items(self):
|
||||
# (page, mcid) joins against extract_text_with_positions to recover
|
||||
# heading text
|
||||
path = fixture_path("firecrawl_docs_tagged.pdf")
|
||||
elements = pdf_inspector.extract_structure_elements(path)
|
||||
items = pdf_inspector.extract_text_with_positions(path)
|
||||
h1_refs = {(e.page, e.mcid) for e in elements if e.role == "H1"}
|
||||
h1_text = "".join(
|
||||
item.text
|
||||
for item in items
|
||||
if item.mcid is not None and (item.page, item.mcid) in h1_refs
|
||||
)
|
||||
assert len(h1_text.strip()) > 0
|
||||
|
||||
def test_with_pages(self):
|
||||
# pages filter is 1-indexed, matching TextItem.page
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("firecrawl_docs_tagged.pdf"), pages=[1]
|
||||
)
|
||||
assert len(elements) > 0
|
||||
assert all(e.page == 1 for e in elements)
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("firecrawl_docs_tagged.pdf")
|
||||
elements = pdf_inspector.extract_structure_elements_bytes(data)
|
||||
assert len(elements) > 0
|
||||
assert any(e.role == "H1" for e in elements)
|
||||
|
||||
def test_untagged_returns_empty(self):
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert elements == []
|
||||
|
||||
def test_repr(self):
|
||||
elements = pdf_inspector.extract_structure_elements(
|
||||
fixture_path("firecrawl_docs_tagged.pdf")
|
||||
)
|
||||
assert "StructureElement" in repr(elements[0])
|
||||
|
||||
def test_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_structure_elements_bytes(b"not a pdf")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text_in_regions / extract_text_in_regions_bytes
|
||||
|
||||
Generated
-1304
File diff suppressed because it is too large
Load Diff
@@ -1,37 +0,0 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "1.14.0"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "README.md"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[dependencies]
|
||||
console_error_panic_hook = "0.1"
|
||||
js-sys = "0.3"
|
||||
pdf-inspector = { path = ".." }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde-wasm-bindgen = "0.6"
|
||||
wasm-bindgen = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
wasm-bindgen-test = "0.3"
|
||||
|
||||
[profile.release]
|
||||
codegen-units = 1
|
||||
lto = true
|
||||
opt-level = "s"
|
||||
strip = true
|
||||
|
||||
[package.metadata.wasm-pack.profile.release]
|
||||
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
|
||||
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
|
||||
# size, and LTO optimizations above.
|
||||
wasm-opt = false
|
||||
@@ -1,58 +0,0 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
Third-party notices
|
||||
===================
|
||||
|
||||
Adobe CMaps
|
||||
-----------
|
||||
|
||||
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
|
||||
|
||||
Copyright 1990-2009 Adobe Systems Incorporated.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
Neither the name of Adobe Systems Incorporated nor the names of its
|
||||
contributors may be used to endorse or promote products derived from this
|
||||
software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -1,60 +0,0 @@
|
||||
# @firecrawl/pdf-inspector-wasm
|
||||
|
||||
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```ts
|
||||
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
|
||||
|
||||
await init();
|
||||
|
||||
const response = await fetch("/annual-report.pdf");
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
Pass options when you need selected pages or compact Markdown:
|
||||
|
||||
```ts
|
||||
const result = processPdf(pdf, {
|
||||
pages: [1, 3, 5],
|
||||
profile: "compact",
|
||||
includePageMarkers: true,
|
||||
});
|
||||
```
|
||||
|
||||
The package also exports:
|
||||
|
||||
- `detectPdf(pdf, options?)` for detection without extraction.
|
||||
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
|
||||
- `extractText(pdf)` for plain text.
|
||||
- `version()` for the WASM package version.
|
||||
|
||||
## Browser behavior
|
||||
|
||||
- Parsing runs locally. PDF bytes are not uploaded anywhere.
|
||||
- The build is single-threaded and does not require cross-origin isolation.
|
||||
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
|
||||
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
|
||||
- Image-only documents still require a separate OCR step.
|
||||
|
||||
## Build from source
|
||||
|
||||
```bash
|
||||
cargo install wasm-pack --version 0.15.0 --locked
|
||||
wasm-pack build wasm --target web --scope firecrawl --release
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
-440
@@ -1,440 +0,0 @@
|
||||
use pdf_inspector::{
|
||||
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
|
||||
ProcessMode,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wasm_bindgen::prelude::*;
|
||||
|
||||
#[wasm_bindgen(typescript_custom_section)]
|
||||
const TYPESCRIPT_TYPES: &str = r#"
|
||||
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
|
||||
export type MarkdownProfile = "fidelity" | "compact";
|
||||
|
||||
export interface ProcessOptions {
|
||||
/** Restrict extraction to these 1-indexed page numbers. */
|
||||
pages?: number[];
|
||||
/** Password for an encrypted PDF. */
|
||||
password?: string;
|
||||
/** Source-faithful output by default, or compact output for fewer tokens. */
|
||||
profile?: MarkdownProfile;
|
||||
/** Insert `<!-- Page N -->` markers between pages. */
|
||||
includePageMarkers?: boolean;
|
||||
/** Include image placeholders in Markdown output. */
|
||||
includeImages?: boolean;
|
||||
}
|
||||
|
||||
export interface PageOcrReasons {
|
||||
/** 1-indexed page number. */
|
||||
page: number;
|
||||
reasons: string[];
|
||||
}
|
||||
|
||||
export interface LayoutComplexity {
|
||||
isComplex: boolean;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithTables: number[];
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithColumns: number[];
|
||||
}
|
||||
|
||||
export interface PdfProcessResult {
|
||||
pdfType: PdfType;
|
||||
markdown?: string;
|
||||
pageCount: number;
|
||||
processingTimeMs: number;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesNeedingOcr: number[];
|
||||
ocrReasonsByPage: PageOcrReasons[];
|
||||
title?: string;
|
||||
confidence: number;
|
||||
layout: LayoutComplexity;
|
||||
hasEncodingIssues: boolean;
|
||||
}
|
||||
|
||||
export interface PdfClassification {
|
||||
pdfType: PdfType;
|
||||
pageCount: number;
|
||||
/** 0-indexed page numbers, matching the native Node.js API. */
|
||||
pagesNeedingOcr: number[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
|
||||
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
|
||||
export function classifyPdf(data: Uint8Array): PdfClassification;
|
||||
export function extractText(data: Uint8Array): string;
|
||||
export function version(): string;
|
||||
"#;
|
||||
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
|
||||
struct WasmProcessOptions {
|
||||
pages: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
profile: Option<WasmMarkdownProfile>,
|
||||
include_page_markers: Option<bool>,
|
||||
include_images: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
enum WasmMarkdownProfile {
|
||||
Fidelity,
|
||||
Compact,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPageOcrReasons {
|
||||
page: u32,
|
||||
reasons: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<PageOcrReasons> for WasmPageOcrReasons {
|
||||
fn from(value: PageOcrReasons) -> Self {
|
||||
Self {
|
||||
page: value.page,
|
||||
reasons: value.reasons,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmLayoutComplexity {
|
||||
is_complex: bool,
|
||||
pages_with_tables: Vec<u32>,
|
||||
pages_with_columns: Vec<u32>,
|
||||
}
|
||||
|
||||
impl From<LayoutComplexity> for WasmLayoutComplexity {
|
||||
fn from(value: LayoutComplexity) -> Self {
|
||||
Self {
|
||||
is_complex: value.is_complex,
|
||||
pages_with_tables: value.pages_with_tables,
|
||||
pages_with_columns: value.pages_with_columns,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfProcessResult {
|
||||
pdf_type: &'static str,
|
||||
markdown: Option<String>,
|
||||
page_count: u32,
|
||||
processing_time_ms: f64,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
|
||||
title: Option<String>,
|
||||
confidence: f64,
|
||||
layout: WasmLayoutComplexity,
|
||||
has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
impl From<PdfProcessResult> for WasmPdfProcessResult {
|
||||
fn from(value: PdfProcessResult) -> Self {
|
||||
Self {
|
||||
pdf_type: pdf_type_name(value.pdf_type),
|
||||
markdown: value.markdown,
|
||||
page_count: value.page_count,
|
||||
processing_time_ms: value.processing_time_ms as f64,
|
||||
pages_needing_ocr: value.pages_needing_ocr,
|
||||
ocr_reasons_by_page: value
|
||||
.ocr_reasons_by_page
|
||||
.into_iter()
|
||||
.map(Into::into)
|
||||
.collect(),
|
||||
title: value.title,
|
||||
confidence: value.confidence as f64,
|
||||
layout: value.layout.into(),
|
||||
has_encoding_issues: value.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfClassification {
|
||||
pdf_type: &'static str,
|
||||
page_count: u32,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
confidence: f64,
|
||||
}
|
||||
|
||||
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
|
||||
match pdf_type {
|
||||
PdfType::TextBased => "TextBased",
|
||||
PdfType::Scanned => "Scanned",
|
||||
PdfType::ImageBased => "ImageBased",
|
||||
PdfType::Mixed => "Mixed",
|
||||
}
|
||||
}
|
||||
|
||||
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
|
||||
js_sys::Error::new(&format!("{context}: {error}")).into()
|
||||
}
|
||||
|
||||
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
|
||||
if value.is_undefined() || value.is_null() {
|
||||
return Ok(WasmProcessOptions::default());
|
||||
}
|
||||
|
||||
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
|
||||
}
|
||||
|
||||
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
|
||||
let options = deserialize_options(value)?;
|
||||
if options
|
||||
.pages
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(js_error(
|
||||
"invalid options",
|
||||
"pages are 1-indexed; page 0 is invalid",
|
||||
));
|
||||
}
|
||||
|
||||
let mut result = PdfOptions::new().mode(mode);
|
||||
if let Some(pages) = options.pages {
|
||||
result = result.pages(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(profile) = options.profile {
|
||||
result.markdown.profile = match profile {
|
||||
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
|
||||
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
|
||||
};
|
||||
}
|
||||
if let Some(include_page_markers) = options.include_page_markers {
|
||||
result.markdown.include_page_numbers = include_page_markers;
|
||||
}
|
||||
if let Some(include_images) = options.include_images {
|
||||
result.markdown.include_images = include_images;
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
|
||||
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
|
||||
}
|
||||
|
||||
fn initialize() {
|
||||
console_error_panic_hook::set_once();
|
||||
}
|
||||
|
||||
/// Process PDF bytes entirely inside WebAssembly.
|
||||
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
|
||||
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::Full)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("process PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Classify PDF bytes without extracting text or producing Markdown.
|
||||
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
|
||||
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::DetectOnly)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("detect PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Return the lightweight classification shape used by the native Node API.
|
||||
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
|
||||
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
|
||||
serialize(&WasmPdfClassification {
|
||||
pdf_type: pdf_type_name(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes without Markdown conversion.
|
||||
#[wasm_bindgen(js_name = extractText, skip_typescript)]
|
||||
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
|
||||
initialize();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
|
||||
.map_err(|error| js_error("extract text", error))?;
|
||||
Ok(
|
||||
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
|
||||
.into_iter()
|
||||
.map(|line| line.text())
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n"),
|
||||
)
|
||||
}
|
||||
|
||||
/// Return the WebAssembly package version.
|
||||
#[wasm_bindgen(skip_typescript)]
|
||||
pub fn version() -> String {
|
||||
env!("CARGO_PKG_VERSION").to_string()
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_arch = "wasm32"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use js_sys::Reflect;
|
||||
use wasm_bindgen_test::*;
|
||||
|
||||
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
|
||||
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
|
||||
|
||||
fn synthetic_korea1_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
|
||||
// There is deliberately no ToUnicode stream: decoding must use the
|
||||
// embedded predefined CMap rather than lopdf's plain-text fallback.
|
||||
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
|
||||
// the bottom edge so they look exactly like a numeric page footer.
|
||||
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn processes_pdf_to_markdown() {
|
||||
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
|
||||
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn rejects_non_pdf_bytes() {
|
||||
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn classifies_and_extracts_plain_text() {
|
||||
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
|
||||
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let text = extract_text(TEXT_PDF).expect("extract text");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!text.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn extracts_cjk_and_preserves_numeric_page_footer() {
|
||||
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
|
||||
|
||||
assert_eq!(text, "가\n42");
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn opens_encrypted_pdf_with_password() {
|
||||
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
|
||||
|
||||
let options = js_sys::Object::new();
|
||||
Reflect::set(
|
||||
&options,
|
||||
&JsValue::from_str("password"),
|
||||
&JsValue::from_str("secret123"),
|
||||
)
|
||||
.expect("set password");
|
||||
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user