Compare commits
20
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a813de5561 | ||
|
|
217e1745fb | ||
|
|
04abab951f | ||
|
|
a410d5aa08 | ||
|
|
7747b3a086 | ||
|
|
ae6246ba0c | ||
|
|
2a7ad5891d | ||
|
|
3fb545284b | ||
|
|
1228a2c2ca | ||
|
|
1c32e4bd69 | ||
|
|
8121ae97ce | ||
|
|
98990cc550 | ||
|
|
a15ec2d68d | ||
|
|
7b7960ee73 | ||
|
|
7188667045 | ||
|
|
6ff104409a | ||
|
|
5e8f1570f6 | ||
|
|
b31e4b1727 | ||
|
|
3c6eb8bf6b | ||
|
|
5b287341a0 |
+55
-11
@@ -14,13 +14,15 @@ jobs:
|
||||
name: Test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
@@ -32,29 +34,34 @@ jobs:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: rustfmt
|
||||
|
||||
- name: Check formatting
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Check WASM formatting
|
||||
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
|
||||
|
||||
clippy:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: clippy
|
||||
|
||||
@@ -68,15 +75,52 @@ jobs:
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: build
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
key: wasm
|
||||
|
||||
- name: Check WebAssembly bindings
|
||||
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
|
||||
|
||||
- name: Check root package for WebAssembly
|
||||
run: cargo check --target wasm32-unknown-unknown
|
||||
|
||||
- name: Lint WebAssembly bindings
|
||||
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Test WebAssembly package
|
||||
run: wasm-pack test --node --release wasm
|
||||
|
||||
@@ -24,13 +24,13 @@ jobs:
|
||||
name: github-pages
|
||||
url: ${{ steps.deploy.outputs.page_url }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Upload site artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
|
||||
with:
|
||||
path: site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deploy
|
||||
uses: actions/deploy-pages@v4
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5.0.0
|
||||
|
||||
@@ -27,7 +27,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -85,17 +85,19 @@ jobs:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
uses: rust-lang/crates-io-auth-action@c6f97d42243bad5fab37ca0427f495c86d5b1a18 # v1.0.5
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -98,21 +98,21 @@ jobs:
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Build wheel
|
||||
uses: PyO3/maturin-action@v1
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: wheels-${{ matrix.target }}
|
||||
path: dist/*.whl
|
||||
@@ -124,16 +124,16 @@ jobs:
|
||||
name: Build sdist
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Build sdist
|
||||
uses: PyO3/maturin-action@v1
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
command: sdist
|
||||
args: --out dist
|
||||
|
||||
- name: Upload sdist
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: sdist
|
||||
path: dist/*.tar.gz
|
||||
@@ -149,7 +149,7 @@ jobs:
|
||||
id-token: write
|
||||
steps:
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
@@ -158,7 +158,7 @@ jobs:
|
||||
run: ls -la dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
with:
|
||||
packages-dir: dist
|
||||
# Tolerate already-uploaded files so a manual re-run can complete
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
name: Publish WebAssembly package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['wasm/Cargo.toml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
package_exists: ${{ steps.check.outputs.package_exists }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
|
||||
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
|
||||
echo "package_exists=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
echo "The initial package must be published once before trusted publishing can be configured."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "package_exists=true" >> "$GITHUB_OUTPUT"
|
||||
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
publish:
|
||||
name: Build and publish
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Build browser package
|
||||
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
|
||||
- name: Prepare package metadata
|
||||
run: |
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const path = "wasm/pkg/package.json"
|
||||
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
|
||||
pkg.name = "@firecrawl/pdf-inspector-wasm"
|
||||
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
|
||||
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
|
||||
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
|
||||
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
|
||||
pkg.publishConfig = { access: "public" }
|
||||
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
- name: Inspect package contents
|
||||
run: npm pack --dry-run ./wasm/pkg
|
||||
|
||||
- name: Publish package
|
||||
run: npm publish ./wasm/pkg --provenance --access public
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -61,31 +61,61 @@ jobs:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
# napi-cross builds gnu targets against an old glibc sysroot for
|
||||
# broad distro compatibility; musl targets cross-compile with
|
||||
# zig via cargo-zigbuild (napi's -x flag).
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
build-flags: --use-napi-cross
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: ${{ matrix.target }}
|
||||
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install zig
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: mlugg/setup-zig@d1434d08867e3ee9daa34448df10607b98908d29 # v2.2.1
|
||||
with:
|
||||
version: 0.14.1
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: taiki-e/install-action@67729d5c413db75907f0ad1e39bb04b9c868ff60 # v2.85.7
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
tool: cargo-zigbuild
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
~/.napi-rs
|
||||
napi/target/
|
||||
key: ${{ runner.os }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
key: ${{ matrix.target }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-napi-
|
||||
${{ matrix.target }}-cargo-napi-
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: napi
|
||||
@@ -93,10 +123,10 @@ jobs:
|
||||
|
||||
- name: Build native addon
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release
|
||||
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
|
||||
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
@@ -104,7 +134,7 @@ jobs:
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
@@ -112,23 +142,68 @@ jobs:
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
smoke-test:
|
||||
name: Smoke test ${{ matrix.target }}
|
||||
needs: [check-version, build]
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-gnu
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-musl
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Download native binary
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi
|
||||
|
||||
- name: Download generated JS bindings
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: napi
|
||||
|
||||
# musl binaries must load under a real musl libc, so run inside Alpine.
|
||||
- name: Run smoke test (Alpine)
|
||||
if: contains(matrix.target, 'musl')
|
||||
run: docker run --rm -v "$PWD:/repo" -w /repo/napi node:24-alpine node test.mjs
|
||||
|
||||
- name: Run smoke test
|
||||
if: ${{ !contains(matrix.target, 'musl') }}
|
||||
working-directory: napi
|
||||
run: node test.mjs
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: [check-version, build]
|
||||
needs: [check-version, build, smoke-test]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
@@ -154,9 +229,12 @@ jobs:
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"linux-x64-musl": { os: ["linux"], cpu: ["x64"], libc: ["musl"] },
|
||||
"linux-arm64-gnu": { os: ["linux"], cpu: ["arm64"], libc: ["glibc"] },
|
||||
"linux-arm64-musl": { os: ["linux"], cpu: ["arm64"], libc: ["musl"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
|
||||
+3
-1
@@ -1,10 +1,13 @@
|
||||
# Rust build artifacts
|
||||
/target/
|
||||
/wasm/target/
|
||||
/wasm/pkg/
|
||||
debug/
|
||||
*.pdb
|
||||
|
||||
# Cargo lock (optional for libraries)
|
||||
Cargo.lock
|
||||
!/wasm/Cargo.lock
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
@@ -39,4 +42,3 @@ test_output/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
+19
-13
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.6"
|
||||
version = "0.1.7"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
@@ -12,13 +12,13 @@ readme = "docs/rust-api.md"
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"src/**",
|
||||
"external/bcmaps/**",
|
||||
"docs/rust-api.md",
|
||||
"LICENSE",
|
||||
"/src/**",
|
||||
"/external/bcmaps/**",
|
||||
"/docs/rust-api.md",
|
||||
"/LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"pdf_inspector.pyi",
|
||||
"/pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
@@ -29,18 +29,11 @@ crate-type = ["lib", "cdylib"]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
# Parallel processing
|
||||
rayon = "1.10"
|
||||
|
||||
# Logging
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Text processing
|
||||
regex = "1.10"
|
||||
@@ -50,6 +43,19 @@ unicode-normalization = "0.1"
|
||||
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
||||
ttf-parser = "0.25"
|
||||
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.42.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.42.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.3"
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -19,6 +19,7 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
@@ -27,17 +28,17 @@ Evaluated on the [opendataloader-bench](https://github.com/opendataloader-projec
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Results were refreshed on July 16, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Speed is the median of three complete corpus runs.
|
||||
Results were refreshed on July 31, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Speed is the median of five alternating or rotating complete corpus runs after an excluded warm-up run, with each parser processing documents sequentially in a single process.
|
||||
|
||||
For context, engines that use OCR or model-based document parsing (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the top of that range without either, in 2.8 seconds.
|
||||
The complete parser configuration, per-document predictions, evaluator output, and generated charts are available in the [reproducible results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. pdf-inspector delivered the highest overall, reading-order, and table scores, along with the fastest complete run in this benchmark. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. In this comparison, pdf-inspector delivered the higher overall, reading-order, and table scores, along with the fastest complete run. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
|
||||
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
|
||||
|
||||
@@ -77,6 +78,26 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Browser WebAssembly
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
```javascript
|
||||
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
|
||||
|
||||
await init();
|
||||
const response = await fetch('/document.pdf');
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
> Full API reference: [wasm/README.md](wasm/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
@@ -188,6 +209,7 @@ src/
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
wasm/ — Browser bindings (wasm-bindgen)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
@@ -216,7 +238,7 @@ The converter handles:
|
||||
|---|---|
|
||||
| Headings (H1-H4) | Font size tiers relative to body text, with 0.5pt clustering |
|
||||
| Bold/italic | Font name patterns (Bold, Italic, Oblique) |
|
||||
| Bullet lists | `*`, `-`, `*`, `○`, `●`, `◦` prefixes |
|
||||
| Bullet lists | `•`, `-`, `*`, `○`, `●`, `◦` prefixes |
|
||||
| Numbered lists | `1.`, `1)`, `(1)` patterns |
|
||||
| Letter lists | `a.`, `a)`, `(a)` patterns |
|
||||
| Code blocks | Monospace fonts (Courier, Consolas, Monaco, Menlo, Fira Code, JetBrains Mono) and keyword detection |
|
||||
|
||||
@@ -30,11 +30,14 @@ overwrite one another.
|
||||
|
||||
## Published comparison protocol
|
||||
|
||||
The public benchmark table was refreshed on July 16, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Every engine processed the same 200
|
||||
PDFs with OCR disabled. Reported speed is the median of three complete corpus
|
||||
runs; quality scores come from the benchmark evaluator over all 200 outputs.
|
||||
The public benchmark table was refreshed on July 31, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Every engine processed the same 200
|
||||
PDFs sequentially in a single process with OCR disabled. Reported speed is the
|
||||
median of five alternating or rotating complete corpus runs after an excluded
|
||||
warm-up run; quality scores come from the benchmark evaluator over all 200
|
||||
outputs. Raw timings, predictions, evaluations, and charts are available in the
|
||||
[results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Optional backend evidence probe
|
||||
|
||||
|
||||
@@ -19,3 +19,20 @@ The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
|
||||
## Browser WebAssembly package
|
||||
|
||||
The browser package is published as `@firecrawl/pdf-inspector-wasm`. Its version lives in `wasm/Cargo.toml`, and `.github/workflows/publish-wasm.yml` builds the `web` target with `wasm-pack` before publishing the generated package.
|
||||
|
||||
The npm package must exist before a trusted publisher can be configured. For the first release only:
|
||||
|
||||
1. Build with `wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release`.
|
||||
2. Inspect with `npm pack --dry-run ./wasm/pkg`.
|
||||
3. Publish with `npm publish ./wasm/pkg --access public` from an authorized maintainer session.
|
||||
4. In the package settings on npm, configure the GitHub Actions trusted publisher:
|
||||
- Organization: `firecrawl`
|
||||
- Repository: `pdf-inspector`
|
||||
- Workflow: `publish-wasm.yml`
|
||||
- Allowed action: `npm publish`
|
||||
|
||||
After that one-time bootstrap, bumping the version in `wasm/Cargo.toml` and merging it to `main` publishes through OIDC. Until the package exists, the workflow exits cleanly without attempting an unauthenticated first publish. See npm's [trusted publishing documentation](https://docs.npmjs.com/trusted-publishers/) for the registry-side setup.
|
||||
|
||||
+20
-9
@@ -18,13 +18,13 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
@@ -111,7 +111,8 @@ class PdfResult: # process_pdf / detect_pdf
|
||||
markdown: str | None # extracted Markdown (None for detect_pdf)
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
title: str | None
|
||||
confidence: float # 0.0 - 1.0
|
||||
is_complex_layout: bool
|
||||
@@ -119,6 +120,10 @@ class PdfResult: # process_pdf / detect_pdf
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool # broken font encodings — consider OCR fallback
|
||||
|
||||
class PageOcrReasons: # per-page OCR diagnostics
|
||||
page: int # 1-indexed
|
||||
reasons: list[str] # machine-readable reason identifiers
|
||||
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
@@ -140,14 +145,20 @@ class TextItem: # extract_text_with_positions
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText: # extract_text_in_regions
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
ocr_reason: str | None # machine-readable OCR reason
|
||||
|
||||
class PageRegionTexts: # extract_text_in_regions
|
||||
page: int # 0-indexed
|
||||
regions: list[RegionText] # RegionText: text: str, needs_ocr: bool
|
||||
regions: list[RegionText]
|
||||
|
||||
class PagesExtractionResult: # extract_pages_markdown
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr, ocr_reason
|
||||
pages_with_tables: list[int] # 1-indexed
|
||||
pages_with_columns: list[int] # 1-indexed
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
is_complex: bool # any page has tables or multi-column layout
|
||||
```
|
||||
|
||||
+6
-6
@@ -18,13 +18,13 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
|
||||
Generated
+25
-3
@@ -499,11 +499,13 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"js-sys",
|
||||
"libc",
|
||||
"r-efi",
|
||||
"rand_core",
|
||||
"wasip2",
|
||||
"wasip3",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -557,6 +559,25 @@ version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
|
||||
|
||||
[[package]]
|
||||
name = "include_dir"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||
dependencies = [
|
||||
"include_dir_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir_macros"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.0"
|
||||
@@ -672,9 +693,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.41.0"
|
||||
version = "0.42.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -830,9 +851,10 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.6"
|
||||
version = "0.1.7"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
"log",
|
||||
"lopdf",
|
||||
"once_cell",
|
||||
|
||||
+13
-10
@@ -18,13 +18,13 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
@@ -34,7 +34,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries for **Linux x64**, **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -116,9 +116,12 @@ Prebuilt binaries ship as platform-specific packages installed automatically via
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| Linux | x64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-x64-musl` |
|
||||
| Linux | ARM64 (glibc) | `@firecrawl/pdf-inspector-linux-arm64-gnu` |
|
||||
| Linux | ARM64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-arm64-musl` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
+6
-3
@@ -8,9 +8,12 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+10
-4
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.11.1",
|
||||
"version": "1.12.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -37,6 +37,9 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-unknown-linux-musl",
|
||||
"aarch64-unknown-linux-gnu",
|
||||
"aarch64-unknown-linux-musl",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
]
|
||||
@@ -49,8 +52,11 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.1"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,9 @@ class PdfResult:
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed page numbers that need OCR."""
|
||||
ocr_reasons_by_page: list["PageOcrReasons"]
|
||||
"""Machine-readable OCR reasons by 1-indexed page."""
|
||||
title: Optional[str]
|
||||
confidence: float
|
||||
is_complex_layout: bool
|
||||
@@ -17,6 +20,13 @@ class PdfResult:
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool
|
||||
|
||||
class PageOcrReasons:
|
||||
"""OCR reasons for a single 1-indexed page."""
|
||||
page: int
|
||||
"""1-indexed page number."""
|
||||
reasons: list[str]
|
||||
"""Machine-readable OCR reason identifiers."""
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
@@ -47,6 +57,8 @@ class RegionText:
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
"""True when the text should not be trusted."""
|
||||
ocr_reason: Optional[str]
|
||||
"""Machine-readable OCR reason when the cause is known."""
|
||||
|
||||
class PageRegionTexts:
|
||||
"""Extracted text for one page's regions."""
|
||||
@@ -62,6 +74,8 @@ class PageMarkdown:
|
||||
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
|
||||
needs_ocr: bool
|
||||
"""True when text on this page is unreliable and OCR should be used instead."""
|
||||
ocr_reason: Optional[str]
|
||||
"""Machine-readable OCR reason when the cause is known."""
|
||||
|
||||
class PagesExtractionResult:
|
||||
"""Per-page markdown output with document-wide layout classification."""
|
||||
@@ -73,6 +87,8 @@ class PagesExtractionResult:
|
||||
"""1-indexed pages where multi-column layout was detected."""
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed pages that need OCR."""
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
"""Machine-readable OCR reasons by 1-indexed page."""
|
||||
is_complex: bool
|
||||
"""True if any page has tables or multi-column layout."""
|
||||
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ build-backend = "maturin"
|
||||
name = "pdf-inspector"
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.5"
|
||||
version = "0.2.6"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
+429
-4
@@ -302,6 +302,171 @@
|
||||
.text-link { color: var(--heat-dark); text-decoration: underline; text-decoration-color: var(--heat-16); text-underline-offset: 4px; }
|
||||
.text-link:hover, .text-link:focus-visible { text-decoration-color: var(--heat); }
|
||||
|
||||
.demo-shell {
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 12px;
|
||||
background: var(--surface);
|
||||
box-shadow: 0 22px 60px rgba(38,38,38,.07);
|
||||
overflow: hidden;
|
||||
}
|
||||
.demo-toolbar {
|
||||
min-height: 48px;
|
||||
padding: 0 17px;
|
||||
border-bottom: 1px solid var(--border);
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
gap: 18px;
|
||||
color: var(--ink-48);
|
||||
font: 11px/1.3 var(--mono);
|
||||
}
|
||||
.demo-engine,
|
||||
.demo-privacy { display: inline-flex; align-items: center; gap: 8px; }
|
||||
.demo-engine strong { color: var(--ink); font-weight: 500; }
|
||||
.demo-status-dot { width: 7px; height: 7px; border-radius: 50%; background: var(--heat); box-shadow: 0 0 0 4px var(--heat-8); }
|
||||
.demo-lock { color: var(--heat); font-size: 12px; }
|
||||
.demo-grid { display: grid; grid-template-columns: minmax(0, .86fr) minmax(0, 1.14fr); min-height: 500px; }
|
||||
.demo-input {
|
||||
padding: 28px;
|
||||
border-right: 1px solid var(--border);
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
background: var(--lighter);
|
||||
}
|
||||
.drop-zone {
|
||||
min-height: 286px;
|
||||
padding: 30px;
|
||||
border: 1px dashed var(--ink-16);
|
||||
border-radius: 10px;
|
||||
display: flex;
|
||||
flex: 1;
|
||||
flex-direction: column;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
text-align: center;
|
||||
background: var(--surface);
|
||||
cursor: pointer;
|
||||
transition: border-color .15s ease, background .15s ease, transform .15s ease;
|
||||
}
|
||||
.drop-zone:hover,
|
||||
.drop-zone:focus-visible,
|
||||
.drop-zone.is-dragging { border-color: var(--heat); background: var(--heat-4); }
|
||||
.drop-zone.is-dragging { transform: scale(.995); }
|
||||
.drop-mark {
|
||||
width: 52px;
|
||||
height: 58px;
|
||||
margin-bottom: 21px;
|
||||
border: 1px solid var(--ink-16);
|
||||
border-radius: 5px 5px 9px 5px;
|
||||
display: grid;
|
||||
place-items: center;
|
||||
position: relative;
|
||||
color: var(--heat-dark);
|
||||
background: var(--heat-4);
|
||||
font: 600 11px/1 var(--mono);
|
||||
letter-spacing: .03em;
|
||||
}
|
||||
.drop-mark::after {
|
||||
content: "";
|
||||
position: absolute;
|
||||
top: -1px;
|
||||
right: -1px;
|
||||
width: 14px;
|
||||
height: 14px;
|
||||
border-left: 1px solid var(--ink-16);
|
||||
border-bottom: 1px solid var(--ink-16);
|
||||
background: var(--surface);
|
||||
}
|
||||
.drop-zone strong { font-size: 18px; font-weight: 500; letter-spacing: -.02em; }
|
||||
.drop-zone > span:not(.drop-mark):not(.demo-choose) { margin-top: 7px; color: var(--ink-48); font-size: 13px; }
|
||||
.demo-choose { margin-top: 20px; color: var(--ink); }
|
||||
.demo-file {
|
||||
margin-top: 14px;
|
||||
padding: 13px 14px;
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
display: grid;
|
||||
grid-template-columns: minmax(0, 1fr) auto;
|
||||
gap: 4px 16px;
|
||||
background: var(--surface);
|
||||
}
|
||||
.demo-file strong { overflow: hidden; text-overflow: ellipsis; font: 500 12px/1.35 var(--mono); white-space: nowrap; }
|
||||
.demo-file span { grid-column: 1; color: var(--ink-32); font: 10px/1.3 var(--mono); }
|
||||
.demo-file button {
|
||||
grid-column: 2;
|
||||
grid-row: 1 / 3;
|
||||
padding: 0;
|
||||
border: 0;
|
||||
color: var(--ink-48);
|
||||
background: transparent;
|
||||
cursor: pointer;
|
||||
font: 11px/1 var(--mono);
|
||||
}
|
||||
.demo-file button:hover,
|
||||
.demo-file button:focus-visible { color: var(--heat-dark); }
|
||||
.demo-message { margin: 13px 2px 0; min-height: 19px; color: var(--ink-48); font: 11px/1.55 var(--mono); }
|
||||
.demo-message[data-tone="error"] { color: #a63520; }
|
||||
.demo-message[data-tone="success"] { color: #34725a; }
|
||||
.demo-output { min-width: 0; display: flex; flex-direction: column; background: var(--code); color: #f5f5f5; }
|
||||
.demo-output-head {
|
||||
min-height: 52px;
|
||||
padding: 0 18px;
|
||||
border-bottom: 1px solid rgba(255,255,255,.08);
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
color: rgba(255,255,255,.42);
|
||||
font: 11px/1 var(--mono);
|
||||
}
|
||||
.demo-copy {
|
||||
height: 29px;
|
||||
padding: 0 10px;
|
||||
border: 1px solid rgba(255,255,255,.12);
|
||||
border-radius: 6px;
|
||||
color: rgba(255,255,255,.68);
|
||||
background: rgba(255,255,255,.04);
|
||||
cursor: pointer;
|
||||
font: 10px/1 var(--mono);
|
||||
}
|
||||
.demo-copy:hover,
|
||||
.demo-copy:focus-visible { border-color: var(--heat); color: #fff; }
|
||||
.demo-copy[disabled] { opacity: .35; cursor: not-allowed; }
|
||||
.demo-empty {
|
||||
min-height: 448px;
|
||||
padding: 40px;
|
||||
display: grid;
|
||||
place-items: center;
|
||||
text-align: center;
|
||||
color: rgba(255,255,255,.28);
|
||||
font: 12px/1.7 var(--mono);
|
||||
}
|
||||
.demo-empty span { display: block; color: var(--heat); font-size: 22px; line-height: 1; }
|
||||
.demo-result { min-height: 0; flex: 1; display: flex; flex-direction: column; }
|
||||
.demo-result[hidden],
|
||||
.demo-file[hidden],
|
||||
.demo-empty[hidden] { display: none; }
|
||||
.demo-metrics {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(3, 1fr);
|
||||
border-bottom: 1px solid rgba(255,255,255,.08);
|
||||
}
|
||||
.demo-metric { min-width: 0; padding: 14px 17px; border-right: 1px solid rgba(255,255,255,.08); }
|
||||
.demo-metric:last-child { border-right: 0; }
|
||||
.demo-metric span { display: block; margin-bottom: 6px; color: rgba(255,255,255,.32); font: 9px/1 var(--mono); letter-spacing: .07em; text-transform: uppercase; }
|
||||
.demo-metric strong { display: block; overflow: hidden; text-overflow: ellipsis; color: #fff; font: 500 12px/1.25 var(--mono); white-space: nowrap; }
|
||||
.demo-metric:first-child strong { color: #ff9c70; }
|
||||
.markdown-output {
|
||||
min-height: 0;
|
||||
max-height: 364px;
|
||||
margin: 0;
|
||||
padding: 23px 20px 28px;
|
||||
flex: 1;
|
||||
overflow: auto;
|
||||
color: rgba(255,255,255,.82);
|
||||
font: 12px/1.7 var(--mono);
|
||||
white-space: pre-wrap;
|
||||
word-break: break-word;
|
||||
}
|
||||
.capability-grid { display: grid; grid-template-columns: repeat(3, 1fr); border: 1px solid var(--border); }
|
||||
.capability {
|
||||
min-height: 245px;
|
||||
@@ -409,6 +574,9 @@
|
||||
.hero-main { grid-template-columns: 1fr; gap: 44px; }
|
||||
.terminal { max-width: 580px; }
|
||||
.section-intro { grid-template-columns: 1fr; gap: 20px; }
|
||||
.demo-grid { grid-template-columns: 1fr; }
|
||||
.demo-input { border-right: 0; border-bottom: 1px solid var(--border); }
|
||||
.demo-empty { min-height: 360px; }
|
||||
.capability-grid { grid-template-columns: repeat(2, 1fr); }
|
||||
.capability:nth-child(3n) { border-right: 1px solid var(--border); }
|
||||
.capability:nth-child(2n) { border-right: 0; }
|
||||
@@ -429,6 +597,13 @@
|
||||
.package-row { grid-template-columns: 1fr; }
|
||||
.package { border-right: 0; border-bottom: 1px solid var(--border); }
|
||||
.package:last-child { border-bottom: 0; }
|
||||
.demo-toolbar { padding: 12px 14px; align-items: flex-start; flex-direction: column; gap: 8px; }
|
||||
.demo-input { padding: 18px; }
|
||||
.drop-zone { min-height: 250px; padding: 24px 18px; }
|
||||
.demo-metric { padding: 12px; }
|
||||
.demo-metric strong { font-size: 11px; }
|
||||
.demo-empty { min-height: 320px; padding: 28px; }
|
||||
.markdown-output { max-height: 340px; padding: 20px 16px 24px; font-size: 11px; }
|
||||
.capability-grid { grid-template-columns: 1fr; }
|
||||
.capability, .capability:nth-child(3n), .capability:nth-child(2n) { min-height: 210px; border-right: 0; border-bottom: 1px solid var(--border); }
|
||||
.capability:last-child { border-bottom: 0; }
|
||||
@@ -463,6 +638,7 @@
|
||||
</a>
|
||||
<div class="nav-links">
|
||||
<a href="#packages">Packages</a>
|
||||
<a href="#demo">Demo</a>
|
||||
<a href="#capabilities">Capabilities</a>
|
||||
<a href="#architecture">Architecture</a>
|
||||
<a href="#benchmark">Benchmark</a>
|
||||
@@ -484,6 +660,7 @@
|
||||
<p class="hero-copy">A Rust-powered, open-source parser that classifies PDFs and turns native text into clean, position-aware Markdown. Use it from Node.js or the bundled CLI, with packages also available from PyPI and crates.io.</p>
|
||||
<div class="hero-actions">
|
||||
<a class="button button-primary" href="https://github.com/firecrawl/pdf-inspector">View on GitHub <span class="arrow" aria-hidden="true">↗</span></a>
|
||||
<a class="button" href="#demo">Try it locally <span class="arrow" aria-hidden="true">↓</span></a>
|
||||
<a class="button" href="https://github.com/firecrawl/pdf-inspector#quick-start">Read the docs <span class="arrow" aria-hidden="true">→</span></a>
|
||||
</div>
|
||||
</div>
|
||||
@@ -538,9 +715,61 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<section id="demo" class="full-rule">
|
||||
<div class="shell">
|
||||
<div class="section-label pad"><span>[ <b>01</b> / 05 ]</span><span>Browser demo</span></div>
|
||||
<div class="section-body pad">
|
||||
<div class="section-intro">
|
||||
<h2>Try it in<br>your browser.</h2>
|
||||
<p>Drop in a native-text PDF to classify it and turn it into Markdown. The Rust core runs locally as WebAssembly in a background worker—your document never leaves this tab.</p>
|
||||
</div>
|
||||
<div class="demo-shell">
|
||||
<div class="demo-toolbar">
|
||||
<span class="demo-engine"><i class="demo-status-dot" aria-hidden="true"></i><strong>Rust core · WebAssembly</strong><span id="demo-engine-status">Loads on first run</span></span>
|
||||
<span class="demo-privacy"><span class="demo-lock" aria-hidden="true">◇</span>PDF bytes stay in your browser</span>
|
||||
</div>
|
||||
<div class="demo-grid">
|
||||
<div class="demo-input">
|
||||
<input id="pdf-input" type="file" accept=".pdf,application/pdf" hidden>
|
||||
<div class="drop-zone" id="drop-zone" role="button" tabindex="0" aria-controls="pdf-input" aria-label="Choose or drop a PDF file">
|
||||
<span class="drop-mark" aria-hidden="true">PDF</span>
|
||||
<strong>Drop a PDF here</strong>
|
||||
<span>Native-text documents · up to 25 MB</span>
|
||||
<span class="button demo-choose" aria-hidden="true">Choose PDF</span>
|
||||
</div>
|
||||
<div class="demo-file" id="demo-file" hidden>
|
||||
<strong id="demo-file-name"></strong>
|
||||
<span id="demo-file-size"></span>
|
||||
<button id="demo-clear" type="button">Remove</button>
|
||||
</div>
|
||||
<p class="demo-message" id="demo-message" role="status" aria-live="polite">Select a PDF to begin.</p>
|
||||
</div>
|
||||
<div class="demo-output" aria-label="PDF parsing result">
|
||||
<div class="demo-output-head">
|
||||
<span>MARKDOWN OUTPUT</span>
|
||||
<button class="demo-copy" id="demo-copy" type="button" disabled>Copy Markdown</button>
|
||||
</div>
|
||||
<div class="demo-empty" id="demo-empty">
|
||||
<p><span aria-hidden="true">_</span><br>Your parsed Markdown will appear here.</p>
|
||||
</div>
|
||||
<div class="demo-result" id="demo-result" hidden>
|
||||
<div class="demo-metrics">
|
||||
<div class="demo-metric"><span>Document type</span><strong id="demo-type">—</strong></div>
|
||||
<div class="demo-metric"><span>Pages</span><strong id="demo-pages">—</strong></div>
|
||||
<div class="demo-metric"><span>Processing</span><strong id="demo-time">—</strong></div>
|
||||
</div>
|
||||
<pre class="markdown-output" id="markdown-output"></pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section id="capabilities" class="full-rule">
|
||||
<div class="shell">
|
||||
<div class="section-label pad"><span>[ <b>01</b> / 04 ]</span><span>Core capabilities</span></div>
|
||||
<div class="section-label pad"><span>[ <b>02</b> / 05 ]</span><span>Core capabilities</span></div>
|
||||
<div class="section-body pad">
|
||||
<div class="section-intro">
|
||||
<h2>A focused PDF<br>toolchain.</h2>
|
||||
@@ -584,7 +813,7 @@
|
||||
|
||||
<section id="architecture" class="full-rule">
|
||||
<div class="shell">
|
||||
<div class="section-label pad"><span>[ <b>02</b> / 04 ]</span><span>Architecture</span></div>
|
||||
<div class="section-label pad"><span>[ <b>03</b> / 05 ]</span><span>Architecture</span></div>
|
||||
<div class="section-body pad">
|
||||
<div class="section-intro">
|
||||
<h2>One parse.<br>Clear stages.</h2>
|
||||
@@ -622,7 +851,7 @@
|
||||
|
||||
<section id="benchmark" class="full-rule">
|
||||
<div class="shell">
|
||||
<div class="section-label pad"><span>[ <b>03</b> / 04 ]</span><span>Benchmark</span></div>
|
||||
<div class="section-label pad"><span>[ <b>04</b> / 05 ]</span><span>Benchmark</span></div>
|
||||
<div class="section-body pad">
|
||||
<div class="section-intro">
|
||||
<h2>Measured on<br>real documents.</h2>
|
||||
@@ -656,7 +885,7 @@
|
||||
|
||||
<section id="usage" class="full-rule">
|
||||
<div class="shell">
|
||||
<div class="section-label pad"><span>[ <b>04</b> / 04 ]</span><span>Usage</span></div>
|
||||
<div class="section-label pad"><span>[ <b>05</b> / 05 ]</span><span>Usage</span></div>
|
||||
<div class="section-body pad">
|
||||
<div class="section-intro">
|
||||
<h2>Start with Node.<br>Use the CLI.</h2>
|
||||
@@ -743,5 +972,201 @@ result = pdf_inspector.<span class="fn">process_pdf</span>(<span class="str">"do
|
||||
</div>
|
||||
</footer>
|
||||
|
||||
<script>
|
||||
(() => {
|
||||
const MAX_FILE_SIZE = 25 * 1024 * 1024;
|
||||
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@0.1.1/pdf_inspector_wasm.js";
|
||||
const input = document.querySelector("#pdf-input");
|
||||
const dropZone = document.querySelector("#drop-zone");
|
||||
const filePanel = document.querySelector("#demo-file");
|
||||
const fileName = document.querySelector("#demo-file-name");
|
||||
const fileSize = document.querySelector("#demo-file-size");
|
||||
const clearButton = document.querySelector("#demo-clear");
|
||||
const copyButton = document.querySelector("#demo-copy");
|
||||
const message = document.querySelector("#demo-message");
|
||||
const engineStatus = document.querySelector("#demo-engine-status");
|
||||
const emptyOutput = document.querySelector("#demo-empty");
|
||||
const resultPanel = document.querySelector("#demo-result");
|
||||
const markdownOutput = document.querySelector("#markdown-output");
|
||||
const typeOutput = document.querySelector("#demo-type");
|
||||
const pagesOutput = document.querySelector("#demo-pages");
|
||||
const timeOutput = document.querySelector("#demo-time");
|
||||
let selectedFile = null;
|
||||
let currentMarkdown = "";
|
||||
let busy = false;
|
||||
|
||||
const formatBytes = (bytes) => {
|
||||
if (bytes < 1024) return `${bytes} B`;
|
||||
const units = ["KB", "MB", "GB"];
|
||||
let value = bytes / 1024;
|
||||
let unit = units[0];
|
||||
for (let index = 1; value >= 1024 && index < units.length; index += 1) {
|
||||
value /= 1024;
|
||||
unit = units[index];
|
||||
}
|
||||
return `${value >= 10 ? value.toFixed(1) : value.toFixed(2)} ${unit}`;
|
||||
};
|
||||
|
||||
const setMessage = (text, tone = "") => {
|
||||
message.textContent = text;
|
||||
message.dataset.tone = tone;
|
||||
};
|
||||
|
||||
const setBusy = (isBusy) => {
|
||||
busy = isBusy;
|
||||
clearButton.disabled = isBusy;
|
||||
input.disabled = isBusy;
|
||||
};
|
||||
|
||||
const clearResult = () => {
|
||||
currentMarkdown = "";
|
||||
copyButton.disabled = true;
|
||||
copyButton.textContent = "Copy Markdown";
|
||||
markdownOutput.textContent = "";
|
||||
resultPanel.hidden = true;
|
||||
emptyOutput.hidden = false;
|
||||
};
|
||||
|
||||
const clearFile = () => {
|
||||
if (busy) return;
|
||||
selectedFile = null;
|
||||
input.value = "";
|
||||
filePanel.hidden = true;
|
||||
clearResult();
|
||||
setMessage("Select a PDF to begin.");
|
||||
};
|
||||
|
||||
const selectFile = (file) => {
|
||||
if (!file) return;
|
||||
const isPdf = file.type === "application/pdf" || file.name.toLowerCase().endsWith(".pdf");
|
||||
if (!isPdf) {
|
||||
clearFile();
|
||||
setMessage("Choose a PDF file to run this demo.", "error");
|
||||
return;
|
||||
}
|
||||
if (file.size > MAX_FILE_SIZE) {
|
||||
clearFile();
|
||||
setMessage("This demo accepts PDFs up to 25 MB.", "error");
|
||||
return;
|
||||
}
|
||||
selectedFile = file;
|
||||
fileName.textContent = file.name;
|
||||
fileSize.textContent = formatBytes(file.size);
|
||||
filePanel.hidden = false;
|
||||
clearResult();
|
||||
parseSelectedFile();
|
||||
};
|
||||
|
||||
const processInWorker = (buffer) => new Promise((resolve, reject) => {
|
||||
const source = `
|
||||
import init, { processPdf, version } from "${WASM_MODULE_URL}";
|
||||
self.onmessage = async ({ data }) => {
|
||||
try {
|
||||
await init();
|
||||
const result = processPdf(new Uint8Array(data.buffer), { profile: "fidelity" });
|
||||
self.postMessage({ ok: true, result, engineVersion: version() });
|
||||
} catch (error) {
|
||||
const detail = error instanceof Error ? error.message : String(error);
|
||||
self.postMessage({ ok: false, error: detail });
|
||||
}
|
||||
};
|
||||
`;
|
||||
const workerUrl = URL.createObjectURL(new Blob([source], { type: "text/javascript" }));
|
||||
let worker;
|
||||
|
||||
try {
|
||||
worker = new Worker(workerUrl, { type: "module" });
|
||||
} catch (error) {
|
||||
URL.revokeObjectURL(workerUrl);
|
||||
reject(error);
|
||||
return;
|
||||
}
|
||||
|
||||
URL.revokeObjectURL(workerUrl);
|
||||
worker.addEventListener("message", ({ data }) => {
|
||||
worker.terminate();
|
||||
if (data.ok) resolve(data);
|
||||
else reject(new Error(data.error));
|
||||
}, { once: true });
|
||||
worker.addEventListener("error", (event) => {
|
||||
worker.terminate();
|
||||
reject(new Error(event.message || "The WebAssembly engine could not be loaded."));
|
||||
}, { once: true });
|
||||
worker.postMessage({ buffer }, [buffer]);
|
||||
});
|
||||
|
||||
const renderResult = ({ result, engineVersion }) => {
|
||||
currentMarkdown = typeof result.markdown === "string" ? result.markdown : "";
|
||||
typeOutput.textContent = result.pdfType || "Unknown";
|
||||
pagesOutput.textContent = String(result.pageCount ?? "—");
|
||||
timeOutput.textContent = Number.isFinite(result.processingTimeMs)
|
||||
? `${Math.max(1, Math.round(result.processingTimeMs))} ms`
|
||||
: "—";
|
||||
markdownOutput.textContent = currentMarkdown || "No Markdown output was produced for this document.";
|
||||
emptyOutput.hidden = true;
|
||||
resultPanel.hidden = false;
|
||||
copyButton.disabled = !currentMarkdown;
|
||||
engineStatus.textContent = engineVersion ? `Ready · v${engineVersion}` : "Ready";
|
||||
};
|
||||
|
||||
const parseSelectedFile = async () => {
|
||||
if (!selectedFile || busy) return;
|
||||
const file = selectedFile;
|
||||
setBusy(true);
|
||||
setMessage("Loading the Rust engine and parsing in this browser…");
|
||||
engineStatus.textContent = "Loading…";
|
||||
|
||||
try {
|
||||
const buffer = await file.arrayBuffer();
|
||||
const response = await processInWorker(buffer);
|
||||
renderResult(response);
|
||||
setMessage(`Parsed ${file.name} locally.`, "success");
|
||||
} catch (error) {
|
||||
engineStatus.textContent = "Load failed";
|
||||
clearResult();
|
||||
const detail = error instanceof Error ? error.message : String(error);
|
||||
setMessage(detail || "The PDF could not be parsed.", "error");
|
||||
} finally {
|
||||
setBusy(false);
|
||||
}
|
||||
};
|
||||
|
||||
dropZone.addEventListener("click", () => {
|
||||
if (!busy) {
|
||||
input.value = "";
|
||||
input.click();
|
||||
}
|
||||
});
|
||||
dropZone.addEventListener("keydown", (event) => {
|
||||
if ((event.key === "Enter" || event.key === " ") && !busy) {
|
||||
event.preventDefault();
|
||||
input.value = "";
|
||||
input.click();
|
||||
}
|
||||
});
|
||||
dropZone.addEventListener("dragover", (event) => {
|
||||
event.preventDefault();
|
||||
if (!busy) dropZone.classList.add("is-dragging");
|
||||
});
|
||||
dropZone.addEventListener("dragleave", () => dropZone.classList.remove("is-dragging"));
|
||||
dropZone.addEventListener("drop", (event) => {
|
||||
event.preventDefault();
|
||||
dropZone.classList.remove("is-dragging");
|
||||
if (!busy) selectFile(event.dataTransfer.files[0]);
|
||||
});
|
||||
input.addEventListener("change", () => selectFile(input.files[0]));
|
||||
clearButton.addEventListener("click", clearFile);
|
||||
copyButton.addEventListener("click", async () => {
|
||||
if (!currentMarkdown) return;
|
||||
try {
|
||||
await navigator.clipboard.writeText(currentMarkdown);
|
||||
copyButton.textContent = "Copied";
|
||||
window.setTimeout(() => { copyButton.textContent = "Copy Markdown"; }, 1600);
|
||||
} catch {
|
||||
setMessage("Copy is unavailable here. Select the Markdown output instead.", "error");
|
||||
}
|
||||
});
|
||||
})();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
||||
@@ -63,6 +63,7 @@ fn json_escape(s: &str) -> String {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
|
||||
+33
-5
@@ -2,8 +2,8 @@
|
||||
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
|
||||
PdfType, ProcessMode, TextItem,
|
||||
extract_text_with_positions_pages_with_password, process_pdf_with_options, LayoutComplexity,
|
||||
PdfOptions, PdfType, ProcessMode, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
use std::env;
|
||||
@@ -103,9 +103,18 @@ fn format_items_json(items: &[TextItem]) -> String {
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_items_json(
|
||||
pdf_path: &str,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<String, pdf_inspector::PdfError> {
|
||||
extract_text_with_positions_pages_with_password(pdf_path, page_filter, password)
|
||||
.map(|items| format_items_json(&items))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::format_items_json;
|
||||
use super::{extract_items_json, format_items_json};
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
@@ -137,6 +146,24 @@ mod tests {
|
||||
assert!(json.contains(r#""item_type":"text""#));
|
||||
assert!(json.contains(r#""mcid":7"#));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn items_json_uses_supplied_pdf_password() {
|
||||
let path = "tests/fixtures/encrypted-secret123.pdf";
|
||||
|
||||
let without_password = extract_items_json(path, None, None);
|
||||
assert!(
|
||||
without_password.is_err(),
|
||||
"encrypted fixture unexpectedly extracted without a password"
|
||||
);
|
||||
|
||||
let json = extract_items_json(path, None, Some("secret123"))
|
||||
.expect("correct password should decrypt positioned text");
|
||||
assert!(
|
||||
json.contains("Procurement"),
|
||||
"decrypted item JSON should contain fixture text, got {json}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
@@ -190,6 +217,7 @@ fn print_layout_info(layout: &LayoutComplexity) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
@@ -256,8 +284,8 @@ fn main() {
|
||||
});
|
||||
|
||||
if items_json_output {
|
||||
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
|
||||
Ok(items) => println!("{}", format_items_json(&items)),
|
||||
match extract_items_json(pdf_path, page_filter.as_ref(), password.as_deref()) {
|
||||
Ok(json) => println!("{}", json),
|
||||
Err(e) => {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
process::exit(1);
|
||||
|
||||
@@ -0,0 +1,649 @@
|
||||
//! Built-in glyph metrics for the 14 standard PDF fonts.
|
||||
//!
|
||||
//! PDFs may omit `/Widths` for non-embedded base-14 fonts (Times, Helvetica,
|
||||
//! Courier, Symbol, ZapfDingbats); per the PDF spec the reader must supply
|
||||
//! the metrics. Without them every text item gets width 0, which breaks
|
||||
//! space synthesis, sub/superscript detection, and table column detection
|
||||
//! (common in 1990s dvips/Distiller output).
|
||||
//!
|
||||
//! Tables are generated from the Adobe Core 14 AFM files (via reportlab's
|
||||
//! `_fontdata`), keyed by Unicode char, sorted for binary search.
|
||||
//! Generator: scratchpad/gen_base14.py (session tooling, not checked in).
|
||||
|
||||
/// Width in 1000ths of an em for `c` in the given base-14 font, or `None`
|
||||
/// if the font is not one of the base 14 (after name normalization) or the
|
||||
/// char has no glyph in its AFM.
|
||||
pub(crate) fn base14_char_width(base_font: &str, c: char) -> Option<u16> {
|
||||
let table = base14_table(base_font)?;
|
||||
// AFM tables key visible glyphs only; alias the invisible variants the
|
||||
// cp1252 fallback can produce so they get the metric of their visible
|
||||
// counterpart instead of the generic default.
|
||||
let c = match c {
|
||||
'\u{00A0}' => ' ', // no-break space -> space
|
||||
'\u{00AD}' => '-', // soft hyphen -> hyphen
|
||||
_ => c,
|
||||
};
|
||||
table
|
||||
.binary_search_by_key(&c, |&(ch, _)| ch)
|
||||
.ok()
|
||||
.map(|i| table[i].1)
|
||||
}
|
||||
|
||||
/// True when the base font name normalizes to one of the standard 14 fonts.
|
||||
pub(crate) fn is_base14_font(base_font: &str) -> bool {
|
||||
base14_table(base_font).is_some()
|
||||
}
|
||||
|
||||
/// Code → Unicode through the font's BUILT-IN encoding, for the base-14
|
||||
/// fonts whose repertoire is not Latin (Symbol, ZapfDingbats). Their glyphs
|
||||
/// live at byte positions that have nothing to do with cp1252 (Symbol 0x61
|
||||
/// renders α, Zapf 0x21 renders ✁), so advance widths must be resolved
|
||||
/// through this mapping — the renderer draws these glyphs regardless of how
|
||||
/// the text decoder transliterates them. Returns `None` for the Latin text
|
||||
/// fonts, which follow standard single-byte encodings.
|
||||
pub(crate) fn builtin_encoding_char(base_font: &str, code: u8) -> Option<char> {
|
||||
let table = base14_table(base_font)?;
|
||||
let enc: &[(u8, char)] = if std::ptr::eq(table, SYMBOL) {
|
||||
SYMBOL_ENCODING
|
||||
} else if std::ptr::eq(table, ZAPFDINGBATS) {
|
||||
ZAPFDINGBATS_ENCODING
|
||||
} else {
|
||||
return None;
|
||||
};
|
||||
enc.binary_search_by_key(&code, |&(b, _)| b)
|
||||
.ok()
|
||||
.map(|i| enc[i].1)
|
||||
}
|
||||
|
||||
/// Map a BaseFont name (possibly subset-prefixed, e.g. "ABCDEF+Times-Bold",
|
||||
/// or a common alias like "Arial" / "TimesNewRomanPSMT") to its width table.
|
||||
fn base14_table(base_font: &str) -> Option<&'static [(char, u16)]> {
|
||||
// Strip subset prefix "ABCDEF+"
|
||||
let name = match base_font.split_once('+') {
|
||||
Some((prefix, rest))
|
||||
if prefix.len() == 6 && prefix.chars().all(|c| c.is_ascii_uppercase()) =>
|
||||
{
|
||||
rest
|
||||
}
|
||||
_ => base_font,
|
||||
};
|
||||
let lower = name.to_ascii_lowercase();
|
||||
let bold = lower.contains("bold");
|
||||
let italic = lower.contains("italic") || lower.contains("oblique");
|
||||
if lower.contains("courier") {
|
||||
return Some(match (bold, italic) {
|
||||
(false, false) => COURIER,
|
||||
(true, false) => COURIER_BOLD,
|
||||
(false, true) => COURIER_OBLIQUE,
|
||||
(true, true) => COURIER_BOLDOBLIQUE,
|
||||
});
|
||||
}
|
||||
if lower.contains("helvetica") || lower.contains("arial") {
|
||||
return Some(match (bold, italic) {
|
||||
(false, false) => HELVETICA,
|
||||
(true, false) => HELVETICA_BOLD,
|
||||
(false, true) => HELVETICA_OBLIQUE,
|
||||
(true, true) => HELVETICA_BOLDOBLIQUE,
|
||||
});
|
||||
}
|
||||
if lower.contains("times") {
|
||||
return Some(match (bold, italic) {
|
||||
(false, false) => TIMES_ROMAN,
|
||||
(true, false) => TIMES_BOLD,
|
||||
(false, true) => TIMES_ITALIC,
|
||||
(true, true) => TIMES_BOLDITALIC,
|
||||
});
|
||||
}
|
||||
// Symbol and ZapfDingbats have unique glyph repertoires, so only exact
|
||||
// names (plus the common MT/ITC aliases) qualify — a custom font that
|
||||
// merely mentions "Symbol" in its name must not get these metrics.
|
||||
match lower.as_str() {
|
||||
"zapfdingbats" | "dingbats" | "itczapfdingbats" | "zapfdingbatsitc" => {
|
||||
return Some(ZAPFDINGBATS)
|
||||
}
|
||||
"symbol" | "symbolmt" | "symbolitc" => return Some(SYMBOL),
|
||||
_ => {}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[rustfmt::skip]
|
||||
static COURIER: &[(char, u16)] = &[
|
||||
(' ', 600), ('!', 600), ('"', 600), ('#', 600), ('$', 600), ('%', 600),
|
||||
('&', 600), ('\'', 600), ('(', 600), (')', 600), ('*', 600), ('+', 600),
|
||||
(',', 600), ('-', 600), ('.', 600), ('/', 600), ('0', 600), ('1', 600),
|
||||
('2', 600), ('3', 600), ('4', 600), ('5', 600), ('6', 600), ('7', 600),
|
||||
('8', 600), ('9', 600), (':', 600), (';', 600), ('<', 600), ('=', 600),
|
||||
('>', 600), ('?', 600), ('@', 600), ('A', 600), ('B', 600), ('C', 600),
|
||||
('D', 600), ('E', 600), ('F', 600), ('G', 600), ('H', 600), ('I', 600),
|
||||
('J', 600), ('K', 600), ('L', 600), ('M', 600), ('N', 600), ('O', 600),
|
||||
('P', 600), ('Q', 600), ('R', 600), ('S', 600), ('T', 600), ('U', 600),
|
||||
('V', 600), ('W', 600), ('X', 600), ('Y', 600), ('Z', 600), ('[', 600),
|
||||
('\\', 600), (']', 600), ('^', 600), ('_', 600), ('`', 600), ('a', 600),
|
||||
('b', 600), ('c', 600), ('d', 600), ('e', 600), ('f', 600), ('g', 600),
|
||||
('h', 600), ('i', 600), ('j', 600), ('k', 600), ('l', 600), ('m', 600),
|
||||
('n', 600), ('o', 600), ('p', 600), ('q', 600), ('r', 600), ('s', 600),
|
||||
('t', 600), ('u', 600), ('v', 600), ('w', 600), ('x', 600), ('y', 600),
|
||||
('z', 600), ('{', 600), ('|', 600), ('}', 600), ('~', 600), ('\u{00A1}', 600),
|
||||
('\u{00A2}', 600), ('\u{00A3}', 600), ('\u{00A4}', 600), ('\u{00A5}', 600), ('\u{00A6}', 600), ('\u{00A7}', 600),
|
||||
('\u{00A8}', 600), ('\u{00A9}', 600), ('\u{00AA}', 600), ('\u{00AB}', 600), ('\u{00AC}', 600), ('\u{00AE}', 600),
|
||||
('\u{00AF}', 600), ('\u{00B0}', 600), ('\u{00B1}', 600), ('\u{00B2}', 600), ('\u{00B3}', 600), ('\u{00B4}', 600),
|
||||
('\u{00B5}', 600), ('\u{00B6}', 600), ('\u{00B7}', 600), ('\u{00B8}', 600), ('\u{00B9}', 600), ('\u{00BA}', 600),
|
||||
('\u{00BB}', 600), ('\u{00BC}', 600), ('\u{00BD}', 600), ('\u{00BE}', 600), ('\u{00BF}', 600), ('\u{00C0}', 600),
|
||||
('\u{00C1}', 600), ('\u{00C2}', 600), ('\u{00C3}', 600), ('\u{00C4}', 600), ('\u{00C5}', 600), ('\u{00C6}', 600),
|
||||
('\u{00C7}', 600), ('\u{00C8}', 600), ('\u{00C9}', 600), ('\u{00CA}', 600), ('\u{00CB}', 600), ('\u{00CC}', 600),
|
||||
('\u{00CD}', 600), ('\u{00CE}', 600), ('\u{00CF}', 600), ('\u{00D0}', 600), ('\u{00D1}', 600), ('\u{00D2}', 600),
|
||||
('\u{00D3}', 600), ('\u{00D4}', 600), ('\u{00D5}', 600), ('\u{00D6}', 600), ('\u{00D7}', 600), ('\u{00D8}', 600),
|
||||
('\u{00D9}', 600), ('\u{00DA}', 600), ('\u{00DB}', 600), ('\u{00DC}', 600), ('\u{00DD}', 600), ('\u{00DE}', 600),
|
||||
('\u{00DF}', 600), ('\u{00E0}', 600), ('\u{00E1}', 600), ('\u{00E2}', 600), ('\u{00E3}', 600), ('\u{00E4}', 600),
|
||||
('\u{00E5}', 600), ('\u{00E6}', 600), ('\u{00E7}', 600), ('\u{00E8}', 600), ('\u{00E9}', 600), ('\u{00EA}', 600),
|
||||
('\u{00EB}', 600), ('\u{00EC}', 600), ('\u{00ED}', 600), ('\u{00EE}', 600), ('\u{00EF}', 600), ('\u{00F0}', 600),
|
||||
('\u{00F1}', 600), ('\u{00F2}', 600), ('\u{00F3}', 600), ('\u{00F4}', 600), ('\u{00F5}', 600), ('\u{00F6}', 600),
|
||||
('\u{00F7}', 600), ('\u{00F8}', 600), ('\u{00F9}', 600), ('\u{00FA}', 600), ('\u{00FB}', 600), ('\u{00FC}', 600),
|
||||
('\u{00FD}', 600), ('\u{00FE}', 600), ('\u{00FF}', 600), ('\u{0131}', 600), ('\u{0141}', 600), ('\u{0142}', 600),
|
||||
('\u{0152}', 600), ('\u{0153}', 600), ('\u{0160}', 600), ('\u{0161}', 600), ('\u{0178}', 600), ('\u{017D}', 600),
|
||||
('\u{017E}', 600), ('\u{0192}', 600), ('\u{02C6}', 600), ('\u{02C7}', 600), ('\u{02D8}', 600), ('\u{02D9}', 600),
|
||||
('\u{02DA}', 600), ('\u{02DB}', 600), ('\u{02DC}', 600), ('\u{02DD}', 600), ('\u{2013}', 600), ('\u{2014}', 600),
|
||||
('\u{2018}', 600), ('\u{2019}', 600), ('\u{201A}', 600), ('\u{201C}', 600), ('\u{201D}', 600), ('\u{201E}', 600),
|
||||
('\u{2020}', 600), ('\u{2021}', 600), ('\u{2022}', 600), ('\u{2026}', 600), ('\u{2030}', 600), ('\u{2039}', 600),
|
||||
('\u{203A}', 600), ('\u{2044}', 600), ('\u{20AC}', 600), ('\u{2122}', 600), ('\u{2212}', 600), ('\u{FB01}', 600),
|
||||
('\u{FB02}', 600),
|
||||
];
|
||||
|
||||
static COURIER_BOLD: &[(char, u16)] = COURIER;
|
||||
|
||||
static COURIER_OBLIQUE: &[(char, u16)] = COURIER;
|
||||
|
||||
static COURIER_BOLDOBLIQUE: &[(char, u16)] = COURIER;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA: &[(char, u16)] = &[
|
||||
(' ', 278), ('!', 278), ('"', 355), ('#', 556), ('$', 556), ('%', 889),
|
||||
('&', 667), ('\'', 191), ('(', 333), (')', 333), ('*', 389), ('+', 584),
|
||||
(',', 278), ('-', 333), ('.', 278), ('/', 278), ('0', 556), ('1', 556),
|
||||
('2', 556), ('3', 556), ('4', 556), ('5', 556), ('6', 556), ('7', 556),
|
||||
('8', 556), ('9', 556), (':', 278), (';', 278), ('<', 584), ('=', 584),
|
||||
('>', 584), ('?', 556), ('@', 1015), ('A', 667), ('B', 667), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 722), ('I', 278),
|
||||
('J', 500), ('K', 667), ('L', 556), ('M', 833), ('N', 722), ('O', 778),
|
||||
('P', 667), ('Q', 778), ('R', 722), ('S', 667), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 944), ('X', 667), ('Y', 667), ('Z', 611), ('[', 278),
|
||||
('\\', 278), (']', 278), ('^', 469), ('_', 556), ('`', 333), ('a', 556),
|
||||
('b', 556), ('c', 500), ('d', 556), ('e', 556), ('f', 278), ('g', 556),
|
||||
('h', 556), ('i', 222), ('j', 222), ('k', 500), ('l', 222), ('m', 833),
|
||||
('n', 556), ('o', 556), ('p', 556), ('q', 556), ('r', 333), ('s', 500),
|
||||
('t', 278), ('u', 556), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 500), ('{', 334), ('|', 260), ('}', 334), ('~', 584), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 556), ('\u{00A3}', 556), ('\u{00A4}', 556), ('\u{00A5}', 556), ('\u{00A6}', 260), ('\u{00A7}', 556),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 737), ('\u{00AA}', 370), ('\u{00AB}', 556), ('\u{00AC}', 584), ('\u{00AE}', 737),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 584), ('\u{00B2}', 333), ('\u{00B3}', 333), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 556), ('\u{00B6}', 537), ('\u{00B7}', 278), ('\u{00B8}', 333), ('\u{00B9}', 333), ('\u{00BA}', 365),
|
||||
('\u{00BB}', 556), ('\u{00BC}', 834), ('\u{00BD}', 834), ('\u{00BE}', 834), ('\u{00BF}', 611), ('\u{00C0}', 667),
|
||||
('\u{00C1}', 667), ('\u{00C2}', 667), ('\u{00C3}', 667), ('\u{00C4}', 667), ('\u{00C5}', 667), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 278),
|
||||
('\u{00CD}', 278), ('\u{00CE}', 278), ('\u{00CF}', 278), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 584), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 667), ('\u{00DE}', 667),
|
||||
('\u{00DF}', 611), ('\u{00E0}', 556), ('\u{00E1}', 556), ('\u{00E2}', 556), ('\u{00E3}', 556), ('\u{00E4}', 556),
|
||||
('\u{00E5}', 556), ('\u{00E6}', 889), ('\u{00E7}', 500), ('\u{00E8}', 556), ('\u{00E9}', 556), ('\u{00EA}', 556),
|
||||
('\u{00EB}', 556), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 556),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 556), ('\u{00F3}', 556), ('\u{00F4}', 556), ('\u{00F5}', 556), ('\u{00F6}', 556),
|
||||
('\u{00F7}', 584), ('\u{00F8}', 611), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 556), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 556), ('\u{0142}', 222),
|
||||
('\u{0152}', 1000), ('\u{0153}', 944), ('\u{0160}', 667), ('\u{0161}', 500), ('\u{0178}', 667), ('\u{017D}', 611),
|
||||
('\u{017E}', 500), ('\u{0192}', 556), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 556), ('\u{2014}', 1000),
|
||||
('\u{2018}', 222), ('\u{2019}', 222), ('\u{201A}', 222), ('\u{201C}', 333), ('\u{201D}', 333), ('\u{201E}', 333),
|
||||
('\u{2020}', 556), ('\u{2021}', 556), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 556), ('\u{2122}', 1000), ('\u{2212}', 584), ('\u{FB01}', 500),
|
||||
('\u{FB02}', 500),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA_BOLD: &[(char, u16)] = &[
|
||||
(' ', 278), ('!', 333), ('"', 474), ('#', 556), ('$', 556), ('%', 889),
|
||||
('&', 722), ('\'', 238), ('(', 333), (')', 333), ('*', 389), ('+', 584),
|
||||
(',', 278), ('-', 333), ('.', 278), ('/', 278), ('0', 556), ('1', 556),
|
||||
('2', 556), ('3', 556), ('4', 556), ('5', 556), ('6', 556), ('7', 556),
|
||||
('8', 556), ('9', 556), (':', 333), (';', 333), ('<', 584), ('=', 584),
|
||||
('>', 584), ('?', 611), ('@', 975), ('A', 722), ('B', 722), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 722), ('I', 278),
|
||||
('J', 556), ('K', 722), ('L', 611), ('M', 833), ('N', 722), ('O', 778),
|
||||
('P', 667), ('Q', 778), ('R', 722), ('S', 667), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 944), ('X', 667), ('Y', 667), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 584), ('_', 556), ('`', 333), ('a', 556),
|
||||
('b', 611), ('c', 556), ('d', 611), ('e', 556), ('f', 333), ('g', 611),
|
||||
('h', 611), ('i', 278), ('j', 278), ('k', 556), ('l', 278), ('m', 889),
|
||||
('n', 611), ('o', 611), ('p', 611), ('q', 611), ('r', 389), ('s', 556),
|
||||
('t', 333), ('u', 611), ('v', 556), ('w', 778), ('x', 556), ('y', 556),
|
||||
('z', 500), ('{', 389), ('|', 280), ('}', 389), ('~', 584), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 556), ('\u{00A3}', 556), ('\u{00A4}', 556), ('\u{00A5}', 556), ('\u{00A6}', 280), ('\u{00A7}', 556),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 737), ('\u{00AA}', 370), ('\u{00AB}', 556), ('\u{00AC}', 584), ('\u{00AE}', 737),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 584), ('\u{00B2}', 333), ('\u{00B3}', 333), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 611), ('\u{00B6}', 556), ('\u{00B7}', 278), ('\u{00B8}', 333), ('\u{00B9}', 333), ('\u{00BA}', 365),
|
||||
('\u{00BB}', 556), ('\u{00BC}', 834), ('\u{00BD}', 834), ('\u{00BE}', 834), ('\u{00BF}', 611), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 278),
|
||||
('\u{00CD}', 278), ('\u{00CE}', 278), ('\u{00CF}', 278), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 584), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 667), ('\u{00DE}', 667),
|
||||
('\u{00DF}', 611), ('\u{00E0}', 556), ('\u{00E1}', 556), ('\u{00E2}', 556), ('\u{00E3}', 556), ('\u{00E4}', 556),
|
||||
('\u{00E5}', 556), ('\u{00E6}', 889), ('\u{00E7}', 556), ('\u{00E8}', 556), ('\u{00E9}', 556), ('\u{00EA}', 556),
|
||||
('\u{00EB}', 556), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 611),
|
||||
('\u{00F1}', 611), ('\u{00F2}', 611), ('\u{00F3}', 611), ('\u{00F4}', 611), ('\u{00F5}', 611), ('\u{00F6}', 611),
|
||||
('\u{00F7}', 584), ('\u{00F8}', 611), ('\u{00F9}', 611), ('\u{00FA}', 611), ('\u{00FB}', 611), ('\u{00FC}', 611),
|
||||
('\u{00FD}', 556), ('\u{00FE}', 611), ('\u{00FF}', 556), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 1000), ('\u{0153}', 944), ('\u{0160}', 667), ('\u{0161}', 556), ('\u{0178}', 667), ('\u{017D}', 611),
|
||||
('\u{017E}', 500), ('\u{0192}', 556), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 556), ('\u{2014}', 1000),
|
||||
('\u{2018}', 278), ('\u{2019}', 278), ('\u{201A}', 278), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 556), ('\u{2021}', 556), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 556), ('\u{2122}', 1000), ('\u{2212}', 584), ('\u{FB01}', 611),
|
||||
('\u{FB02}', 611),
|
||||
];
|
||||
|
||||
static HELVETICA_OBLIQUE: &[(char, u16)] = HELVETICA;
|
||||
|
||||
static HELVETICA_BOLDOBLIQUE: &[(char, u16)] = HELVETICA_BOLD;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_ROMAN: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('"', 408), ('#', 500), ('$', 500), ('%', 833),
|
||||
('&', 778), ('\'', 180), ('(', 333), (')', 333), ('*', 500), ('+', 564),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 278), (';', 278), ('<', 564), ('=', 564),
|
||||
('>', 564), ('?', 444), ('@', 921), ('A', 722), ('B', 667), ('C', 667),
|
||||
('D', 722), ('E', 611), ('F', 556), ('G', 722), ('H', 722), ('I', 333),
|
||||
('J', 389), ('K', 722), ('L', 611), ('M', 889), ('N', 722), ('O', 722),
|
||||
('P', 556), ('Q', 722), ('R', 667), ('S', 556), ('T', 611), ('U', 722),
|
||||
('V', 722), ('W', 944), ('X', 722), ('Y', 722), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 469), ('_', 500), ('`', 333), ('a', 444),
|
||||
('b', 500), ('c', 444), ('d', 500), ('e', 444), ('f', 333), ('g', 500),
|
||||
('h', 500), ('i', 278), ('j', 278), ('k', 500), ('l', 278), ('m', 778),
|
||||
('n', 500), ('o', 500), ('p', 500), ('q', 500), ('r', 333), ('s', 389),
|
||||
('t', 278), ('u', 500), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 444), ('{', 480), ('|', 200), ('}', 480), ('~', 541), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 200), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 760), ('\u{00AA}', 276), ('\u{00AB}', 500), ('\u{00AC}', 564), ('\u{00AE}', 760),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 564), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 500), ('\u{00B6}', 453), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 310),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 444), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 889),
|
||||
('\u{00C7}', 667), ('\u{00C8}', 611), ('\u{00C9}', 611), ('\u{00CA}', 611), ('\u{00CB}', 611), ('\u{00CC}', 333),
|
||||
('\u{00CD}', 333), ('\u{00CE}', 333), ('\u{00CF}', 333), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 722),
|
||||
('\u{00D3}', 722), ('\u{00D4}', 722), ('\u{00D5}', 722), ('\u{00D6}', 722), ('\u{00D7}', 564), ('\u{00D8}', 722),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 722), ('\u{00DE}', 556),
|
||||
('\u{00DF}', 500), ('\u{00E0}', 444), ('\u{00E1}', 444), ('\u{00E2}', 444), ('\u{00E3}', 444), ('\u{00E4}', 444),
|
||||
('\u{00E5}', 444), ('\u{00E6}', 667), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 500), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 564), ('\u{00F8}', 500), ('\u{00F9}', 500), ('\u{00FA}', 500), ('\u{00FB}', 500), ('\u{00FC}', 500),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 500), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 889), ('\u{0153}', 722), ('\u{0160}', 556), ('\u{0161}', 389), ('\u{0178}', 722), ('\u{017D}', 611),
|
||||
('\u{017E}', 444), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 1000),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 444), ('\u{201D}', 444), ('\u{201E}', 444),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 980), ('\u{2212}', 564), ('\u{FB01}', 556),
|
||||
('\u{FB02}', 556),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_BOLD: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('"', 555), ('#', 500), ('$', 500), ('%', 1000),
|
||||
('&', 833), ('\'', 278), ('(', 333), (')', 333), ('*', 500), ('+', 570),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 333), (';', 333), ('<', 570), ('=', 570),
|
||||
('>', 570), ('?', 500), ('@', 930), ('A', 722), ('B', 667), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 778), ('I', 389),
|
||||
('J', 500), ('K', 778), ('L', 667), ('M', 944), ('N', 722), ('O', 778),
|
||||
('P', 611), ('Q', 778), ('R', 722), ('S', 556), ('T', 667), ('U', 722),
|
||||
('V', 722), ('W', 1000), ('X', 722), ('Y', 722), ('Z', 667), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 581), ('_', 500), ('`', 333), ('a', 500),
|
||||
('b', 556), ('c', 444), ('d', 556), ('e', 444), ('f', 333), ('g', 500),
|
||||
('h', 556), ('i', 278), ('j', 333), ('k', 556), ('l', 278), ('m', 833),
|
||||
('n', 556), ('o', 500), ('p', 556), ('q', 556), ('r', 444), ('s', 389),
|
||||
('t', 333), ('u', 556), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 444), ('{', 394), ('|', 220), ('}', 394), ('~', 520), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 220), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 747), ('\u{00AA}', 300), ('\u{00AB}', 500), ('\u{00AC}', 570), ('\u{00AE}', 747),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 570), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 556), ('\u{00B6}', 540), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 330),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 500), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 389),
|
||||
('\u{00CD}', 389), ('\u{00CE}', 389), ('\u{00CF}', 389), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 570), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 722), ('\u{00DE}', 611),
|
||||
('\u{00DF}', 556), ('\u{00E0}', 500), ('\u{00E1}', 500), ('\u{00E2}', 500), ('\u{00E3}', 500), ('\u{00E4}', 500),
|
||||
('\u{00E5}', 500), ('\u{00E6}', 722), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 570), ('\u{00F8}', 500), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 556), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 667), ('\u{0142}', 278),
|
||||
('\u{0152}', 1000), ('\u{0153}', 722), ('\u{0160}', 556), ('\u{0161}', 389), ('\u{0178}', 722), ('\u{017D}', 667),
|
||||
('\u{017E}', 444), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 1000),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 1000), ('\u{2212}', 570), ('\u{FB01}', 556),
|
||||
('\u{FB02}', 556),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_ITALIC: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('"', 420), ('#', 500), ('$', 500), ('%', 833),
|
||||
('&', 778), ('\'', 214), ('(', 333), (')', 333), ('*', 500), ('+', 675),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 333), (';', 333), ('<', 675), ('=', 675),
|
||||
('>', 675), ('?', 500), ('@', 920), ('A', 611), ('B', 611), ('C', 667),
|
||||
('D', 722), ('E', 611), ('F', 611), ('G', 722), ('H', 722), ('I', 333),
|
||||
('J', 444), ('K', 667), ('L', 556), ('M', 833), ('N', 667), ('O', 722),
|
||||
('P', 611), ('Q', 722), ('R', 611), ('S', 500), ('T', 556), ('U', 722),
|
||||
('V', 611), ('W', 833), ('X', 611), ('Y', 556), ('Z', 556), ('[', 389),
|
||||
('\\', 278), (']', 389), ('^', 422), ('_', 500), ('`', 333), ('a', 500),
|
||||
('b', 500), ('c', 444), ('d', 500), ('e', 444), ('f', 278), ('g', 500),
|
||||
('h', 500), ('i', 278), ('j', 278), ('k', 444), ('l', 278), ('m', 722),
|
||||
('n', 500), ('o', 500), ('p', 500), ('q', 500), ('r', 389), ('s', 389),
|
||||
('t', 278), ('u', 500), ('v', 444), ('w', 667), ('x', 444), ('y', 444),
|
||||
('z', 389), ('{', 400), ('|', 275), ('}', 400), ('~', 541), ('\u{00A1}', 389),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 275), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 760), ('\u{00AA}', 276), ('\u{00AB}', 500), ('\u{00AC}', 675), ('\u{00AE}', 760),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 675), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 500), ('\u{00B6}', 523), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 310),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 500), ('\u{00C0}', 611),
|
||||
('\u{00C1}', 611), ('\u{00C2}', 611), ('\u{00C3}', 611), ('\u{00C4}', 611), ('\u{00C5}', 611), ('\u{00C6}', 889),
|
||||
('\u{00C7}', 667), ('\u{00C8}', 611), ('\u{00C9}', 611), ('\u{00CA}', 611), ('\u{00CB}', 611), ('\u{00CC}', 333),
|
||||
('\u{00CD}', 333), ('\u{00CE}', 333), ('\u{00CF}', 333), ('\u{00D0}', 722), ('\u{00D1}', 667), ('\u{00D2}', 722),
|
||||
('\u{00D3}', 722), ('\u{00D4}', 722), ('\u{00D5}', 722), ('\u{00D6}', 722), ('\u{00D7}', 675), ('\u{00D8}', 722),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 556), ('\u{00DE}', 611),
|
||||
('\u{00DF}', 500), ('\u{00E0}', 500), ('\u{00E1}', 500), ('\u{00E2}', 500), ('\u{00E3}', 500), ('\u{00E4}', 500),
|
||||
('\u{00E5}', 500), ('\u{00E6}', 667), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 500), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 675), ('\u{00F8}', 500), ('\u{00F9}', 500), ('\u{00FA}', 500), ('\u{00FB}', 500), ('\u{00FC}', 500),
|
||||
('\u{00FD}', 444), ('\u{00FE}', 500), ('\u{00FF}', 444), ('\u{0131}', 278), ('\u{0141}', 556), ('\u{0142}', 278),
|
||||
('\u{0152}', 944), ('\u{0153}', 667), ('\u{0160}', 500), ('\u{0161}', 389), ('\u{0178}', 556), ('\u{017D}', 556),
|
||||
('\u{017E}', 389), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 889),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 556), ('\u{201D}', 556), ('\u{201E}', 556),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 889), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 980), ('\u{2212}', 675), ('\u{FB01}', 500),
|
||||
('\u{FB02}', 500),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_BOLDITALIC: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 389), ('"', 555), ('#', 500), ('$', 500), ('%', 833),
|
||||
('&', 778), ('\'', 278), ('(', 333), (')', 333), ('*', 500), ('+', 570),
|
||||
(',', 250), ('-', 333), ('.', 250), ('/', 278), ('0', 500), ('1', 500),
|
||||
('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500), ('7', 500),
|
||||
('8', 500), ('9', 500), (':', 333), (';', 333), ('<', 570), ('=', 570),
|
||||
('>', 570), ('?', 500), ('@', 832), ('A', 667), ('B', 667), ('C', 667),
|
||||
('D', 722), ('E', 667), ('F', 667), ('G', 722), ('H', 778), ('I', 389),
|
||||
('J', 500), ('K', 667), ('L', 611), ('M', 889), ('N', 722), ('O', 722),
|
||||
('P', 611), ('Q', 722), ('R', 667), ('S', 556), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 889), ('X', 667), ('Y', 611), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 570), ('_', 500), ('`', 333), ('a', 500),
|
||||
('b', 500), ('c', 444), ('d', 500), ('e', 444), ('f', 333), ('g', 500),
|
||||
('h', 556), ('i', 278), ('j', 278), ('k', 500), ('l', 278), ('m', 778),
|
||||
('n', 556), ('o', 500), ('p', 500), ('q', 500), ('r', 389), ('s', 389),
|
||||
('t', 278), ('u', 556), ('v', 444), ('w', 667), ('x', 500), ('y', 444),
|
||||
('z', 389), ('{', 348), ('|', 220), ('}', 348), ('~', 570), ('\u{00A1}', 389),
|
||||
('\u{00A2}', 500), ('\u{00A3}', 500), ('\u{00A4}', 500), ('\u{00A5}', 500), ('\u{00A6}', 220), ('\u{00A7}', 500),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 747), ('\u{00AA}', 266), ('\u{00AB}', 500), ('\u{00AC}', 606), ('\u{00AE}', 747),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 570), ('\u{00B2}', 300), ('\u{00B3}', 300), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 576), ('\u{00B6}', 500), ('\u{00B7}', 250), ('\u{00B8}', 333), ('\u{00B9}', 300), ('\u{00BA}', 300),
|
||||
('\u{00BB}', 500), ('\u{00BC}', 750), ('\u{00BD}', 750), ('\u{00BE}', 750), ('\u{00BF}', 500), ('\u{00C0}', 667),
|
||||
('\u{00C1}', 667), ('\u{00C2}', 667), ('\u{00C3}', 667), ('\u{00C4}', 667), ('\u{00C5}', 667), ('\u{00C6}', 944),
|
||||
('\u{00C7}', 667), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 389),
|
||||
('\u{00CD}', 389), ('\u{00CE}', 389), ('\u{00CF}', 389), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 722),
|
||||
('\u{00D3}', 722), ('\u{00D4}', 722), ('\u{00D5}', 722), ('\u{00D6}', 722), ('\u{00D7}', 570), ('\u{00D8}', 722),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 611), ('\u{00DE}', 611),
|
||||
('\u{00DF}', 500), ('\u{00E0}', 500), ('\u{00E1}', 500), ('\u{00E2}', 500), ('\u{00E3}', 500), ('\u{00E4}', 500),
|
||||
('\u{00E5}', 500), ('\u{00E6}', 722), ('\u{00E7}', 444), ('\u{00E8}', 444), ('\u{00E9}', 444), ('\u{00EA}', 444),
|
||||
('\u{00EB}', 444), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 500),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 500), ('\u{00F3}', 500), ('\u{00F4}', 500), ('\u{00F5}', 500), ('\u{00F6}', 500),
|
||||
('\u{00F7}', 570), ('\u{00F8}', 500), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 444), ('\u{00FE}', 500), ('\u{00FF}', 444), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 944), ('\u{0153}', 722), ('\u{0160}', 556), ('\u{0161}', 389), ('\u{0178}', 611), ('\u{017D}', 611),
|
||||
('\u{017E}', 389), ('\u{0192}', 500), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 500), ('\u{2014}', 1000),
|
||||
('\u{2018}', 333), ('\u{2019}', 333), ('\u{201A}', 333), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 500), ('\u{2021}', 500), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 500), ('\u{2122}', 1000), ('\u{2212}', 606), ('\u{FB01}', 556),
|
||||
('\u{FB02}', 556),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static SYMBOL: &[(char, u16)] = &[
|
||||
(' ', 250), ('!', 333), ('#', 500), ('%', 833), ('&', 778), ('(', 333),
|
||||
(')', 333), ('+', 549), (',', 250), ('.', 250), ('/', 278), ('0', 500),
|
||||
('1', 500), ('2', 500), ('3', 500), ('4', 500), ('5', 500), ('6', 500),
|
||||
('7', 500), ('8', 500), ('9', 500), (':', 278), (';', 278), ('<', 549),
|
||||
('=', 549), ('>', 549), ('?', 444), ('[', 333), (']', 333), ('_', 500),
|
||||
('{', 480), ('|', 200), ('}', 480), ('\u{00AC}', 713), ('\u{00B0}', 400), ('\u{00B1}', 549),
|
||||
('\u{00B5}', 576), ('\u{00D7}', 549), ('\u{00F7}', 549), ('\u{0192}', 500), ('\u{0391}', 722), ('\u{0392}', 667),
|
||||
('\u{0393}', 603), ('\u{0395}', 611), ('\u{0396}', 611), ('\u{0397}', 722), ('\u{0398}', 741), ('\u{0399}', 333),
|
||||
('\u{039A}', 722), ('\u{039B}', 686), ('\u{039C}', 889), ('\u{039D}', 722), ('\u{039E}', 645), ('\u{039F}', 722),
|
||||
('\u{03A0}', 768), ('\u{03A1}', 556), ('\u{03A3}', 592), ('\u{03A4}', 611), ('\u{03A5}', 690), ('\u{03A6}', 763),
|
||||
('\u{03A7}', 722), ('\u{03A8}', 795), ('\u{03B1}', 631), ('\u{03B2}', 549), ('\u{03B3}', 411), ('\u{03B4}', 494),
|
||||
('\u{03B5}', 439), ('\u{03B6}', 494), ('\u{03B7}', 603), ('\u{03B8}', 521), ('\u{03B9}', 329), ('\u{03BA}', 549),
|
||||
('\u{03BB}', 549), ('\u{03BD}', 521), ('\u{03BE}', 493), ('\u{03BF}', 549), ('\u{03C0}', 549), ('\u{03C1}', 549),
|
||||
('\u{03C2}', 439), ('\u{03C3}', 603), ('\u{03C4}', 439), ('\u{03C5}', 576), ('\u{03C6}', 521), ('\u{03C7}', 549),
|
||||
('\u{03C8}', 686), ('\u{03C9}', 686), ('\u{03D1}', 631), ('\u{03D2}', 620), ('\u{03D5}', 603), ('\u{03D6}', 713),
|
||||
('\u{2022}', 460), ('\u{2026}', 1000), ('\u{2032}', 247), ('\u{2033}', 411), ('\u{2044}', 167), ('\u{20AC}', 750),
|
||||
('\u{2111}', 686), ('\u{2118}', 987), ('\u{211C}', 795), ('\u{2126}', 768), ('\u{2135}', 823), ('\u{2190}', 987),
|
||||
('\u{2191}', 603), ('\u{2192}', 987), ('\u{2193}', 603), ('\u{2194}', 1042), ('\u{21B5}', 658), ('\u{21D0}', 987),
|
||||
('\u{21D1}', 603), ('\u{21D2}', 987), ('\u{21D3}', 603), ('\u{21D4}', 1042), ('\u{2200}', 713), ('\u{2202}', 494),
|
||||
('\u{2203}', 549), ('\u{2205}', 823), ('\u{2206}', 612), ('\u{2207}', 713), ('\u{2208}', 713), ('\u{2209}', 713),
|
||||
('\u{220B}', 439), ('\u{220F}', 823), ('\u{2211}', 713), ('\u{2212}', 549), ('\u{2217}', 500), ('\u{221A}', 549),
|
||||
('\u{221D}', 713), ('\u{221E}', 713), ('\u{2220}', 768), ('\u{2227}', 603), ('\u{2228}', 603), ('\u{2229}', 768),
|
||||
('\u{222A}', 768), ('\u{222B}', 274), ('\u{2234}', 863), ('\u{223C}', 549), ('\u{2245}', 549), ('\u{2248}', 549),
|
||||
('\u{2260}', 549), ('\u{2261}', 549), ('\u{2264}', 549), ('\u{2265}', 549), ('\u{2282}', 713), ('\u{2283}', 713),
|
||||
('\u{2284}', 713), ('\u{2286}', 713), ('\u{2287}', 713), ('\u{2295}', 768), ('\u{2297}', 768), ('\u{22A5}', 658),
|
||||
('\u{22C5}', 250), ('\u{2320}', 686), ('\u{2321}', 686), ('\u{2329}', 329), ('\u{232A}', 329), ('\u{25CA}', 494),
|
||||
('\u{2660}', 753), ('\u{2663}', 753), ('\u{2665}', 753), ('\u{2666}', 753), ('\u{F6D9}', 790), ('\u{F6DA}', 790),
|
||||
('\u{F6DB}', 890), ('\u{F8E5}', 500), ('\u{F8E6}', 603), ('\u{F8E7}', 1000), ('\u{F8E8}', 790), ('\u{F8E9}', 790),
|
||||
('\u{F8EA}', 786), ('\u{F8EB}', 384), ('\u{F8EC}', 384), ('\u{F8ED}', 384), ('\u{F8EE}', 384), ('\u{F8EF}', 384),
|
||||
('\u{F8F0}', 384), ('\u{F8F1}', 494), ('\u{F8F2}', 494), ('\u{F8F3}', 494), ('\u{F8F4}', 494), ('\u{F8F5}', 686),
|
||||
('\u{F8F6}', 384), ('\u{F8F7}', 384), ('\u{F8F8}', 384), ('\u{F8F9}', 384), ('\u{F8FA}', 384), ('\u{F8FB}', 384),
|
||||
('\u{F8FC}', 494), ('\u{F8FD}', 494), ('\u{F8FE}', 494), ('\u{F8FF}', 790),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static ZAPFDINGBATS: &[(char, u16)] = &[
|
||||
(' ', 278), ('\u{2192}', 838), ('\u{2194}', 1016), ('\u{2195}', 458), ('\u{2460}', 788), ('\u{2461}', 788),
|
||||
('\u{2462}', 788), ('\u{2463}', 788), ('\u{2464}', 788), ('\u{2465}', 788), ('\u{2466}', 788), ('\u{2467}', 788),
|
||||
('\u{2468}', 788), ('\u{2469}', 788), ('\u{25A0}', 761), ('\u{25B2}', 892), ('\u{25BC}', 892), ('\u{25C6}', 788),
|
||||
('\u{25CF}', 791), ('\u{25D7}', 438), ('\u{2605}', 816), ('\u{260E}', 719), ('\u{261B}', 960), ('\u{261E}', 939),
|
||||
('\u{2660}', 626), ('\u{2663}', 776), ('\u{2665}', 694), ('\u{2666}', 595), ('\u{2701}', 974), ('\u{2702}', 961),
|
||||
('\u{2703}', 974), ('\u{2704}', 980), ('\u{2706}', 789), ('\u{2707}', 790), ('\u{2708}', 791), ('\u{2709}', 690),
|
||||
('\u{270C}', 549), ('\u{270D}', 855), ('\u{270E}', 911), ('\u{270F}', 933), ('\u{2710}', 911), ('\u{2711}', 945),
|
||||
('\u{2712}', 974), ('\u{2713}', 755), ('\u{2714}', 846), ('\u{2715}', 762), ('\u{2716}', 761), ('\u{2717}', 571),
|
||||
('\u{2718}', 677), ('\u{2719}', 763), ('\u{271A}', 760), ('\u{271B}', 759), ('\u{271C}', 754), ('\u{271D}', 494),
|
||||
('\u{271E}', 552), ('\u{271F}', 537), ('\u{2720}', 577), ('\u{2721}', 692), ('\u{2722}', 786), ('\u{2723}', 788),
|
||||
('\u{2724}', 788), ('\u{2725}', 790), ('\u{2726}', 793), ('\u{2727}', 794), ('\u{2729}', 823), ('\u{272A}', 789),
|
||||
('\u{272B}', 841), ('\u{272C}', 823), ('\u{272D}', 833), ('\u{272E}', 816), ('\u{272F}', 831), ('\u{2730}', 923),
|
||||
('\u{2731}', 744), ('\u{2732}', 723), ('\u{2733}', 749), ('\u{2734}', 790), ('\u{2735}', 792), ('\u{2736}', 695),
|
||||
('\u{2737}', 776), ('\u{2738}', 768), ('\u{2739}', 792), ('\u{273A}', 759), ('\u{273B}', 707), ('\u{273C}', 708),
|
||||
('\u{273D}', 682), ('\u{273E}', 701), ('\u{273F}', 826), ('\u{2740}', 815), ('\u{2741}', 789), ('\u{2742}', 789),
|
||||
('\u{2743}', 707), ('\u{2744}', 687), ('\u{2745}', 696), ('\u{2746}', 689), ('\u{2747}', 786), ('\u{2748}', 787),
|
||||
('\u{2749}', 713), ('\u{274A}', 791), ('\u{274B}', 785), ('\u{274D}', 873), ('\u{274F}', 762), ('\u{2750}', 762),
|
||||
('\u{2751}', 759), ('\u{2752}', 759), ('\u{2756}', 784), ('\u{2758}', 138), ('\u{2759}', 277), ('\u{275A}', 415),
|
||||
('\u{275B}', 392), ('\u{275C}', 392), ('\u{275D}', 668), ('\u{275E}', 668), ('\u{2761}', 732), ('\u{2762}', 544),
|
||||
('\u{2763}', 544), ('\u{2764}', 910), ('\u{2765}', 667), ('\u{2766}', 760), ('\u{2767}', 760), ('\u{2768}', 390),
|
||||
('\u{2769}', 390), ('\u{276A}', 317), ('\u{276B}', 317), ('\u{276C}', 276), ('\u{276D}', 276), ('\u{276E}', 509),
|
||||
('\u{276F}', 509), ('\u{2770}', 410), ('\u{2771}', 410), ('\u{2772}', 234), ('\u{2773}', 234), ('\u{2774}', 334),
|
||||
('\u{2775}', 334), ('\u{2776}', 788), ('\u{2777}', 788), ('\u{2778}', 788), ('\u{2779}', 788), ('\u{277A}', 788),
|
||||
('\u{277B}', 788), ('\u{277C}', 788), ('\u{277D}', 788), ('\u{277E}', 788), ('\u{277F}', 788), ('\u{2780}', 788),
|
||||
('\u{2781}', 788), ('\u{2782}', 788), ('\u{2783}', 788), ('\u{2784}', 788), ('\u{2785}', 788), ('\u{2786}', 788),
|
||||
('\u{2787}', 788), ('\u{2788}', 788), ('\u{2789}', 788), ('\u{278A}', 788), ('\u{278B}', 788), ('\u{278C}', 788),
|
||||
('\u{278D}', 788), ('\u{278E}', 788), ('\u{278F}', 788), ('\u{2790}', 788), ('\u{2791}', 788), ('\u{2792}', 788),
|
||||
('\u{2793}', 788), ('\u{2794}', 894), ('\u{2798}', 748), ('\u{2799}', 924), ('\u{279A}', 748), ('\u{279B}', 918),
|
||||
('\u{279C}', 927), ('\u{279D}', 928), ('\u{279E}', 928), ('\u{279F}', 834), ('\u{27A0}', 873), ('\u{27A1}', 828),
|
||||
('\u{27A2}', 924), ('\u{27A3}', 924), ('\u{27A4}', 917), ('\u{27A5}', 930), ('\u{27A6}', 931), ('\u{27A7}', 463),
|
||||
('\u{27A8}', 883), ('\u{27A9}', 836), ('\u{27AA}', 836), ('\u{27AB}', 867), ('\u{27AC}', 867), ('\u{27AD}', 696),
|
||||
('\u{27AE}', 696), ('\u{27AF}', 874), ('\u{27B1}', 874), ('\u{27B2}', 760), ('\u{27B3}', 946), ('\u{27B4}', 771),
|
||||
('\u{27B5}', 865), ('\u{27B6}', 771), ('\u{27B7}', 888), ('\u{27B8}', 967), ('\u{27B9}', 888), ('\u{27BA}', 831),
|
||||
('\u{27BB}', 873), ('\u{27BC}', 927), ('\u{27BD}', 970), ('\u{27BE}', 918),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static SYMBOL_ENCODING: &[(u8, char)] = &[
|
||||
(0x20, ' '), (0x21, '!'), (0x22, '\u{2200}'), (0x23, '#'), (0x24, '\u{2203}'), (0x25, '%'),
|
||||
(0x26, '&'), (0x27, '\u{220B}'), (0x28, '('), (0x29, ')'), (0x2A, '\u{2217}'), (0x2B, '+'),
|
||||
(0x2C, ','), (0x2D, '\u{2212}'), (0x2E, '.'), (0x2F, '/'), (0x30, '0'), (0x31, '1'),
|
||||
(0x32, '2'), (0x33, '3'), (0x34, '4'), (0x35, '5'), (0x36, '6'), (0x37, '7'),
|
||||
(0x38, '8'), (0x39, '9'), (0x3A, ':'), (0x3B, ';'), (0x3C, '<'), (0x3D, '='),
|
||||
(0x3E, '>'), (0x3F, '?'), (0x40, '\u{2245}'), (0x41, '\u{0391}'), (0x42, '\u{0392}'), (0x43, '\u{03A7}'),
|
||||
(0x44, '\u{2206}'), (0x45, '\u{0395}'), (0x46, '\u{03A6}'), (0x47, '\u{0393}'), (0x48, '\u{0397}'), (0x49, '\u{0399}'),
|
||||
(0x4A, '\u{03D1}'), (0x4B, '\u{039A}'), (0x4C, '\u{039B}'), (0x4D, '\u{039C}'), (0x4E, '\u{039D}'), (0x4F, '\u{039F}'),
|
||||
(0x50, '\u{03A0}'), (0x51, '\u{0398}'), (0x52, '\u{03A1}'), (0x53, '\u{03A3}'), (0x54, '\u{03A4}'), (0x55, '\u{03A5}'),
|
||||
(0x56, '\u{03C2}'), (0x57, '\u{2126}'), (0x58, '\u{039E}'), (0x59, '\u{03A8}'), (0x5A, '\u{0396}'), (0x5B, '['),
|
||||
(0x5C, '\u{2234}'), (0x5D, ']'), (0x5E, '\u{22A5}'), (0x5F, '_'), (0x60, '\u{F8E5}'), (0x61, '\u{03B1}'),
|
||||
(0x62, '\u{03B2}'), (0x63, '\u{03C7}'), (0x64, '\u{03B4}'), (0x65, '\u{03B5}'), (0x66, '\u{03C6}'), (0x67, '\u{03B3}'),
|
||||
(0x68, '\u{03B7}'), (0x69, '\u{03B9}'), (0x6A, '\u{03D5}'), (0x6B, '\u{03BA}'), (0x6C, '\u{03BB}'), (0x6D, '\u{00B5}'),
|
||||
(0x6E, '\u{03BD}'), (0x6F, '\u{03BF}'), (0x70, '\u{03C0}'), (0x71, '\u{03B8}'), (0x72, '\u{03C1}'), (0x73, '\u{03C3}'),
|
||||
(0x74, '\u{03C4}'), (0x75, '\u{03C5}'), (0x76, '\u{03D6}'), (0x77, '\u{03C9}'), (0x78, '\u{03BE}'), (0x79, '\u{03C8}'),
|
||||
(0x7A, '\u{03B6}'), (0x7B, '{'), (0x7C, '|'), (0x7D, '}'), (0x7E, '\u{223C}'), (0xA0, '\u{20AC}'),
|
||||
(0xA1, '\u{03D2}'), (0xA2, '\u{2032}'), (0xA3, '\u{2264}'), (0xA4, '\u{2044}'), (0xA5, '\u{221E}'), (0xA6, '\u{0192}'),
|
||||
(0xA7, '\u{2663}'), (0xA8, '\u{2666}'), (0xA9, '\u{2665}'), (0xAA, '\u{2660}'), (0xAB, '\u{2194}'), (0xAC, '\u{2190}'),
|
||||
(0xAD, '\u{2191}'), (0xAE, '\u{2192}'), (0xAF, '\u{2193}'), (0xB0, '\u{00B0}'), (0xB1, '\u{00B1}'), (0xB2, '\u{2033}'),
|
||||
(0xB3, '\u{2265}'), (0xB4, '\u{00D7}'), (0xB5, '\u{221D}'), (0xB6, '\u{2202}'), (0xB7, '\u{2022}'), (0xB8, '\u{00F7}'),
|
||||
(0xB9, '\u{2260}'), (0xBA, '\u{2261}'), (0xBB, '\u{2248}'), (0xBC, '\u{2026}'), (0xBD, '\u{F8E6}'), (0xBE, '\u{F8E7}'),
|
||||
(0xBF, '\u{21B5}'), (0xC0, '\u{2135}'), (0xC1, '\u{2111}'), (0xC2, '\u{211C}'), (0xC3, '\u{2118}'), (0xC4, '\u{2297}'),
|
||||
(0xC5, '\u{2295}'), (0xC6, '\u{2205}'), (0xC7, '\u{2229}'), (0xC8, '\u{222A}'), (0xC9, '\u{2283}'), (0xCA, '\u{2287}'),
|
||||
(0xCB, '\u{2284}'), (0xCC, '\u{2282}'), (0xCD, '\u{2286}'), (0xCE, '\u{2208}'), (0xCF, '\u{2209}'), (0xD0, '\u{2220}'),
|
||||
(0xD1, '\u{2207}'), (0xD2, '\u{F6DA}'), (0xD3, '\u{F6D9}'), (0xD4, '\u{F6DB}'), (0xD5, '\u{220F}'), (0xD6, '\u{221A}'),
|
||||
(0xD7, '\u{22C5}'), (0xD8, '\u{00AC}'), (0xD9, '\u{2227}'), (0xDA, '\u{2228}'), (0xDB, '\u{21D4}'), (0xDC, '\u{21D0}'),
|
||||
(0xDD, '\u{21D1}'), (0xDE, '\u{21D2}'), (0xDF, '\u{21D3}'), (0xE0, '\u{25CA}'), (0xE1, '\u{2329}'), (0xE2, '\u{F8E8}'),
|
||||
(0xE3, '\u{F8E9}'), (0xE4, '\u{F8EA}'), (0xE5, '\u{2211}'), (0xE6, '\u{F8EB}'), (0xE7, '\u{F8EC}'), (0xE8, '\u{F8ED}'),
|
||||
(0xE9, '\u{F8EE}'), (0xEA, '\u{F8EF}'), (0xEB, '\u{F8F0}'), (0xEC, '\u{F8F1}'), (0xED, '\u{F8F2}'), (0xEE, '\u{F8F3}'),
|
||||
(0xEF, '\u{F8F4}'), (0xF1, '\u{232A}'), (0xF2, '\u{222B}'), (0xF3, '\u{2320}'), (0xF4, '\u{F8F5}'), (0xF5, '\u{2321}'),
|
||||
(0xF6, '\u{F8F6}'), (0xF7, '\u{F8F7}'), (0xF8, '\u{F8F8}'), (0xF9, '\u{F8F9}'), (0xFA, '\u{F8FA}'), (0xFB, '\u{F8FB}'),
|
||||
(0xFC, '\u{F8FC}'), (0xFD, '\u{F8FD}'), (0xFE, '\u{F8FE}'),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static ZAPFDINGBATS_ENCODING: &[(u8, char)] = &[
|
||||
(0x20, ' '), (0x21, '\u{2701}'), (0x22, '\u{2702}'), (0x23, '\u{2703}'), (0x24, '\u{2704}'), (0x25, '\u{260E}'),
|
||||
(0x26, '\u{2706}'), (0x27, '\u{2707}'), (0x28, '\u{2708}'), (0x29, '\u{2709}'), (0x2A, '\u{261B}'), (0x2B, '\u{261E}'),
|
||||
(0x2C, '\u{270C}'), (0x2D, '\u{270D}'), (0x2E, '\u{270E}'), (0x2F, '\u{270F}'), (0x30, '\u{2710}'), (0x31, '\u{2711}'),
|
||||
(0x32, '\u{2712}'), (0x33, '\u{2713}'), (0x34, '\u{2714}'), (0x35, '\u{2715}'), (0x36, '\u{2716}'), (0x37, '\u{2717}'),
|
||||
(0x38, '\u{2718}'), (0x39, '\u{2719}'), (0x3A, '\u{271A}'), (0x3B, '\u{271B}'), (0x3C, '\u{271C}'), (0x3D, '\u{271D}'),
|
||||
(0x3E, '\u{271E}'), (0x3F, '\u{271F}'), (0x40, '\u{2720}'), (0x41, '\u{2721}'), (0x42, '\u{2722}'), (0x43, '\u{2723}'),
|
||||
(0x44, '\u{2724}'), (0x45, '\u{2725}'), (0x46, '\u{2726}'), (0x47, '\u{2727}'), (0x48, '\u{2605}'), (0x49, '\u{2729}'),
|
||||
(0x4A, '\u{272A}'), (0x4B, '\u{272B}'), (0x4C, '\u{272C}'), (0x4D, '\u{272D}'), (0x4E, '\u{272E}'), (0x4F, '\u{272F}'),
|
||||
(0x50, '\u{2730}'), (0x51, '\u{2731}'), (0x52, '\u{2732}'), (0x53, '\u{2733}'), (0x54, '\u{2734}'), (0x55, '\u{2735}'),
|
||||
(0x56, '\u{2736}'), (0x57, '\u{2737}'), (0x58, '\u{2738}'), (0x59, '\u{2739}'), (0x5A, '\u{273A}'), (0x5B, '\u{273B}'),
|
||||
(0x5C, '\u{273C}'), (0x5D, '\u{273D}'), (0x5E, '\u{273E}'), (0x5F, '\u{273F}'), (0x60, '\u{2740}'), (0x61, '\u{2741}'),
|
||||
(0x62, '\u{2742}'), (0x63, '\u{2743}'), (0x64, '\u{2744}'), (0x65, '\u{2745}'), (0x66, '\u{2746}'), (0x67, '\u{2747}'),
|
||||
(0x68, '\u{2748}'), (0x69, '\u{2749}'), (0x6A, '\u{274A}'), (0x6B, '\u{274B}'), (0x6C, '\u{25CF}'), (0x6D, '\u{274D}'),
|
||||
(0x6E, '\u{25A0}'), (0x6F, '\u{274F}'), (0x70, '\u{2750}'), (0x71, '\u{2751}'), (0x72, '\u{2752}'), (0x73, '\u{25B2}'),
|
||||
(0x74, '\u{25BC}'), (0x75, '\u{25C6}'), (0x76, '\u{2756}'), (0x77, '\u{25D7}'), (0x78, '\u{2758}'), (0x79, '\u{2759}'),
|
||||
(0x7A, '\u{275A}'), (0x7B, '\u{275B}'), (0x7C, '\u{275C}'), (0x7D, '\u{275D}'), (0x7E, '\u{275E}'), (0x80, '\u{2768}'),
|
||||
(0x81, '\u{2769}'), (0x82, '\u{276A}'), (0x83, '\u{276B}'), (0x84, '\u{276C}'), (0x85, '\u{276D}'), (0x86, '\u{276E}'),
|
||||
(0x87, '\u{276F}'), (0x88, '\u{2770}'), (0x89, '\u{2771}'), (0x8A, '\u{2772}'), (0x8B, '\u{2773}'), (0x8C, '\u{2774}'),
|
||||
(0x8D, '\u{2775}'), (0xA1, '\u{2761}'), (0xA2, '\u{2762}'), (0xA3, '\u{2763}'), (0xA4, '\u{2764}'), (0xA5, '\u{2765}'),
|
||||
(0xA6, '\u{2766}'), (0xA7, '\u{2767}'), (0xA8, '\u{2663}'), (0xA9, '\u{2666}'), (0xAA, '\u{2665}'), (0xAB, '\u{2660}'),
|
||||
(0xAC, '\u{2460}'), (0xAD, '\u{2461}'), (0xAE, '\u{2462}'), (0xAF, '\u{2463}'), (0xB0, '\u{2464}'), (0xB1, '\u{2465}'),
|
||||
(0xB2, '\u{2466}'), (0xB3, '\u{2467}'), (0xB4, '\u{2468}'), (0xB5, '\u{2469}'), (0xB6, '\u{2776}'), (0xB7, '\u{2777}'),
|
||||
(0xB8, '\u{2778}'), (0xB9, '\u{2779}'), (0xBA, '\u{277A}'), (0xBB, '\u{277B}'), (0xBC, '\u{277C}'), (0xBD, '\u{277D}'),
|
||||
(0xBE, '\u{277E}'), (0xBF, '\u{277F}'), (0xC0, '\u{2780}'), (0xC1, '\u{2781}'), (0xC2, '\u{2782}'), (0xC3, '\u{2783}'),
|
||||
(0xC4, '\u{2784}'), (0xC5, '\u{2785}'), (0xC6, '\u{2786}'), (0xC7, '\u{2787}'), (0xC8, '\u{2788}'), (0xC9, '\u{2789}'),
|
||||
(0xCA, '\u{278A}'), (0xCB, '\u{278B}'), (0xCC, '\u{278C}'), (0xCD, '\u{278D}'), (0xCE, '\u{278E}'), (0xCF, '\u{278F}'),
|
||||
(0xD0, '\u{2790}'), (0xD1, '\u{2791}'), (0xD2, '\u{2792}'), (0xD3, '\u{2793}'), (0xD4, '\u{2794}'), (0xD5, '\u{2192}'),
|
||||
(0xD6, '\u{2194}'), (0xD7, '\u{2195}'), (0xD8, '\u{2798}'), (0xD9, '\u{2799}'), (0xDA, '\u{279A}'), (0xDB, '\u{279B}'),
|
||||
(0xDC, '\u{279C}'), (0xDD, '\u{279D}'), (0xDE, '\u{279E}'), (0xDF, '\u{279F}'), (0xE0, '\u{27A0}'), (0xE1, '\u{27A1}'),
|
||||
(0xE2, '\u{27A2}'), (0xE3, '\u{27A3}'), (0xE4, '\u{27A4}'), (0xE5, '\u{27A5}'), (0xE6, '\u{27A6}'), (0xE7, '\u{27A7}'),
|
||||
(0xE8, '\u{27A8}'), (0xE9, '\u{27A9}'), (0xEA, '\u{27AA}'), (0xEB, '\u{27AB}'), (0xEC, '\u{27AC}'), (0xED, '\u{27AD}'),
|
||||
(0xEE, '\u{27AE}'), (0xEF, '\u{27AF}'), (0xF1, '\u{27B1}'), (0xF2, '\u{27B2}'), (0xF3, '\u{27B3}'), (0xF4, '\u{27B4}'),
|
||||
(0xF5, '\u{27B5}'), (0xF6, '\u{27B6}'), (0xF7, '\u{27B7}'), (0xF8, '\u{27B8}'), (0xF9, '\u{27B9}'), (0xFA, '\u{27BA}'),
|
||||
(0xFB, '\u{27BB}'), (0xFC, '\u{27BC}'), (0xFD, '\u{27BD}'), (0xFE, '\u{27BE}'),
|
||||
];
|
||||
|
||||
/// Every width table, for exhaustive testing.
|
||||
#[cfg(test)]
|
||||
static ALL_TABLES: &[(&str, &[(char, u16)])] = &[
|
||||
("COURIER", COURIER),
|
||||
("COURIER_BOLD", COURIER_BOLD),
|
||||
("COURIER_OBLIQUE", COURIER_OBLIQUE),
|
||||
("COURIER_BOLDOBLIQUE", COURIER_BOLDOBLIQUE),
|
||||
("HELVETICA", HELVETICA),
|
||||
("HELVETICA_BOLD", HELVETICA_BOLD),
|
||||
("HELVETICA_OBLIQUE", HELVETICA_OBLIQUE),
|
||||
("HELVETICA_BOLDOBLIQUE", HELVETICA_BOLDOBLIQUE),
|
||||
("TIMES_ROMAN", TIMES_ROMAN),
|
||||
("TIMES_BOLD", TIMES_BOLD),
|
||||
("TIMES_ITALIC", TIMES_ITALIC),
|
||||
("TIMES_BOLDITALIC", TIMES_BOLDITALIC),
|
||||
("SYMBOL", SYMBOL),
|
||||
("ZAPFDINGBATS", ZAPFDINGBATS),
|
||||
];
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn times_roman_ascii_widths() {
|
||||
assert_eq!(base14_char_width("Times-Roman", ' '), Some(250));
|
||||
assert_eq!(base14_char_width("Times-Roman", 'M'), Some(889));
|
||||
assert_eq!(base14_char_width("Times-Roman", 'i'), Some(278));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subset_prefix_and_aliases_normalize() {
|
||||
assert_eq!(
|
||||
base14_char_width("ABCDEF+Times-Bold", ' '),
|
||||
base14_char_width("Times-Bold", ' ')
|
||||
);
|
||||
assert!(base14_char_width("ArialMT", 'a').is_some());
|
||||
assert!(base14_char_width("TimesNewRomanPSMT", 'a').is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_base14_returns_none() {
|
||||
assert_eq!(base14_char_width("DejaVuSans", 'a'), None);
|
||||
assert!(!is_base14_font("Garamond"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builtin_encoding_resolves_symbol_and_zapf_codes() {
|
||||
// Symbol 0x61 renders alpha; Zapf 0x21 renders U+2701.
|
||||
assert_eq!(builtin_encoding_char("Symbol", 0x61), Some('\u{03B1}'));
|
||||
assert_eq!(builtin_encoding_char("Symbol", 0xA5), Some('\u{221E}'));
|
||||
assert_eq!(
|
||||
builtin_encoding_char("ZapfDingbats", 0x21),
|
||||
Some('\u{2701}')
|
||||
);
|
||||
// Latin text fonts follow standard encodings — no builtin override.
|
||||
assert_eq!(builtin_encoding_char("Times-Roman", 0x61), None);
|
||||
// The resolved chars have real AFM widths.
|
||||
let alpha_w = base14_char_width("Symbol", '\u{03B1}');
|
||||
assert!(alpha_w.is_some() && alpha_w != Some(500));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encoding_tables_are_sorted_for_binary_search() {
|
||||
for table in [SYMBOL_ENCODING, ZAPFDINGBATS_ENCODING] {
|
||||
assert!(table.windows(2).all(|w| w[0].0 < w[1].0));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tables_are_sorted_for_binary_search() {
|
||||
// Every table is queried by binary search, so all of them must be
|
||||
// sorted — not just a sample.
|
||||
for (name, table) in ALL_TABLES {
|
||||
assert!(
|
||||
table.windows(2).all(|w| w[0].0 < w[1].0),
|
||||
"{name} is not sorted"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -14,9 +14,9 @@ use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, descriptor_style_flags,
|
||||
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
FontStyleCache,
|
||||
build_font_encodings, build_font_widths, build_type3_scales, compute_string_width_ts,
|
||||
descriptor_style_flags, extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes,
|
||||
CMapDecisionCache, FontStyleCache,
|
||||
};
|
||||
use super::underline::UnderlineLine;
|
||||
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
|
||||
@@ -162,10 +162,11 @@ pub(crate) fn extract_page_text_items(
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
|
||||
// Build font encoding maps from Differences arrays
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
|
||||
|
||||
// Build font width info for accurate text positioning
|
||||
let font_widths = build_font_widths(doc, &fonts);
|
||||
let type3_scales = build_type3_scales(doc, &fonts);
|
||||
|
||||
// Build maps of font resource names to their base font names and ToUnicode object refs
|
||||
let mut font_base_names: std::collections::HashMap<String, String> =
|
||||
@@ -502,7 +503,8 @@ pub(crate) fn extract_page_text_items(
|
||||
) {
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
@@ -673,7 +675,8 @@ pub(crate) fn extract_page_text_items(
|
||||
} else {
|
||||
rotation_votes.rotated += 1;
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let base_font = font_base_names
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
@@ -780,7 +783,8 @@ pub(crate) fn extract_page_text_items(
|
||||
} else {
|
||||
rotation_votes.rotated += 1;
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = w_ts_opt
|
||||
.map(|w_ts| {
|
||||
@@ -932,7 +936,8 @@ pub(crate) fn extract_page_text_items(
|
||||
} else {
|
||||
rotation_votes.rotated += 1;
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
// Width in device space from text matrix delta
|
||||
let delta_ts = text_matrix[4] - start_tm[4];
|
||||
|
||||
+388
-13
@@ -138,6 +138,93 @@ pub(crate) fn build_font_widths(
|
||||
widths
|
||||
}
|
||||
|
||||
/// Visual-size scale factors for Type3 fonts, keyed by resource name.
|
||||
///
|
||||
/// A Type3 font's glyph space maps to text space through FontMatrix, so the
|
||||
/// visual height of its glyphs is `nominal_size × |matrix_y| × FontBBox
|
||||
/// height`. For a well-behaved font (matrix 0.001, bbox ≈ 1000 units) that
|
||||
/// factor is ≈ 1.0 and the nominal size is already right. TeX PK bitmap
|
||||
/// fonts (dvips → Distiller) instead use FontMatrix [1 0 0 -1 0 0] with
|
||||
/// nominal sizes like 0.12, which makes every downstream font-size heuristic
|
||||
/// (drop caps, sub/superscripts, small-font tables, line heights) see
|
||||
/// nonsense. Fonts without a usable FontBBox are omitted (treated as 1.0).
|
||||
pub(crate) fn build_type3_scales(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
) -> HashMap<String, f32> {
|
||||
let mut scales = HashMap::new();
|
||||
for (font_name, font_dict) in fonts {
|
||||
let is_type3 = font_dict
|
||||
.get(b"Subtype")
|
||||
.ok()
|
||||
.and_then(|o| o.as_name().ok())
|
||||
.is_some_and(|n| n == b"Type3");
|
||||
if !is_type3 {
|
||||
continue;
|
||||
}
|
||||
// Array elements may themselves be indirect references per PDF
|
||||
// syntax — resolve before reading the numeric value.
|
||||
let num = |o: &Object| {
|
||||
let resolved = match o {
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(inner) => inner,
|
||||
Err(_) => return 0.0,
|
||||
},
|
||||
other => other,
|
||||
};
|
||||
match resolved {
|
||||
Object::Integer(i) => *i as f32,
|
||||
Object::Real(r) => *r,
|
||||
_ => 0.0,
|
||||
}
|
||||
};
|
||||
let Some(matrix) = font_dict
|
||||
.get(b"FontMatrix")
|
||||
.ok()
|
||||
.and_then(|o| resolve_array(doc, o))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let Some(bbox) = font_dict
|
||||
.get(b"FontBBox")
|
||||
.ok()
|
||||
.and_then(|o| resolve_array(doc, o))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
if matrix.len() < 4 || bbox.len() < 4 {
|
||||
continue;
|
||||
}
|
||||
let scale_y = (num(&matrix[2]).powi(2) + num(&matrix[3]).powi(2)).sqrt();
|
||||
let bbox_h = (num(&bbox[3]) - num(&bbox[1])).abs();
|
||||
let scale = bbox_h * scale_y;
|
||||
|
||||
// `scale` is the glyph box measured in text-space units. For a
|
||||
// self-consistent font it lands near 1.0 — the FontMatrix is the
|
||||
// reciprocal of the glyph-space em by construction — so the Tf
|
||||
// operand is already the rendered size and must be left alone.
|
||||
// A modest deviation is normal and must NOT trigger rescaling:
|
||||
// FontBBox is the glyph bounding box, not the em box, so it is
|
||||
// routinely somewhat smaller (descender..ascender ≈ 0.7) or larger
|
||||
// (tall accents > 1.0).
|
||||
//
|
||||
// Only a wildly inconsistent font gets renormalized. dvips/PK
|
||||
// bitmap fonts declare [1 0 0 -1 0 0] with glyphs spanning
|
||||
// hundreds of units, giving scale ≈ 159 against a nominal size of
|
||||
// 0.12pt — there the declared size carries no information. The
|
||||
// band is deliberately wide so that only that class qualifies,
|
||||
// while any matrix scale (including non-standard ones like 0.005
|
||||
// with a full-em bbox, scale = 5.0) is judged on the product
|
||||
// rather than on the matrix alone.
|
||||
const CONSISTENT_LO: f32 = 0.25;
|
||||
const CONSISTENT_HI: f32 = 4.0;
|
||||
if scale.is_finite() && scale > 0.0 && !(CONSISTENT_LO..=CONSISTENT_HI).contains(&scale) {
|
||||
scales.insert(String::from_utf8_lossy(font_name).to_string(), scale);
|
||||
}
|
||||
}
|
||||
scales
|
||||
}
|
||||
|
||||
/// Parse font widths from a font dictionary, dispatching by Subtype
|
||||
pub(crate) fn parse_font_widths(
|
||||
doc: &Document,
|
||||
@@ -149,11 +236,71 @@ pub(crate) fn parse_font_widths(
|
||||
|
||||
match subtype_name {
|
||||
b"Type0" => parse_type0_widths(doc, font_dict),
|
||||
b"Type1" | b"TrueType" | b"MMType1" | b"Type3" => parse_simple_font_widths(doc, font_dict),
|
||||
b"Type1" | b"TrueType" | b"MMType1" => parse_simple_font_widths(doc, font_dict)
|
||||
.or_else(|| base14_fallback_widths(doc, font_dict)),
|
||||
b"Type3" => parse_simple_font_widths(doc, font_dict),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Fallback metrics for non-embedded base-14 fonts whose dictionary omits
|
||||
/// `/FirstChar`/`/Widths` (legal per the PDF spec — the reader must supply
|
||||
/// standard-font metrics). Without this, every glyph advances 0 and all
|
||||
/// downstream gap-based logic (space synthesis, script detection, table
|
||||
/// columns) collapses — common in 1990s dvips/Distiller PDFs.
|
||||
///
|
||||
/// Widths are resolved per code through the font's Differences encoding when
|
||||
/// present, falling back to the same single-byte decode the text extractor
|
||||
/// uses (cp1252-style smart punctuation for 0x80..=0x9F, Latin-1 elsewhere) —
|
||||
/// so the width of a code always matches the char we extract for it.
|
||||
fn base14_fallback_widths(doc: &Document, font_dict: &lopdf::Dictionary) -> Option<FontWidthInfo> {
|
||||
let base_font = font_dict
|
||||
.get(b"BaseFont")
|
||||
.ok()
|
||||
.and_then(|o| o.as_name().ok())
|
||||
.map(|n| String::from_utf8_lossy(n).to_string())?;
|
||||
if !crate::extractor::base14::is_base14_font(&base_font) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let enc_map = parse_font_encoding(doc, font_dict)
|
||||
.map(|r| r.map)
|
||||
.unwrap_or_default();
|
||||
|
||||
let mut widths = HashMap::new();
|
||||
for code in 0u16..=255 {
|
||||
// Resolution order: Differences override, then the font's BUILT-IN
|
||||
// encoding (Symbol/ZapfDingbats glyphs live at positions unrelated
|
||||
// to cp1252 — the renderer draws α for Symbol 0x61 no matter how
|
||||
// the text decoder transliterates it, so the advance must be α's),
|
||||
// then the cp1252-style fallback used by the text decoder.
|
||||
let ch = enc_map
|
||||
.get(&(code as u8))
|
||||
.copied()
|
||||
.or_else(|| crate::extractor::base14::builtin_encoding_char(&base_font, code as u8))
|
||||
.unwrap_or_else(|| decode_single_byte_fallback_char(code as u8, true));
|
||||
if let Some(w) = crate::extractor::base14::base14_char_width(&base_font, ch) {
|
||||
widths.insert(code, w);
|
||||
}
|
||||
}
|
||||
let space_width = widths.get(&32).copied().unwrap_or(250);
|
||||
|
||||
debug!(
|
||||
" base14 fallback widths for {} ({} codes mapped)",
|
||||
base_font,
|
||||
widths.len()
|
||||
);
|
||||
|
||||
Some(FontWidthInfo {
|
||||
widths,
|
||||
default_width: 500,
|
||||
space_width,
|
||||
is_cid: false,
|
||||
units_scale: 0.001,
|
||||
wmode: 0,
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse widths for simple fonts (Type1, TrueType, MMType1, Type3)
|
||||
/// Reads FirstChar, LastChar, and Widths array.
|
||||
/// For Type3 fonts, reads FontMatrix to determine the correct units_scale.
|
||||
@@ -497,9 +644,13 @@ pub(crate) fn get_operand_bytes(obj: &Object) -> Option<&[u8]> {
|
||||
/// Build encoding maps for all fonts on a page.
|
||||
/// Returns `(encodings, has_gid_fonts)` where `has_gid_fonts` is true when
|
||||
/// any font uses raw glyph ID names (gidNNNNN) that can't be decoded.
|
||||
/// Gid names whose codes the font's own ToUnicode CMap maps are decodable
|
||||
/// and do not set the flag (LibreOffice subsets write /gidNNNN Differences
|
||||
/// names alongside a complete ToUnicode CMap).
|
||||
pub(crate) fn build_font_encodings(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
cmaps: &FontCMaps,
|
||||
) -> (PageFontEncodings, bool) {
|
||||
let mut encodings = PageFontEncodings::new();
|
||||
let mut has_gid_fonts = false;
|
||||
@@ -508,7 +659,9 @@ pub(crate) fn build_font_encodings(
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
|
||||
if let Some(result) = parse_font_encoding(doc, font_dict) {
|
||||
if result.gid_glyph_count > 0 {
|
||||
if !result.gid_codes.is_empty()
|
||||
&& !tounicode_maps_codes(font_dict, cmaps, &result.gid_codes)
|
||||
{
|
||||
has_gid_fonts = true;
|
||||
}
|
||||
if !result.map.is_empty() {
|
||||
@@ -520,6 +673,34 @@ pub(crate) fn build_font_encodings(
|
||||
(encodings, has_gid_fonts)
|
||||
}
|
||||
|
||||
/// True when the font's ToUnicode CMap maps the gid-named character codes,
|
||||
/// so the Differences entries still decode through the CMap.
|
||||
fn tounicode_maps_codes(font_dict: &lopdf::Dictionary, cmaps: &FontCMaps, codes: &[u8]) -> bool {
|
||||
let Some(obj_ref) = font_dict
|
||||
.get(b"ToUnicode")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
else {
|
||||
return false;
|
||||
};
|
||||
let Some(entry) = cmaps.get_by_obj(obj_ref.0) else {
|
||||
return false;
|
||||
};
|
||||
// At least one gid code usably mapped means the CMap addresses these
|
||||
// codes; remaining unmapped codes are subset leftovers (e.g. the
|
||||
// component glyphs of an emoji ZWJ sequence mapped whole on its first
|
||||
// code). A mapping is usable only when extraction would accept it —
|
||||
// empty or U+FFFD results are rejected there as invalid. Fonts whose
|
||||
// CMap ignores the gid codes entirely stay flagged, and the downstream
|
||||
// garbage/encoding checks still catch partial damage.
|
||||
codes.iter().any(|&code| {
|
||||
entry
|
||||
.primary
|
||||
.lookup(code as u16)
|
||||
.is_some_and(|s| !s.is_empty() && !s.contains('\u{FFFD}'))
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse font encoding from a font dictionary
|
||||
pub(crate) fn parse_font_encoding(
|
||||
doc: &Document,
|
||||
@@ -558,11 +739,10 @@ pub(crate) fn parse_font_encoding(
|
||||
/// Result of parsing an encoding dictionary's Differences array.
|
||||
pub(crate) struct EncodingResult {
|
||||
pub map: FontEncodingMap,
|
||||
/// Number of glyph names matching the `gidNNNNN` pattern (raw glyph IDs).
|
||||
/// These indicate a font with unresolvable encoding — the glyph IDs
|
||||
/// reference the original font's glyph table, but without the original
|
||||
/// font's cmap there is no way to map them to Unicode.
|
||||
pub gid_glyph_count: u32,
|
||||
/// Character codes whose glyph names match the `gidNNNNN` pattern (raw
|
||||
/// glyph IDs). These reference the original font's glyph table and are
|
||||
/// only decodable when the font's ToUnicode CMap maps the code.
|
||||
pub gid_codes: Vec<u8>,
|
||||
}
|
||||
|
||||
/// Parse an encoding dictionary with Differences array
|
||||
@@ -588,7 +768,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
let mut encoding_map = FontEncodingMap::new();
|
||||
let mut current_code: u8 = 0;
|
||||
let mut ligature_count = 0u32;
|
||||
let mut gid_glyph_count = 0u32;
|
||||
let mut gid_codes: Vec<u8> = Vec::new();
|
||||
|
||||
for item in diff_array {
|
||||
match item {
|
||||
@@ -614,7 +794,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
&& glyph_name.len() >= 4
|
||||
&& glyph_name[3..].chars().all(|c| c.is_ascii_digit())
|
||||
{
|
||||
gid_glyph_count += 1;
|
||||
gid_codes.push(current_code);
|
||||
}
|
||||
if let Some(ch) = mapped_char {
|
||||
encoding_map.insert(current_code, ch);
|
||||
@@ -638,16 +818,16 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
);
|
||||
}
|
||||
|
||||
if gid_glyph_count > 0 {
|
||||
if !gid_codes.is_empty() {
|
||||
debug!(
|
||||
" Differences: {} gid-encoded glyphs (unresolvable without original font)",
|
||||
gid_glyph_count
|
||||
" Differences: {} gid-encoded glyphs (decodable only via ToUnicode)",
|
||||
gid_codes.len()
|
||||
);
|
||||
}
|
||||
|
||||
Some(EncodingResult {
|
||||
map: encoding_map,
|
||||
gid_glyph_count,
|
||||
gid_codes,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1437,6 +1617,92 @@ fn score_text(text: &str) -> i32 {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn type3_scale_resolves_indirect_matrix_and_bbox_numbers() {
|
||||
use lopdf::{dictionary, Document, Object};
|
||||
// FontMatrix/FontBBox elements may be indirect references per PDF
|
||||
// syntax; the scale must use their resolved values, not zero.
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let matrix_d = doc.add_object(Object::Real(-1.0));
|
||||
let bbox_top = doc.add_object(Object::Integer(3));
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type3",
|
||||
"FontMatrix" => vec![
|
||||
Object::Integer(1),
|
||||
Object::Integer(0),
|
||||
Object::Integer(0),
|
||||
Object::Reference(matrix_d),
|
||||
Object::Integer(0),
|
||||
Object::Integer(0),
|
||||
],
|
||||
"FontBBox" => vec![
|
||||
Object::Integer(1),
|
||||
Object::Integer(-156),
|
||||
Object::Integer(37),
|
||||
Object::Reference(bbox_top),
|
||||
],
|
||||
};
|
||||
let mut fonts = std::collections::BTreeMap::new();
|
||||
fonts.insert(b"T2".to_vec(), &font_dict);
|
||||
let scales = super::build_type3_scales(&doc, &fonts);
|
||||
let scale = scales.get("T2").copied().unwrap_or(1.0);
|
||||
// bbox height 159 x |matrix_y| 1.0
|
||||
assert!(
|
||||
(scale - 159.0).abs() < 0.5,
|
||||
"scale should use resolved indirect values, got {scale}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Build a one-font Type3 document and return its computed scale, if any.
|
||||
#[cfg(test)]
|
||||
fn type3_scale_for(matrix_y: f32, bbox_lo: i64, bbox_hi: i64) -> Option<f32> {
|
||||
use lopdf::{dictionary, Document, Object};
|
||||
let doc = Document::with_version("1.4");
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type3",
|
||||
"FontMatrix" => vec![
|
||||
Object::Real(matrix_y), Object::Integer(0), Object::Integer(0),
|
||||
Object::Real(matrix_y), Object::Integer(0), Object::Integer(0),
|
||||
],
|
||||
"FontBBox" => vec![
|
||||
Object::Integer(0), Object::Integer(bbox_lo),
|
||||
Object::Integer(600), Object::Integer(bbox_hi),
|
||||
],
|
||||
};
|
||||
let mut fonts = std::collections::BTreeMap::new();
|
||||
fonts.insert(b"T9".to_vec(), &font_dict);
|
||||
super::build_type3_scales(&doc, &fonts).get("T9").copied()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_skips_self_consistent_fonts() {
|
||||
// Conventional 1/1000 matrix with a descender..ascender bbox of 700
|
||||
// units: scale 0.7. The Tf operand is already the rendered size, so
|
||||
// renormalizing would report every size at 0.7x.
|
||||
assert_eq!(type3_scale_for(0.001, -200, 500), None);
|
||||
// Tall-accent bbox slightly over the em (1100 units, scale 1.1).
|
||||
assert_eq!(type3_scale_for(0.001, -100, 1000), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_applies_to_inconsistent_fonts_at_any_matrix_scale() {
|
||||
// Non-standard but valid matrix (0.005) with a full-em bbox:
|
||||
// scale 5.0, so the declared size is off by 5x and must be fixed.
|
||||
let s = type3_scale_for(0.005, 0, 1000).expect("0.005 matrix should rescale");
|
||||
assert!((s - 5.0).abs() < 0.01, "got {s}");
|
||||
// dvips/PK bitmap pattern: unit matrix, glyphs spanning ~159 units.
|
||||
let s = type3_scale_for(1.0, -156, 3).expect("PK pattern should rescale");
|
||||
assert!((s - 159.0).abs() < 0.5, "got {s}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_ignores_degenerate_bbox() {
|
||||
// [0 0 0 0] is legal and carries no size information.
|
||||
assert_eq!(type3_scale_for(0.001, 0, 0), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn texcm_math_symbols_remap() {
|
||||
assert_eq!(
|
||||
@@ -1938,4 +2204,113 @@ mod tests {
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
fn gid_font_doc(bfchar: Option<&str>) -> (Document, lopdf::ObjectId) {
|
||||
use lopdf::Stream;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let cmap = format!(
|
||||
"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
{}
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end",
|
||||
bfchar.unwrap_or_default()
|
||||
);
|
||||
let tounicode_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
cmap.into_bytes(),
|
||||
)));
|
||||
let enc_id = doc.add_object(dictionary! {
|
||||
"Type" => "Encoding",
|
||||
"Differences" => vec![
|
||||
1.into(),
|
||||
Object::Name(b"gid1283".to_vec()),
|
||||
Object::Name(b"gid1464".to_vec()),
|
||||
],
|
||||
});
|
||||
let mut font = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "ABCDEF+OpenSymbol",
|
||||
"Encoding" => Object::Reference(enc_id),
|
||||
};
|
||||
if bfchar.is_some() {
|
||||
font.set("ToUnicode", Object::Reference(tounicode_id));
|
||||
}
|
||||
let font_id = doc.add_object(font);
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn gid_flagged(bfchar: Option<&str>) -> bool {
|
||||
let (doc, page_id) = gid_font_doc(bfchar);
|
||||
let cmaps = FontCMaps::from_doc(&doc);
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap();
|
||||
let (_, has_gid_fonts) = build_font_encodings(&doc, &fonts, &cmaps);
|
||||
has_gid_fonts
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_covering_tounicode_are_not_flagged() {
|
||||
// LibreOffice subsets write /gidNNNN Differences names alongside a
|
||||
// ToUnicode CMap that decodes those codes; the page must not be
|
||||
// flagged as unresolvable (which would suppress the whole document's
|
||||
// markdown when every page carries such a font).
|
||||
assert!(!gid_flagged(Some("<01> <2022>\n<02> <25E6>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_partial_tounicode_are_not_flagged() {
|
||||
// An emoji ZWJ sequence maps whole on its first code; the remaining
|
||||
// component-glyph codes are subset leftovers, not damage.
|
||||
assert!(!gid_flagged(Some(
|
||||
"<01> <D83DDC68200DD83DDC69200DD83DDC67>"
|
||||
)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_without_tounicode_are_flagged() {
|
||||
assert!(
|
||||
gid_flagged(None),
|
||||
"gid glyphs without ToUnicode are unresolvable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_disjoint_tounicode_are_flagged() {
|
||||
// A ToUnicode that never addresses the gid codes leaves them
|
||||
// unresolvable.
|
||||
assert!(gid_flagged(Some("<10> <0041>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_replacement_char_tounicode_are_flagged() {
|
||||
// A mapping to U+FFFD is not usable — extraction rejects it as an
|
||||
// invalid CMap result — so it must not clear the gid flag.
|
||||
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
|
||||
}
|
||||
}
|
||||
|
||||
+810
-17
@@ -960,22 +960,713 @@ fn spans_multiple_columns(item: &TextItem, columns: &[ColumnRegion]) -> bool {
|
||||
overlap_count >= 2
|
||||
}
|
||||
|
||||
/// Check if a text item is likely a page number
|
||||
fn is_page_number(item: &TextItem) -> bool {
|
||||
const PAGE_NUMBER_Y_TOLERANCE: f32 = 3.0;
|
||||
const PAGE_NUMBER_CONTEXT_GAP_EM: f32 = 1.5;
|
||||
const PAGE_NUMBER_BOTTOM_Y: f32 = 100.0;
|
||||
const PAGE_NUMBER_TOP_Y: f32 = 720.0;
|
||||
const SPREAD_MIN_CONTENT_WIDTH_EM: f32 = 40.0;
|
||||
const SPREAD_EDGE_FRACTION: f32 = 0.25;
|
||||
const ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM: f32 = 26.0;
|
||||
|
||||
type ContextualCandidateOccurrence = (u32, f32, Vec<(usize, u32)>);
|
||||
|
||||
fn page_number_value(item: &TextItem) -> Option<u32> {
|
||||
if !matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let text = item.text.trim();
|
||||
|
||||
// Must be 1-4 digits only
|
||||
if text.is_empty() || text.len() > 4 {
|
||||
return false;
|
||||
}
|
||||
if !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return false;
|
||||
if text.is_empty() || text.len() > 4 || !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Must be at top or bottom of page.
|
||||
// US Letter = 792pt, A4 = 841pt. Page numbers are typically in the
|
||||
// top ~5% or bottom ~12% of the page.
|
||||
item.y > 720.0 || item.y < 100.0
|
||||
if item.y <= PAGE_NUMBER_TOP_Y && item.y >= PAGE_NUMBER_BOTTOM_Y {
|
||||
return None;
|
||||
}
|
||||
|
||||
text.parse().ok()
|
||||
}
|
||||
|
||||
/// Mark numeric slots that advance inside a repeated deep-margin line.
|
||||
///
|
||||
/// A folio can be emitted as part of a footer text run (for example,
|
||||
/// `42 Company report`) and therefore look contextual on a single page. Across
|
||||
/// the document, however, the surrounding text and Y position repeat while the
|
||||
/// numeric slot advances. Require that full signal before treating the slot as
|
||||
/// a folio so constant metadata and substantive rows near the page edge remain
|
||||
/// untouched.
|
||||
fn mark_repeated_folio_candidates(
|
||||
occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>>,
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
// Both thresholds are evidence floors: short documents still need four
|
||||
// occurrences, while long documents also need meaningful coverage. Using
|
||||
// `min` here would make one occurrence sufficient in a one-page document.
|
||||
let min_pages = 4usize.max((document_page_count * 30).div_ceil(100));
|
||||
|
||||
for occurrences in occurrences_by_signature.into_values() {
|
||||
if occurrences.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let distinct_pages: HashSet<u32> = occurrences.iter().map(|(page, _, _)| *page).collect();
|
||||
// Repeated table rows or duplicated drawing labels can share a
|
||||
// signature multiple times on one page. They are not running folios.
|
||||
if distinct_pages.len() != occurrences.len() || distinct_pages.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let min_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if max_y - min_y >= PAGE_NUMBER_Y_TOLERANCE {
|
||||
continue;
|
||||
}
|
||||
|
||||
let slot_count = occurrences[0].2.len();
|
||||
if slot_count == 0
|
||||
|| occurrences
|
||||
.iter()
|
||||
.any(|(_, _, candidates)| candidates.len() != slot_count)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
for slot in 0..slot_count {
|
||||
let mut values: Vec<(u32, u32, usize)> = occurrences
|
||||
.iter()
|
||||
.map(|(page, _, candidates)| {
|
||||
let (index, value) = candidates[slot];
|
||||
(*page, value, index)
|
||||
})
|
||||
.collect();
|
||||
values.sort_by_key(|(page, _, _)| *page);
|
||||
|
||||
let unique_values: HashSet<u32> = values.iter().map(|(_, value, _)| *value).collect();
|
||||
let mostly_unique = unique_values.len() * 5 >= values.len() * 4;
|
||||
let page_tracking_pairs = values
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let page_delta = pair[1].0 - pair[0].0;
|
||||
let value_delta = pair[1].1.saturating_sub(pair[0].1);
|
||||
value_delta == page_delta || value_delta == page_delta.saturating_mul(2)
|
||||
})
|
||||
.count();
|
||||
let mostly_tracks_page_order = page_tracking_pairs * 5 >= (values.len() - 1) * 4;
|
||||
// A running folio can be offset by front matter or advance twice per
|
||||
// PDF page in a two-page spread, but its magnitude should still be
|
||||
// plausible for the document. This keeps changing metadata such as
|
||||
// a sequence of years from becoming a deletion signal.
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
let plausible_magnitude = values
|
||||
.iter()
|
||||
.all(|(_, value, _)| *value <= max_plausible_folio);
|
||||
|
||||
if mostly_unique && mostly_tracks_page_order && plausible_magnitude {
|
||||
for (_, _, index) in values {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark the contextual half of a facing-page folio pair.
|
||||
///
|
||||
/// A landscape PDF can contain two printed pages per PDF page. One folio may be
|
||||
/// isolated while the other touches footer text; they remain a pair because
|
||||
/// they are consecutive, share a deep-margin baseline, and sit on opposite
|
||||
/// sides of the spread. The isolated half is strong evidence that the touching
|
||||
/// half is also a folio.
|
||||
fn mark_spread_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let mut candidates_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
if value.is_some() {
|
||||
candidates_by_page
|
||||
.entry(items[index].page)
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
}
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
for (page, page_candidates) in candidates_by_page {
|
||||
let Some(&(page_left, page_right)) = page_bounds.get(&page) else {
|
||||
continue;
|
||||
};
|
||||
let page_width = page_right - page_left;
|
||||
if page_width <= 0.0 {
|
||||
continue;
|
||||
}
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
let max_pair_font_size = page_width / SPREAD_MIN_CONTENT_WIDTH_EM;
|
||||
let edge_side = |index: usize| {
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Index strong folio evidence by value and spread edge. Sorted
|
||||
// baselines let each contextual candidate query only the two adjacent
|
||||
// values on the opposite edge in O(log n), rather than comparing every
|
||||
// candidate pair on numeric-heavy pages.
|
||||
let mut known_baselines: HashMap<(u32, bool), Vec<f32>> = HashMap::new();
|
||||
for &index in &page_candidates {
|
||||
if (contextual[index] && !explicit_folio[index])
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let value = candidate_values[index].unwrap();
|
||||
known_baselines
|
||||
.entry((value, side))
|
||||
.or_default()
|
||||
.push(items[index].y);
|
||||
}
|
||||
for baselines in known_baselines.values_mut() {
|
||||
baselines.sort_by(f32::total_cmp);
|
||||
}
|
||||
|
||||
for index in page_candidates {
|
||||
if !contextual[index]
|
||||
|| explicit_folio[index]
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(value) = candidate_values[index] else {
|
||||
continue;
|
||||
};
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let y = items[index].y;
|
||||
let paired = [value.checked_sub(1), value.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|other_value| known_baselines.get(&(other_value, !side)))
|
||||
.any(|baselines| {
|
||||
let first = baselines
|
||||
.partition_point(|baseline| *baseline <= y - PAGE_NUMBER_Y_TOLERANCE);
|
||||
baselines
|
||||
.get(first)
|
||||
.is_some_and(|baseline| *baseline < y + PAGE_NUMBER_Y_TOLERANCE)
|
||||
});
|
||||
if paired {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark a contextual folio that alternates with an isolated folio on the
|
||||
/// neighboring PDF page.
|
||||
///
|
||||
/// Facing pages commonly put folios on opposite outer edges. A running header
|
||||
/// can touch the right-hand folio while the next left-hand folio is isolated.
|
||||
/// A single isolated candidate is not enough to remove nearby contextual text.
|
||||
/// Require a second pre-existing anchor in the same advancing sequence, along
|
||||
/// with a genuinely wide content span, stable baselines/font sizes, and
|
||||
/// alternating outer edges.
|
||||
fn mark_adjacent_page_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Do not let newly inferred candidates recursively become evidence for
|
||||
// later candidates; every match must be anchored by evidence established
|
||||
// before this cross-page pass.
|
||||
let strong_folio_evidence = explicit_folio.to_vec();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
let edge_side = |index: usize| {
|
||||
let &(page_left, page_right) = page_bounds.get(&items[index].page)?;
|
||||
let page_width = page_right - page_left;
|
||||
if page_width < items[index].font_size * ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM {
|
||||
return None;
|
||||
}
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Anchor sequences by outer edge and the value/page offset. This enforces
|
||||
// forward page tracking and lets candidates query adjacent pages directly,
|
||||
// while a baseline-sorted index finds a second independent anchor without
|
||||
// a document-wide quadratic scan.
|
||||
let mut anchors_by_page: HashMap<(u32, bool, i64), Vec<usize>> = HashMap::new();
|
||||
let mut anchors_by_sequence: HashMap<(bool, i64), Vec<usize>> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if *value > max_plausible_folio || (contextual[index] && !strong_folio_evidence[index]) {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
anchors_by_page
|
||||
.entry((page, side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
anchors_by_sequence
|
||||
.entry((side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
for anchors in anchors_by_sequence.values_mut() {
|
||||
anchors.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
}
|
||||
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if !contextual[index] || explicit_folio[index] || *value > max_plausible_folio {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
let Some(sequence_anchors) = anchors_by_sequence.get(&(!side, offset)) else {
|
||||
continue;
|
||||
};
|
||||
let neighbor = [page.checked_sub(1), page.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|neighbor_page| anchors_by_page.get(&(neighbor_page, !side, offset)))
|
||||
.flatten()
|
||||
.copied()
|
||||
.find(|&neighbor_index| {
|
||||
(items[index].y - items[neighbor_index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
&& (items[index].font_size - items[neighbor_index].font_size).abs() < 1.0
|
||||
});
|
||||
let Some(neighbor_index) = neighbor else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let first = sequence_anchors.partition_point(|&anchor_index| {
|
||||
items[anchor_index].y <= items[index].y - PAGE_NUMBER_Y_TOLERANCE
|
||||
});
|
||||
let has_second_anchor = sequence_anchors[first..]
|
||||
.iter()
|
||||
.take_while(|&&anchor_index| {
|
||||
items[anchor_index].y < items[index].y + PAGE_NUMBER_Y_TOLERANCE
|
||||
})
|
||||
.any(|&anchor_index| {
|
||||
items[anchor_index].page != page
|
||||
&& items[anchor_index].page != items[neighbor_index].page
|
||||
&& (items[index].font_size - items[anchor_index].font_size).abs() < 1.0
|
||||
});
|
||||
if has_second_anchor {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Identify page-edge numeric items that belong to a nearby content run.
|
||||
///
|
||||
/// Numeric candidates on their own do not establish context for one another.
|
||||
/// A connected same-baseline run is contextual only when it also contains a
|
||||
/// non-candidate item, preserving lines such as `Chapter 1 2026` while still
|
||||
/// removing isolated numeric footer clusters.
|
||||
fn page_number_context_masks(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
document_page_count: usize,
|
||||
) -> (Vec<bool>, Vec<bool>) {
|
||||
let mut contextual = vec![false; items.len()];
|
||||
let mut explicit_folio = vec![false; items.len()];
|
||||
let mut occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>> =
|
||||
HashMap::new();
|
||||
let mut indices_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
for (index, item) in items.iter().enumerate() {
|
||||
if matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) && !item.text.trim().is_empty()
|
||||
{
|
||||
indices_by_page.entry(item.page).or_default().push(index);
|
||||
}
|
||||
}
|
||||
for mut page_indices in indices_by_page.into_values() {
|
||||
page_indices.sort_by(|&left, &right| {
|
||||
items[right]
|
||||
.y
|
||||
.total_cmp(&items[left].y)
|
||||
.then(items[left].x.total_cmp(&items[right].x))
|
||||
});
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in page_indices {
|
||||
if rows.last().is_some_and(|row| {
|
||||
(items[row[0]].y - items[index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
}) {
|
||||
rows.last_mut().unwrap().push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
let mut start = 0;
|
||||
while start < row.len() {
|
||||
let mut end = start + 1;
|
||||
let first = &items[row[start]];
|
||||
let mut group_right = first.x + effective_width(first);
|
||||
let mut group_font_size = first.font_size;
|
||||
|
||||
while end < row.len() {
|
||||
let item = &items[row[end]];
|
||||
let gap = item.x - group_right;
|
||||
if gap > group_font_size.max(item.font_size) * PAGE_NUMBER_CONTEXT_GAP_EM {
|
||||
break;
|
||||
}
|
||||
group_right = group_right.max(item.x + effective_width(item));
|
||||
group_font_size = group_font_size.max(item.font_size);
|
||||
end += 1;
|
||||
}
|
||||
|
||||
let group = &row[start..end];
|
||||
let has_lexical_context = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.chars()
|
||||
.any(|character| character.is_alphabetic())
|
||||
});
|
||||
// Numeric data near a page edge also needs protection, but a
|
||||
// lone long integer beside a short candidate is not enough to
|
||||
// establish context. Preserve explicit numeric structures
|
||||
// (list markers, ranges, comma-formatted values, dotted index
|
||||
// entries) and dense runs with at least one long integer.
|
||||
let numeric_like = |text: &str| {
|
||||
text.chars().any(|character| character.is_numeric())
|
||||
&& !text.chars().any(|character| character.is_alphabetic())
|
||||
};
|
||||
let is_structured_numeric_context = |index: usize| {
|
||||
if candidate_values[index].is_some() {
|
||||
return false;
|
||||
}
|
||||
let text = items[index].text.trim();
|
||||
numeric_like(text)
|
||||
&& text
|
||||
.chars()
|
||||
.any(|character| !character.is_numeric() && !character.is_whitespace())
|
||||
};
|
||||
let has_structured_numeric_context = group
|
||||
.iter()
|
||||
.any(|&index| is_structured_numeric_context(index));
|
||||
let numeric_item_count = group
|
||||
.iter()
|
||||
.filter(|&&index| numeric_like(items[index].text.trim()))
|
||||
.count();
|
||||
let has_long_integer = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_numeric())
|
||||
});
|
||||
let has_dense_numeric_context = numeric_item_count >= 3 && has_long_integer;
|
||||
let has_context = has_lexical_context
|
||||
|| has_structured_numeric_context
|
||||
|| has_dense_numeric_context;
|
||||
let has_candidate = row[start..end]
|
||||
.iter()
|
||||
.any(|&index| candidate_values[index].is_some());
|
||||
// Decorative centered folios have no lexical context, so
|
||||
// recognize the complete delimiter-number-delimiter triplet
|
||||
// before the contextual-content gate. This prevents `- 42 -`
|
||||
// from leaving a malformed `- -` line.
|
||||
if group.len() == 3
|
||||
&& items[group[0]].text.trim() == "-"
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim() == "-"
|
||||
{
|
||||
for &index in group {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
if has_context && has_candidate {
|
||||
let group = &row[start..end];
|
||||
let group_text = group
|
||||
.iter()
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let group_is_folio =
|
||||
crate::text_utils::is_explicit_page_number_expression(&group_text);
|
||||
let context_text = group
|
||||
.iter()
|
||||
.filter(|&&index| candidate_values[index].is_none())
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let candidates: Vec<(usize, u32)> = group
|
||||
.iter()
|
||||
.filter_map(|&index| candidate_values[index].map(|value| (index, value)))
|
||||
.collect();
|
||||
// Recurrence is only evidence for numeric slots at the
|
||||
// outer boundary of a contextual run. An embedded number
|
||||
// in repeated prose such as `Page 42 explains the result`
|
||||
// is substantive content, not a running folio.
|
||||
let recurrence_candidates: Vec<(usize, u32)> = candidates
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(index, _)| {
|
||||
group.first() == Some(index) || group.last() == Some(index)
|
||||
})
|
||||
.collect();
|
||||
let in_deep_margin = candidates.iter().all(|(index, _)| {
|
||||
items[*index].y < PAGE_NUMBER_BOTTOM_Y
|
||||
|| items[*index].y > PAGE_NUMBER_TOP_Y
|
||||
});
|
||||
if in_deep_margin
|
||||
&& !recurrence_candidates.is_empty()
|
||||
&& context_text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphanumeric())
|
||||
.count()
|
||||
>= 8
|
||||
{
|
||||
let signature = group
|
||||
.iter()
|
||||
.map(|&index| {
|
||||
if candidate_values[index].is_some() {
|
||||
"{number}".to_string()
|
||||
} else {
|
||||
items[index]
|
||||
.text
|
||||
.split_whitespace()
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ")
|
||||
.to_lowercase()
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
occurrences_by_signature
|
||||
.entry(signature)
|
||||
.or_default()
|
||||
.push((
|
||||
items[group[0]].page,
|
||||
items[group[0]].y,
|
||||
recurrence_candidates,
|
||||
));
|
||||
}
|
||||
for (position, &index) in group.iter().enumerate() {
|
||||
if let Some(value) = candidate_values[index] {
|
||||
let adjacent_context = [position.checked_sub(1), Some(position + 1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|position| group.get(position).copied())
|
||||
.any(|adjacent| {
|
||||
candidate_values[adjacent].is_none()
|
||||
&& items[adjacent].text.chars().any(|character| {
|
||||
!character.is_numeric() && !character.is_whitespace()
|
||||
})
|
||||
});
|
||||
let max_plausible_folio =
|
||||
(document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Large year/identifier-like values stay attached
|
||||
// to their lexical run even when a smaller numeric
|
||||
// candidate sits between them and the text.
|
||||
let implausible_folio_with_lexical_context =
|
||||
value > max_plausible_folio && has_lexical_context;
|
||||
contextual[index] = adjacent_context
|
||||
|| has_dense_numeric_context
|
||||
|| implausible_folio_with_lexical_context;
|
||||
let previous = position
|
||||
.checked_sub(1)
|
||||
.map(|position| items[group[position]].text.trim());
|
||||
let next = group
|
||||
.get(position + 1)
|
||||
.map(|&index| items[index].text.trim());
|
||||
let follows_page_label =
|
||||
previous.is_some_and(|text| text.eq_ignore_ascii_case("page"));
|
||||
let starts_of_expression =
|
||||
next.is_some_and(|text| text.eq_ignore_ascii_case("of"));
|
||||
let is_centered_folio = previous == Some("-") && next == Some("-");
|
||||
explicit_folio[index] |= group_is_folio
|
||||
&& (follows_page_label
|
||||
|| starts_of_expression
|
||||
|| is_centered_folio);
|
||||
}
|
||||
}
|
||||
// Remove the complete labeled expression rather than
|
||||
// leaving fragments such as `Page of 15`. A trailing
|
||||
// running-header suffix remains untouched.
|
||||
if group_is_folio
|
||||
&& group.len() >= 4
|
||||
&& items[group[0]].text.trim().eq_ignore_ascii_case("page")
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim().eq_ignore_ascii_case("of")
|
||||
&& items[group[3]]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_ascii_digit())
|
||||
{
|
||||
for &index in &group[..4] {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mark_repeated_folio_candidates(
|
||||
occurrences_by_signature,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
mark_spread_folio_pairs(items, candidate_values, &contextual, &mut explicit_folio);
|
||||
mark_adjacent_page_folio_pairs(
|
||||
items,
|
||||
candidate_values,
|
||||
&contextual,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
|
||||
(contextual, explicit_folio)
|
||||
}
|
||||
|
||||
/// Decide which digit-only page-edge items can be removed before layout.
|
||||
///
|
||||
/// PDF producers commonly emit one text-showing operation per word. A numeric
|
||||
/// item attached to neighboring content on the same baseline is therefore kept.
|
||||
/// Complete page-number expressions such as `Page 42` remain removable even
|
||||
/// though their numeric item has lexical context.
|
||||
fn page_number_removal_mask(items: &[TextItem], document_page_count: usize) -> Vec<bool> {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, value)| explicit_folio[index] || (value.is_some() && !contextual[index]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return whether selected-page extraction contains a page-edge number whose
|
||||
/// folio status depends on evidence from other pages. Isolated and explicitly
|
||||
/// labeled folios can be decided locally; only contextual candidates require
|
||||
/// a document-wide extraction pass.
|
||||
pub(super) fn needs_document_page_number_context(
|
||||
items: &[TextItem],
|
||||
document_page_count: usize,
|
||||
) -> bool {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.any(|(index, value)| value.is_some() && contextual[index] && !explicit_folio[index])
|
||||
}
|
||||
|
||||
/// Remove numeric folios using complete document context before downstream
|
||||
/// non-table layout partitions could separate the evidence needed to recognize
|
||||
/// them.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn filter_markdown_page_numbers(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> Vec<TextItem> {
|
||||
filter_markdown_page_numbers_with_removed_pages(items, document_page_count).0
|
||||
}
|
||||
|
||||
/// Filter Markdown folios while retaining the pages where items were removed.
|
||||
///
|
||||
/// The page set lets downstream table-continuation classification preserve its
|
||||
/// pre-filter semantics even though structural layout consumes the cleaned
|
||||
/// item collection.
|
||||
pub(crate) fn filter_markdown_page_numbers_with_removed_pages(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> (Vec<TextItem>, HashSet<u32>, Vec<bool>) {
|
||||
let remove = page_number_removal_mask(&items, document_page_count as usize);
|
||||
let mut removed_pages = HashSet::new();
|
||||
let items = items
|
||||
.into_iter()
|
||||
.zip(remove.iter().copied())
|
||||
.filter_map(|(item, remove)| {
|
||||
if remove {
|
||||
removed_pages.insert(item.page);
|
||||
None
|
||||
} else {
|
||||
Some(item)
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(items, removed_pages, remove)
|
||||
}
|
||||
|
||||
/// Group text items into lines, with multi-column support
|
||||
@@ -1153,6 +1844,22 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds(items, &HashMap::new(), &HashSet::new())
|
||||
}
|
||||
|
||||
/// Group text items into lines without removing numeric page headers or footers.
|
||||
///
|
||||
/// Plain-text extraction uses this path because every extracted item is part of
|
||||
/// the API result. Markdown conversion keeps using [`group_into_lines`], where
|
||||
/// page-number suppression is an intentional presentation cleanup.
|
||||
pub fn group_into_lines_preserving_all_text(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
/// Group text items into lines, using pre-computed per-page adaptive thresholds
|
||||
/// from Canva-style letter-spacing detection. Falls back to computing the
|
||||
/// threshold from item gaps when no pre-computed value is available.
|
||||
@@ -1189,22 +1896,96 @@ pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
)
|
||||
}
|
||||
|
||||
/// Group items after document-level page-number filtering has already run.
|
||||
///
|
||||
/// Partitioned Markdown layout uses this path so a contextual candidate that
|
||||
/// was preserved with its complete baseline context is not reconsidered after
|
||||
/// its neighboring text lands in another band or chart/prose zone.
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
true,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
filter_page_numbers: bool,
|
||||
) -> Vec<TextLine> {
|
||||
if items.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Filter out page numbers (standalone numbers at top/bottom of page)
|
||||
let items: Vec<TextItem> = items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.collect();
|
||||
// Markdown output omits standalone numeric headers/footers. Determine
|
||||
// standalone status from rough baseline context before layout analysis so
|
||||
// removed page numbers cannot affect column detection. Plain-text callers
|
||||
// opt out because dropping extracted text violates that API.
|
||||
let items = if filter_page_numbers {
|
||||
// Item-only grouping has no document metadata, so use the highest
|
||||
// observed 1-based page as its best available coverage denominator.
|
||||
// The Markdown document path passes the authoritative PDF page count
|
||||
// through `filter_markdown_page_numbers` before reaching this helper.
|
||||
let observed_page_count = items
|
||||
.iter()
|
||||
.map(|item| item.page as usize)
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
let remove = page_number_removal_mask(&items, observed_page_count);
|
||||
items
|
||||
.into_iter()
|
||||
.zip(remove)
|
||||
.filter_map(|(item, remove)| (!remove).then_some(item))
|
||||
.collect()
|
||||
} else {
|
||||
items
|
||||
};
|
||||
|
||||
// Get unique pages
|
||||
let mut pages: Vec<u32> = items.iter().map(|i| i.page).collect();
|
||||
@@ -1215,6 +1996,15 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
|
||||
for page in pages {
|
||||
let page_items: Vec<TextItem> = items.iter().filter(|i| i.page == page).cloned().collect();
|
||||
// Page-edge numeric runs are weak evidence for column geometry. Keep
|
||||
// contextual values for line assembly, but prevent their preservation
|
||||
// from changing the page's inferred layout.
|
||||
let column_detection_items: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|item| page_number_value(item).is_none())
|
||||
.cloned()
|
||||
.collect();
|
||||
let column_detection_items = column_detection_items.as_slice();
|
||||
|
||||
// Use pre-computed threshold from fix_letterspaced_items if available
|
||||
// (computed before embedded-space removal, with full signal).
|
||||
@@ -1226,9 +2016,9 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
// their own positioned-region ordering and therefore stay on that path.
|
||||
if !chart_regions.contains_key(&page) {
|
||||
let preliminary_columns =
|
||||
detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
detect_columns(column_detection_items, page, table_pages.contains(&page));
|
||||
let detected_split =
|
||||
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
|
||||
(preliminary_columns.len() == 2).then(|| preliminary_columns[0].x_max);
|
||||
if let Some(band) = image_regions.get(&page).and_then(|regions| {
|
||||
super::reading_order::infer_image_anchored_flow(
|
||||
&page_items,
|
||||
@@ -1268,6 +2058,9 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
let col_input: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
if page_number_value(it).is_some() {
|
||||
return false;
|
||||
}
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Tight bounds: this only blinds the histogram to
|
||||
// chart-internal text; rows adjacent to the chart
|
||||
@@ -1280,7 +2073,7 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
.collect();
|
||||
detect_columns(&col_input, page, table_pages.contains(&page))
|
||||
}
|
||||
None => detect_columns(&page_items, page, table_pages.contains(&page)),
|
||||
None => detect_columns(column_detection_items, page, table_pages.contains(&page)),
|
||||
};
|
||||
|
||||
if columns.len() <= 1 {
|
||||
|
||||
+90
-2
@@ -153,16 +153,54 @@ pub(crate) fn extract_form_fields(
|
||||
},
|
||||
Err(_) => return items,
|
||||
};
|
||||
if fields.is_empty() {
|
||||
return items;
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(doc, field_ref, None, "", page_map, &mut items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
field_ref,
|
||||
None,
|
||||
"",
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
items
|
||||
}
|
||||
|
||||
/// Map widget annotation objects back to the page whose `/Annots` array owns
|
||||
/// them. Some valid widgets omit `/P`, so the page tree is the only reliable
|
||||
/// ownership signal available for page-filtered extraction.
|
||||
fn annotation_page_map(
|
||||
doc: &Document,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
) -> HashMap<ObjectId, u32> {
|
||||
let mut annotation_pages = HashMap::new();
|
||||
for (&page_id, &page_num) in page_map {
|
||||
let Some(annotations) = doc
|
||||
.get_dictionary(page_id)
|
||||
.ok()
|
||||
.and_then(|page| page.get(b"Annots").ok())
|
||||
.and_then(|annotations| resolve_array(doc, annotations))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
for annotation in annotations {
|
||||
if let Ok(annotation_id) = annotation.as_reference() {
|
||||
annotation_pages.insert(annotation_id, page_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
annotation_pages
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
@@ -170,6 +208,7 @@ pub(crate) fn walk_form_fields(
|
||||
parent_ft: Option<&[u8]>,
|
||||
parent_name: &str,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
) {
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
@@ -206,7 +245,15 @@ pub(crate) fn walk_form_fields(
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(doc, kid_ref, ft, &full_name, page_map, items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
kid_ref,
|
||||
ft,
|
||||
&full_name,
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -299,6 +346,7 @@ pub(crate) fn walk_form_fields(
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.and_then(|p| page_map.get(&p).copied())
|
||||
.or_else(|| annotation_pages.get(&field_id).copied())
|
||||
.unwrap_or(1);
|
||||
|
||||
let text = if full_name.is_empty() {
|
||||
@@ -324,3 +372,43 @@ pub(crate) fn walk_form_fields(
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::{dictionary, Object};
|
||||
|
||||
#[test]
|
||||
fn widget_without_page_reference_uses_owning_page_annotation() {
|
||||
let mut doc = Document::new();
|
||||
let widget_id = doc.add_object(dictionary! {
|
||||
"Type" => "Annot",
|
||||
"Subtype" => "Widget",
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("customer"),
|
||||
"V" => Object::string_literal("Alice"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
});
|
||||
let page_one_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
});
|
||||
let page_two_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Annots" => vec![Object::Reference(widget_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(widget_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::from([(page_one_id, 1), (page_two_id, 2)]);
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
}
|
||||
|
||||
+860
-22
@@ -2,6 +2,7 @@
|
||||
//!
|
||||
//! This module extracts text with position information for structure detection.
|
||||
|
||||
mod base14;
|
||||
pub(crate) mod content_stream;
|
||||
mod fonts;
|
||||
mod layout;
|
||||
@@ -27,12 +28,15 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
pub use layout::group_into_lines;
|
||||
#[cfg(test)]
|
||||
use layout::filter_markdown_page_numbers;
|
||||
pub(crate) use layout::filter_markdown_page_numbers_with_removed_pages;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
@@ -81,17 +85,33 @@ pub fn extract_text_with_positions_pages<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<Vec<TextItem>, PdfError> {
|
||||
let (items, _rects, _lines) = extract_text_with_positions_and_rects(path, page_filter)?;
|
||||
let (items, _rects, _lines) =
|
||||
extract_text_with_positions_and_rects_with_password(path, page_filter, None)?;
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
/// Extract text with positions and rectangles from a file.
|
||||
pub(crate) fn extract_text_with_positions_and_rects<P: AsRef<Path>>(
|
||||
/// Extract text with positions from a file, limited to specific pages and
|
||||
/// decrypting with `password` when the PDF is encrypted.
|
||||
///
|
||||
/// `page_filter` is an optional set of 1-indexed page numbers to process.
|
||||
/// When `None`, all pages are processed.
|
||||
pub fn extract_text_with_positions_pages_with_password<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<Vec<TextItem>, PdfError> {
|
||||
let (items, _rects, _lines) =
|
||||
extract_text_with_positions_and_rects_with_password(path, page_filter, password)?;
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
pub(crate) fn extract_text_with_positions_and_rects_with_password<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
let (doc, _) = crate::load_document_from_path_with_password(&path, password)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
@@ -140,17 +160,92 @@ pub(crate) fn extract_positioned_text_from_doc(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false)
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false, None)
|
||||
}
|
||||
|
||||
/// Extract with option to include invisible (Tr=3) text.
|
||||
/// Used for Mixed/template PDFs where the OCR text layer is invisible.
|
||||
pub(crate) fn extract_positioned_text_include_invisible(
|
||||
/// Extract selected pages and gather document-wide folio evidence only when a
|
||||
/// selected page contains an ambiguous contextual page-edge number. Errors on
|
||||
/// selected pages remain fatal; errors on context-only pages are skipped.
|
||||
pub(crate) fn extract_positioned_text_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, true)
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, false)
|
||||
}
|
||||
|
||||
/// Invisible-text variant of [`extract_positioned_text_with_folio_context`].
|
||||
pub(crate) fn extract_positioned_text_include_invisible_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, true)
|
||||
}
|
||||
|
||||
fn extract_positioned_text_with_folio_context_impl(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let Some(required_pages) = page_filter else {
|
||||
return extract_positioned_text_impl(doc, font_cmaps, None, include_invisible, None);
|
||||
};
|
||||
|
||||
let (
|
||||
(mut selected_items, mut selected_rects, mut selected_lines),
|
||||
mut page_thresholds,
|
||||
mut gid_encoded_pages,
|
||||
) = extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(required_pages),
|
||||
include_invisible,
|
||||
None,
|
||||
)?;
|
||||
if !layout::needs_document_page_number_context(&selected_items, doc.get_pages().len()) {
|
||||
return Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
));
|
||||
}
|
||||
|
||||
let context_pages: HashSet<u32> = doc
|
||||
.get_pages()
|
||||
.keys()
|
||||
.copied()
|
||||
.filter(|page| !required_pages.contains(page))
|
||||
.collect();
|
||||
let ((context_items, context_rects, context_lines), context_thresholds, context_gid_pages) =
|
||||
extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(&context_pages),
|
||||
include_invisible,
|
||||
Some(required_pages),
|
||||
)?;
|
||||
selected_items.extend(context_items);
|
||||
selected_rects.extend(context_rects);
|
||||
selected_lines.extend(context_lines);
|
||||
page_thresholds.extend(context_thresholds);
|
||||
gid_encoded_pages.extend(context_gid_pages);
|
||||
Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
))
|
||||
}
|
||||
|
||||
/// Extract all pages for document-wide analysis while allowing malformed
|
||||
/// unselected pages to be skipped. Any requested page still fails normally.
|
||||
pub(crate) fn extract_positioned_text_for_document_analysis(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
required_pages: &HashSet<u32>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, None, false, Some(required_pages))
|
||||
}
|
||||
|
||||
fn extract_positioned_text_impl(
|
||||
@@ -158,6 +253,7 @@ fn extract_positioned_text_impl(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
required_pages: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let pages = doc.get_pages();
|
||||
let mut all_items = Vec::new();
|
||||
@@ -179,15 +275,25 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) =
|
||||
extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
let page_result = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
);
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) = match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error) if required_pages.is_some_and(|required| !required.contains(page_num)) => {
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
@@ -317,7 +423,9 @@ fn extract_positioned_text_impl(
|
||||
}
|
||||
|
||||
// Extract AcroForm field values
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num);
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num)
|
||||
.into_iter()
|
||||
.filter(|item| page_filter.is_none_or(|filter| filter.contains(&item.page)));
|
||||
all_items.extend(form_items);
|
||||
|
||||
Ok((
|
||||
@@ -1519,6 +1627,736 @@ mod tests {
|
||||
assert_eq!(lines[1].text(), "Next line");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserving_all_text_keeps_numeric_page_footer() {
|
||||
let mut page_number = make_merge_item("42", 100.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
assert!(group_into_lines(vec![page_number.clone()]).is_empty());
|
||||
|
||||
let lines = group_into_lines_preserving_all_text(vec![page_number]);
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_numeric_run_near_page_edge_is_not_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Total 730 seats");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_page_footer_separated_from_label_is_removed() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut footer_label = make_merge_item("DOCUMENT FOOTER", 60.0, 100.0);
|
||||
footer_label.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, footer_label]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "DOCUMENT FOOTER");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decorative_marker_does_not_contextualize_numeric_page_footer() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut page_number = make_merge_item("42", 37.0, 10.0);
|
||||
page_number.y = 30.0;
|
||||
let mut footer_label = make_merge_item("Company report footer", 68.0, 120.0);
|
||||
footer_label.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, page_number, footer_label]);
|
||||
|
||||
assert!(lines.iter().all(|line| !line.text().contains("42")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_is_removed_in_a_short_document() {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item("42", 57.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![label, page_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_with_running_header_suffix_is_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("of", 73.0, 12.0),
|
||||
make_merge_item("100", 89.0, 18.0),
|
||||
make_merge_item("Report header", 111.0, 78.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report header");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_of_total_expression_is_removed_without_leaving_fragments() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 482.0, 27.0),
|
||||
make_merge_item("1", 513.0, 6.0),
|
||||
make_merge_item("of", 523.0, 10.0),
|
||||
make_merge_item("15", 537.0, 12.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 46.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn document_folio_filter_survives_per_page_layout_splitting() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
items.extend([label, page_number]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 3);
|
||||
assert!(filtered
|
||||
.iter()
|
||||
.all(|item| !matches!(item.text.as_str(), "42" | "43" | "44")));
|
||||
let mut lines = Vec::new();
|
||||
for page in 1..=3 {
|
||||
let page_items = filtered
|
||||
.iter()
|
||||
.filter(|item| item.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
lines.extend(
|
||||
group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
page_items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert!(lines.iter().all(|line| line.text() == "Page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_page_edge_runs_do_not_contextualize_folios() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut long_number = make_merge_item("12345", 43.0, 30.0);
|
||||
long_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, long_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12345");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structured_and_dense_numeric_page_edge_runs_are_preserved() {
|
||||
let mut list_marker = make_merge_item("11)", 25.0, 18.0);
|
||||
list_marker.y = 50.0;
|
||||
let mut chapter = make_merge_item("13", 47.0, 12.0);
|
||||
chapter.y = 50.0;
|
||||
|
||||
let mut isbn_prefix = make_merge_item("9", 25.0, 6.0);
|
||||
isbn_prefix.page = 2;
|
||||
isbn_prefix.y = 50.0;
|
||||
let mut isbn_mid = make_merge_item("780113", 35.0, 36.0);
|
||||
isbn_mid.page = 2;
|
||||
isbn_mid.y = 50.0;
|
||||
let mut isbn_end = make_merge_item("227426", 75.0, 36.0);
|
||||
isbn_end.page = 2;
|
||||
isbn_end.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![list_marker, chapter, isbn_prefix, isbn_mid, isbn_end]);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "11) 13");
|
||||
assert_eq!(lines[1].text(), "9 780113 227426");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn incrementing_numeric_body_column_is_not_treated_as_a_folio() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "13"), (2, "14"), (3, "15")] {
|
||||
let mut row_number = make_merge_item(value, 72.0, 12.0);
|
||||
row_number.page = page;
|
||||
row_number.y = 730.0;
|
||||
let mut name = make_merge_item("Person", 90.0, 42.0);
|
||||
name.page = page;
|
||||
name.y = 730.0;
|
||||
items.extend([row_number, name]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert_eq!(lines[0].text(), "13 Person");
|
||||
assert_eq!(lines[1].text(), "14 Person");
|
||||
assert_eq!(lines[2].text(), "15 Person");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn advancing_number_in_repeated_deep_margin_footer_is_removed() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_substantive_page_number_prose_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44"), (4, "45")] {
|
||||
let mut page_label = make_merge_item("Page", 25.0, 28.0);
|
||||
page_label.page = page;
|
||||
page_label.y = 30.0;
|
||||
let mut number = make_merge_item(value, 57.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut explanation = make_merge_item("explains the result", 73.0, 108.0);
|
||||
explanation.page = page;
|
||||
explanation.y = 30.0;
|
||||
items.extend([page_label, number, explanation]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
for (line, value) in lines.iter().zip(["42", "43", "44", "45"]) {
|
||||
assert_eq!(line.text(), format!("Page {value} explains the result"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_candidates_do_not_bridge_lexical_context() {
|
||||
let mut report = make_merge_item("Report", 25.0, 40.0);
|
||||
report.y = 30.0;
|
||||
let mut year = make_merge_item("2026", 69.0, 24.0);
|
||||
year.y = 30.0;
|
||||
let mut folio = make_merge_item("42", 97.0, 12.0);
|
||||
folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![report, year, folio]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_folio_delimiters_are_removed_with_the_number() {
|
||||
let mut left = make_merge_item("-", 270.0, 6.0);
|
||||
left.y = 30.0;
|
||||
let mut number = make_merge_item("42", 280.0, 12.0);
|
||||
number.y = 30.0;
|
||||
let mut right = make_merge_item("-", 296.0, 6.0);
|
||||
right.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![left, number, right]);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_delimiters_inside_substantive_text_are_preserved() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Result", 240.0, 36.0),
|
||||
make_merge_item("-", 280.0, 6.0),
|
||||
make_merge_item("42", 290.0, 12.0),
|
||||
make_merge_item("-", 306.0, 6.0),
|
||||
make_merge_item("approved", 316.0, 48.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 30.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Result-42-approved");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn changing_year_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, year) in [(1, "2020"), (2, "2021"), (3, "2022"), (4, "2023")] {
|
||||
let mut year = make_merge_item(year, 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert_eq!(lines[0].text(), "2020 Annual report");
|
||||
assert_eq!(lines[3].text(), "2023 Annual report");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_repeated_margin_numbers_do_not_meet_the_folio_evidence_floor() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
if page <= 2 {
|
||||
let value = if page == 1 { "2" } else { "4" };
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
} else {
|
||||
let mut body = make_merge_item("Body text", 72.0, 54.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
}
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "2 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "4 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_document_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (10, "10"), (19, "19"), (28, "28")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "1 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "28 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_blank_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 20);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().any(|item| item.text == "4"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefiltered_contextual_number_survives_layout_partitioning() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 1);
|
||||
let partitioned_number: Vec<TextItem> = filtered
|
||||
.into_iter()
|
||||
.filter(|item| item.text == "730")
|
||||
.collect();
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
partitioned_number,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "730");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_partition_does_not_define_columns() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..20 {
|
||||
let y = 90.0 - row as f32 * 4.0;
|
||||
let mut left = make_merge_item(&(row + 1).to_string(), 50.0, 20.0);
|
||||
left.y = y;
|
||||
let mut right = make_merge_item(&(row + 101).to_string(), 350.0, 20.0);
|
||||
right.y = y;
|
||||
items.extend([left, right]);
|
||||
}
|
||||
assert_eq!(detect_columns(&items, 1, false).len(), 2);
|
||||
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 20);
|
||||
assert!(lines.iter().all(|line| line.items.len() == 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separated_content_is_not_treated_as_a_spread_folio_pair() {
|
||||
let mut value = make_merge_item("12", 100.0, 12.0);
|
||||
value.y = 30.0;
|
||||
let mut label = make_merge_item("Total", 116.0, 30.0);
|
||||
label.y = 30.0;
|
||||
let mut unrelated_number = make_merge_item("13", 300.0, 12.0);
|
||||
unrelated_number.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![value, label, unrelated_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12 Total");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_folio_uses_the_full_page_edge_band() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 80.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 80.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folio_on_facing_page_spread_is_removed() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut left_folio = make_merge_item("326", 35.0, 17.0);
|
||||
left_folio.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 61.0, 120.0);
|
||||
footer.y = 30.0;
|
||||
let mut right_folio = make_merge_item("327", 1148.0, 17.0);
|
||||
right_folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, left_folio, footer, right_folio]);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| !line.text().contains("326") && !line.text().contains("327")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folios_alternating_across_pages_are_removed() {
|
||||
let headers = [
|
||||
"Letter to shareholders",
|
||||
"Corporate governance report",
|
||||
"Business environment overview",
|
||||
"Consolidated financial statements",
|
||||
];
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=8 {
|
||||
let mut body = make_merge_item("Body text", 50.0, 500.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
|
||||
let mut folio = make_merge_item(&(page + 22).to_string(), 0.0, 14.0);
|
||||
folio.page = page;
|
||||
folio.y = 780.0;
|
||||
if page % 2 == 0 {
|
||||
folio.x = 50.0;
|
||||
items.push(folio);
|
||||
} else {
|
||||
let mut header = make_merge_item(headers[(page / 2) as usize], 350.0, 180.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
folio.x = 536.0;
|
||||
items.extend([header, folio]);
|
||||
}
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 8);
|
||||
|
||||
assert!(filtered.iter().all(|item| {
|
||||
!matches!(
|
||||
item.text.as_str(),
|
||||
"23" | "24" | "25" | "26" | "27" | "28" | "29" | "30"
|
||||
)
|
||||
}));
|
||||
assert!(headers
|
||||
.iter()
|
||||
.all(|header| filtered.iter().any(|item| item.text == *header)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_isolated_neighbor_does_not_remove_contextual_number() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 450.0, 70.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 526.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated = make_merge_item("2", 50.0, 7.0);
|
||||
isolated.page = 2;
|
||||
isolated.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, label, contextual, body_two, isolated], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().all(|item| item.text != "2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_content_span_does_not_establish_adjacent_page_edges() {
|
||||
let mut body_one = make_merge_item("Body text", 100.0, 120.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 170.0, 60.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 235.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated_two = make_merge_item("2", 100.0, 7.0);
|
||||
isolated_two.page = 2;
|
||||
isolated_two.y = 780.0;
|
||||
|
||||
let mut body_four = body_one.clone();
|
||||
body_four.page = 4;
|
||||
let mut isolated_four = make_merge_item("4", 100.0, 7.0);
|
||||
isolated_four.page = 4;
|
||||
isolated_four.y = 780.0;
|
||||
|
||||
let filtered = filter_markdown_page_numbers(
|
||||
vec![
|
||||
body_one,
|
||||
label,
|
||||
contextual,
|
||||
body_two,
|
||||
isolated_two,
|
||||
body_four,
|
||||
isolated_four,
|
||||
],
|
||||
4,
|
||||
);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_edge_number_on_an_adjacent_page_is_not_folio_evidence() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut isolated = make_merge_item("42", 50.0, 14.0);
|
||||
isolated.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut contextual = make_merge_item("43", 50.0, 14.0);
|
||||
contextual.page = 2;
|
||||
contextual.y = 780.0;
|
||||
let mut label = make_merge_item("cases reviewed", 70.0, 90.0);
|
||||
label.page = 2;
|
||||
label.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, isolated, body_two, contextual, label], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "43"));
|
||||
assert!(filtered.iter().any(|item| item.text == "cases reviewed"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_number_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
let mut year = make_merge_item("2026", 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines.iter().all(|line| line.text() == "2026 Annual report"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_prefix_does_not_remove_substantive_text_during_layout() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("explains", 73.0, 44.0),
|
||||
make_merge_item("the result", 121.0, 55.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42 explains the result");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_page_number_prefix_with_substantive_text_is_preserved_during_layout() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value, chapter) in [(1, "42", "Chapter 1"), (2, "43", "Chapter 2")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
let mut suffix = make_merge_item(chapter, 73.0, 58.0);
|
||||
suffix.page = page;
|
||||
suffix.y = 50.0;
|
||||
items.extend([label, page_number, suffix]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "Page 42 Chapter 1");
|
||||
assert_eq!(lines[1].text(), "Page 43 Chapter 2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_phrase_in_the_page_body_is_preserved() {
|
||||
let items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
];
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_numeric_context_near_page_edge_is_preserved() {
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
|
||||
let chapter_lines = group_into_lines(vec![chapter, chapter_number]);
|
||||
assert_eq!(chapter_lines.len(), 1);
|
||||
assert_eq!(chapter_lines[0].text(), "Chapter 1");
|
||||
|
||||
let mut year = make_merge_item("2026", 100.0, 24.0);
|
||||
year.y = 760.0;
|
||||
let mut report = make_merge_item("Report", 130.0, 36.0);
|
||||
report.y = 760.0;
|
||||
|
||||
let report_lines = group_into_lines(vec![year, report]);
|
||||
assert_eq!(report_lines.len(), 1);
|
||||
assert_eq!(report_lines[0].text(), "2026 Report");
|
||||
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
let mut edition_year = make_merge_item("2026", 163.0, 24.0);
|
||||
edition_year.y = 760.0;
|
||||
|
||||
let chained_lines = group_into_lines(vec![chapter, chapter_number, edition_year]);
|
||||
assert_eq!(chained_lines.len(), 1);
|
||||
assert_eq!(chained_lines[0].text(), "Chapter 1 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bold_italic_detection() {
|
||||
// Test bold detection
|
||||
|
||||
+385
-17
@@ -44,6 +44,16 @@ const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
/// Strikeout decorations are text-sized. Diagram connectors, signature
|
||||
/// lines, and chart rules often cross glyphs too, but extend well beyond the
|
||||
/// text they happen to intersect.
|
||||
const STRIKE_OWNER_PAD_EM: f32 = 0.75;
|
||||
const STRIKE_OWNER_MIN_PAD: f32 = 4.0;
|
||||
const STRIKE_ROW_Y_TOLERANCE_EM: f32 = 0.15;
|
||||
const STRIKE_ROW_Y_TOLERANCE_MIN: f32 = 5.0;
|
||||
const GRAPHIC_CONNECTION_EPS: f32 = 2.0;
|
||||
const GRAPHIC_CONNECTOR_MAX_THICKNESS: f32 = 4.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
@@ -387,6 +397,206 @@ fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
fn is_bare_list_marker(text: &str) -> bool {
|
||||
matches!(
|
||||
text.trim(),
|
||||
"•" | "◦" | "▪" | "▫" | "‣" | "⁃" | "●" | "○" | "■" | "□" | "-" | "*"
|
||||
)
|
||||
}
|
||||
|
||||
fn same_strike_row(left: &TextItem, right: &TextItem) -> bool {
|
||||
let font_size = left.font_size.max(right.font_size);
|
||||
let tolerance = (font_size * STRIKE_ROW_Y_TOLERANCE_EM).max(STRIKE_ROW_Y_TOLERANCE_MIN);
|
||||
(left.y - right.y).abs() <= tolerance
|
||||
}
|
||||
|
||||
fn is_inline_script(rule: &Rule, candidate: &TextItem, parent: &TextItem) -> bool {
|
||||
if !is_underline_candidate(candidate)
|
||||
|| is_bare_list_marker(&candidate.text)
|
||||
|| candidate.font_size <= 0.0
|
||||
|| candidate.font_size >= parent.font_size * 0.75
|
||||
|| candidate.text.len() > 4
|
||||
|| !candidate.text.chars().all(|c| c.is_ascii_digit())
|
||||
|| (candidate.y - parent.y).abs() > 5.0
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_ends_with_letter = parent.text.chars().last().is_some_and(char::is_alphabetic);
|
||||
if !parent_ends_with_letter {
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_right = parent.x + parent.width;
|
||||
let gap = candidate.x - parent_right;
|
||||
if gap >= parent.font_size * 0.2 || gap <= -parent.font_size * 0.3 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = rule.x2.min(candidate.x + candidate.width) - rule.x1.max(candidate.x);
|
||||
overlap >= candidate.width * MIN_X_OVERLAP
|
||||
}
|
||||
|
||||
/// Return the items owned by a snug mid-glyph rule.
|
||||
///
|
||||
/// Real strikeout decorations track the width of the deleted text, including
|
||||
/// runs split by font/style changes and adjacent numeric super/subscripts.
|
||||
/// Non-text graphics can cross the same vertical window, but arrow shafts,
|
||||
/// signature lines, fraction bars, and chart rules extend materially beyond
|
||||
/// the intersected glyphs. Requiring the rule to stay within a small em-sized
|
||||
/// pad of a contiguous matched row separates those cases without relying on
|
||||
/// document-specific fonts or coordinates.
|
||||
///
|
||||
/// Ownership is computed once per rule. This keeps the strikeout pass at the
|
||||
/// same rule-by-item scale as underline detection instead of rescanning the
|
||||
/// whole page for every matching item.
|
||||
fn snug_strike_owner_indices(rule: &Rule, items: &[TextItem]) -> Vec<usize> {
|
||||
let mut struck_indices: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
(is_underline_candidate(candidate)
|
||||
&& !is_bare_list_marker(&candidate.text)
|
||||
&& rule_strikes_item(rule, candidate))
|
||||
.then_some(index)
|
||||
})
|
||||
.collect();
|
||||
if struck_indices.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
struck_indices.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in struck_indices {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| same_strike_row(&items[row[0]], &items[index]))
|
||||
{
|
||||
row.push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
let mut owned_indices = Vec::new();
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
|
||||
// Underline detection runs before the extractor's script-merging
|
||||
// pass. Include the same tightly adjacent numeric script shape here
|
||||
// when the rule spans it, so the owner width and semantic mark both
|
||||
// survive that later merge.
|
||||
let scripts: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
let parent_pos =
|
||||
row.partition_point(|&row_index| items[row_index].x <= candidate.x);
|
||||
let parent_index = parent_pos.checked_sub(1).map(|pos| row[pos])?;
|
||||
is_inline_script(rule, candidate, &items[parent_index]).then_some(index)
|
||||
})
|
||||
.collect();
|
||||
row.extend(scripts);
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
row.dedup();
|
||||
|
||||
let x1 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x2 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x + items[index].width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let max_font_size = row
|
||||
.iter()
|
||||
.map(|&index| items[index].font_size)
|
||||
.fold(0.0, f32::max);
|
||||
let pad = (max_font_size * STRIKE_OWNER_PAD_EM).max(STRIKE_OWNER_MIN_PAD);
|
||||
|
||||
if rule.x1 < x1 - pad || rule.x2 > x2 + pad {
|
||||
continue;
|
||||
}
|
||||
|
||||
let contiguous = row.windows(2).all(|pair| {
|
||||
let gap = items[pair[1]].x - (items[pair[0]].x + items[pair[0]].width);
|
||||
gap <= (max_font_size * 2.0).max(12.0)
|
||||
});
|
||||
if contiguous {
|
||||
owned_indices.extend(row);
|
||||
}
|
||||
}
|
||||
|
||||
owned_indices.sort_unstable();
|
||||
owned_indices.dedup();
|
||||
owned_indices
|
||||
}
|
||||
|
||||
/// Diagram and table rules participate in larger path geometry. A vertical
|
||||
/// or diagonal segment meeting the candidate rule is strong evidence that
|
||||
/// the horizontal segment is a connector, border, arrow, or symbol rather
|
||||
/// than an isolated text decoration.
|
||||
fn has_connected_nonhorizontal_segment(rule: &Rule, lines: &[UnderlineLine], page: u32) -> bool {
|
||||
lines.iter().any(|line| {
|
||||
if line.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let dx = line.x2 - line.x1;
|
||||
let dy = line.y2 - line.y1;
|
||||
if dy.abs() <= MAX_RULE_THICKNESS {
|
||||
return false;
|
||||
}
|
||||
|
||||
let y_min = line.y1.min(line.y2) - GRAPHIC_CONNECTION_EPS;
|
||||
let y_max = line.y1.max(line.y2) + GRAPHIC_CONNECTION_EPS;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let t = (rule.y - line.y1) / dy;
|
||||
if !(-0.05..=1.05).contains(&t) {
|
||||
return false;
|
||||
}
|
||||
let intersection_x = line.x1 + t * dx;
|
||||
intersection_x >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& intersection_x <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Filled diagrams often build connectors from intersecting thin rectangles
|
||||
/// instead of stroked path segments. Treat only narrow, vertically elongated
|
||||
/// rectangles as connector geometry; broad fills can legitimately sit behind
|
||||
/// struck text and must not veto its decoration.
|
||||
fn has_connected_nonhorizontal_rect(rule: &Rule, rects: &[PdfRect], page: u32) -> bool {
|
||||
rects.iter().any(|rect| {
|
||||
if rect.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let (x1, x2) = if rect.width >= 0.0 {
|
||||
(rect.x, rect.x + rect.width)
|
||||
} else {
|
||||
(rect.x + rect.width, rect.x)
|
||||
};
|
||||
let (y1, y2) = if rect.height >= 0.0 {
|
||||
(rect.y, rect.y + rect.height)
|
||||
} else {
|
||||
(rect.y + rect.height, rect.y)
|
||||
};
|
||||
let width = x2 - x1;
|
||||
let height = y2 - y1;
|
||||
|
||||
width > 0.0
|
||||
&& width <= GRAPHIC_CONNECTOR_MAX_THICKNESS
|
||||
&& height > width * 2.0
|
||||
&& rule.y >= y1 - GRAPHIC_CONNECTION_EPS
|
||||
&& rule.y <= y2 + GRAPHIC_CONNECTION_EPS
|
||||
&& x2 >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& x1 <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
@@ -440,27 +650,43 @@ pub(crate) fn mark_underlined_items(
|
||||
.map(|(i, _)| i)
|
||||
.collect();
|
||||
|
||||
for item in items.iter_mut() {
|
||||
let mut strikeout_items = vec![false; items.len()];
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx)
|
||||
|| has_connected_nonhorizontal_segment(rule, lines, page)
|
||||
|| has_connected_nonhorizontal_rect(rule, rects, page)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
for item_idx in snug_strike_owner_indices(rule, items) {
|
||||
strikeout_items[item_idx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
let underlined_items: HashSet<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
is_underline_candidate(item)
|
||||
&& rules.iter().enumerate().any(|(rule_idx, rule)| {
|
||||
!tabular_rules.contains(&rule_idx)
|
||||
&& !fraction_rules.contains(&rule_idx)
|
||||
&& rule_matches_item(rule, item)
|
||||
})
|
||||
})
|
||||
.map(|(item_idx, _)| item_idx)
|
||||
.collect();
|
||||
|
||||
for (item_idx, item) in items.iter_mut().enumerate() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
// The fraction guard only gates UNDERLINE marking — a rule that
|
||||
// reads as a fraction bar from below can still legitimately
|
||||
// strike through a line above it.
|
||||
if !fraction_rules.contains(&rule_idx) && rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
if strikeout_items[item_idx] {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if underlined_items.contains(&item_idx) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -580,6 +806,148 @@ mod tests {
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_connector_crossing_text_is_not_a_strikeout() {
|
||||
let mut items = vec![item("diagram label", 160.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 280.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_rule_ending_inside_short_label_is_not_a_strikeout() {
|
||||
let mut items = vec![item("T 18", 300.0, 500.0, 20.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 315.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_diagram_segment_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let lines = vec![
|
||||
hline(99.0, 113.0, 503.0),
|
||||
UnderlineLine {
|
||||
x1: 106.0,
|
||||
y1: 496.0,
|
||||
x2: 109.0,
|
||||
y2: 510.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_filled_rect_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 14.0),
|
||||
PdfRect {
|
||||
x: 106.0,
|
||||
y: 496.0,
|
||||
width: 2.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn broad_fill_behind_text_does_not_block_strikeout() {
|
||||
let mut items = vec![item("deleted", 100.0, 500.0, 40.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 42.0),
|
||||
PdfRect {
|
||||
x: 90.0,
|
||||
y: 490.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_bullet_is_not_a_strikeout() {
|
||||
for marker in ["•", "-", "*"] {
|
||||
let mut items = vec![item(marker, 100.0, 500.0, 6.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 107.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout, "marker {marker:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_marks_adjacent_split_runs_as_strikeout() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 500.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_groups_split_runs_with_baseline_drift() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 498.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_superscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 503.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_subscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 497.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
|
||||
@@ -8,8 +8,9 @@ use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache, FontStyleCache,
|
||||
build_font_encodings, build_font_widths, build_type3_scales, compute_string_width_ts,
|
||||
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
FontStyleCache,
|
||||
};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
@@ -162,10 +163,11 @@ fn extract_form_xobject_text_inner(
|
||||
|
||||
// Get fonts from the Form's Resources
|
||||
let form_fonts = get_form_fonts(doc, &stream.dict);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
|
||||
|
||||
// Build font width info for the form
|
||||
let font_widths = build_font_widths(doc, &form_fonts);
|
||||
let type3_scales = build_type3_scales(doc, &form_fonts);
|
||||
|
||||
// Build font base names and ToUnicode refs for the form
|
||||
let mut font_base_names: HashMap<String, String> = HashMap::new();
|
||||
@@ -413,7 +415,8 @@ fn extract_form_xobject_text_inner(
|
||||
&font_widths,
|
||||
) {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = if let Some(font_info) = font_widths.get(¤t_font) {
|
||||
if let Some(raw_bytes) = get_operand_bytes(&op.operands[0]) {
|
||||
@@ -572,7 +575,8 @@ fn extract_form_xobject_text_inner(
|
||||
}
|
||||
if !sub_items.is_empty() {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined)
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let base_font = font_base_names
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
|
||||
+264
-33
@@ -50,11 +50,11 @@ pub use detector::{
|
||||
};
|
||||
pub use extractor::{
|
||||
extract_text, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
extract_text_with_positions_pages,
|
||||
extract_text_with_positions_pages, extract_text_with_positions_pages_with_password,
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
MarkdownProfile,
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -68,6 +68,40 @@ use text_quality::{
|
||||
};
|
||||
use tounicode::FontCMaps;
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
struct ProcessingTimer(std::time::Instant);
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
struct ProcessingTimer;
|
||||
|
||||
impl ProcessingTimer {
|
||||
fn start() -> Self {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
Self(std::time::Instant::now())
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
Self
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(&self) -> u64 {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
self.0.elapsed().as_millis() as u64
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
// The wasm32-unknown-unknown standard library has no clock.
|
||||
// Browser bindings measure with JavaScript's host clock.
|
||||
0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reason emitted when the extracted text layer appears garbled due to
|
||||
/// broken font decoding or mojibake.
|
||||
pub const OCR_REASON_SUSPECTED_GARBLED_TEXT: &str = "suspected_garbled_text";
|
||||
@@ -250,7 +284,7 @@ pub fn process_pdf_with_options<P: AsRef<Path>>(
|
||||
path: P,
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = std::time::Instant::now();
|
||||
let start = ProcessingTimer::start();
|
||||
validate_pdf_file(&path)?;
|
||||
|
||||
// Load the document once — shared by detection AND extraction.
|
||||
@@ -277,7 +311,7 @@ pub fn process_pdf_mem_with_options(
|
||||
buffer: &[u8],
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = std::time::Instant::now();
|
||||
let start = ProcessingTimer::start();
|
||||
validate_pdf_bytes(buffer)?;
|
||||
|
||||
let (doc, page_count) =
|
||||
@@ -428,16 +462,39 @@ pub fn extract_pages_markdown_mem(
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats.
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
// unselected page cannot make a valid requested page fail, but errors on a
|
||||
// requested page retain the normal extraction semantics.
|
||||
let required_pages: Option<HashSet<u32>> = pages.map(|pages| {
|
||||
pages
|
||||
.iter()
|
||||
.filter_map(|page| page.checked_add(1))
|
||||
.collect()
|
||||
});
|
||||
let ((all_items, all_rects, all_lines), page_thresholds, gid_pages) =
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||
if let Some(required_pages) = required_pages.as_ref() {
|
||||
extractor::extract_positioned_text_for_document_analysis(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
required_pages,
|
||||
)?
|
||||
} else {
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?
|
||||
};
|
||||
let text_quality = analyze_text_quality(&all_items);
|
||||
|
||||
// Compute layout complexity from full document (near-zero cost).
|
||||
let complexity = compute_layout_complexity(&all_items, &all_rects, &all_lines);
|
||||
// Resolve page numbers with full-document context before partitioning.
|
||||
// Per-page Markdown receives the original items plus these decisions so
|
||||
// table detection can retain legitimate numeric cells.
|
||||
let (filtered_items, removed_page_number_pages, page_number_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
|
||||
// Tables need the original numeric cells; columns use folio-cleaned
|
||||
// evidence so removed page numbers cannot create false layout metadata.
|
||||
let complexity = compute_layout_complexity(&all_items, &filtered_items, &all_rects, &all_lines);
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -468,12 +525,13 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
let page_1idx = page_0idx + 1;
|
||||
|
||||
// Filter items/rects for this page only
|
||||
let page_items: Vec<TextItem> = all_items
|
||||
// Partition items, removal decisions, and rects for this page only.
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page_1idx)
|
||||
.cloned()
|
||||
.collect();
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
let page_rects: Vec<PdfRect> = all_rects
|
||||
.iter()
|
||||
@@ -500,9 +558,14 @@ pub fn extract_pages_markdown_mem(
|
||||
options,
|
||||
&page_rects,
|
||||
&[],
|
||||
&page_thresholds,
|
||||
None,
|
||||
&[],
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_page_number_pages),
|
||||
prefiltered_page_number_mask: Some(&page_number_removal_mask),
|
||||
},
|
||||
)
|
||||
};
|
||||
|
||||
@@ -1027,7 +1090,12 @@ pub fn extract_tables_in_regions_mem(
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
let detected = tables::detect_tables(&matched, base_font_size, false);
|
||||
let detected = tables::detect_tables_with_page_width(
|
||||
&matched,
|
||||
base_font_size,
|
||||
false,
|
||||
items.map_or(1.0, |items| tables::content_width(items)),
|
||||
);
|
||||
if let Some(candidate) = detected
|
||||
.iter()
|
||||
.find_map(|t| evaluate(TableCandidateSource::Heuristic, t))
|
||||
@@ -3526,7 +3594,7 @@ fn process_document(
|
||||
doc: Document,
|
||||
page_count: u32,
|
||||
options: PdfOptions,
|
||||
start: std::time::Instant,
|
||||
start: ProcessingTimer,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
// Step 1 — Detection (cheap: scans content streams for text operators)
|
||||
let detection = detector::detect_from_document(&doc, page_count, &options.detection)?;
|
||||
@@ -3542,7 +3610,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3558,7 +3626,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3571,7 +3639,10 @@ fn process_document(
|
||||
// Step 2 — Extraction (reuses the already-loaded document)
|
||||
let extracted = {
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extractor::extract_positioned_text_from_doc(
|
||||
// Most page-filtered requests extract only the selected pages. Gather
|
||||
// other pages only when a selected contextual folio needs cross-page
|
||||
// evidence; failures on those context-only pages are non-fatal.
|
||||
let result = extractor::extract_positioned_text_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3582,9 +3653,19 @@ fn process_document(
|
||||
// This unlocks OCR text layers behind scanned images.
|
||||
if pdf_type == PdfType::Mixed {
|
||||
if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) {
|
||||
let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect();
|
||||
let sample: String = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&item.page))
|
||||
})
|
||||
.take(200)
|
||||
.map(|item| item.text.as_str())
|
||||
.collect();
|
||||
if is_garbage_text(&sample) || sample.trim().is_empty() {
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3594,7 +3675,7 @@ fn process_document(
|
||||
}
|
||||
} else {
|
||||
// Normal extraction failed — try invisible as fallback
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3658,6 +3739,13 @@ fn process_document(
|
||||
let mut garbage_pages: std::collections::HashSet<u32> =
|
||||
std::collections::HashSet::new();
|
||||
for &pg in &ocr_set {
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|filter| !filter.contains(&pg))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let page_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.page == pg)
|
||||
@@ -3697,9 +3785,38 @@ fn process_document(
|
||||
}
|
||||
};
|
||||
|
||||
let selected_page = |page: u32| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&page))
|
||||
};
|
||||
let rects: Vec<_> = rects
|
||||
.into_iter()
|
||||
.filter(|rect| selected_page(rect.page))
|
||||
.collect();
|
||||
let lines: Vec<_> = lines
|
||||
.into_iter()
|
||||
.filter(|line| selected_page(line.page))
|
||||
.collect();
|
||||
let gid_encoded_pages: HashSet<_> = gid_encoded_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
let FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
} = select_items_with_document_folio_context(
|
||||
items,
|
||||
page_count,
|
||||
options.page_filter.as_ref(),
|
||||
);
|
||||
|
||||
let text_quality = analyze_text_quality(&items);
|
||||
merge_ocr_reasons(&mut ocr_reasons_by_page, text_quality.reasons_by_page);
|
||||
let layout = compute_layout_complexity(&items, &rects, &lines);
|
||||
let layout = compute_layout_complexity(&items, &layout_items, &rects, &lines);
|
||||
|
||||
let md = if options.mode == ProcessMode::Analyze {
|
||||
None
|
||||
@@ -3709,9 +3826,14 @@ fn process_document(
|
||||
options.markdown,
|
||||
&rects,
|
||||
&lines,
|
||||
&page_thresholds,
|
||||
struct_roles.as_ref(),
|
||||
&struct_tables,
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: struct_roles.as_ref(),
|
||||
struct_tables: &struct_tables,
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(removal_mask.as_slice()),
|
||||
},
|
||||
))
|
||||
};
|
||||
|
||||
@@ -3824,7 +3946,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: {
|
||||
// Detector reasons (scanned / no_text / vector_text / garbled) merged
|
||||
@@ -5470,9 +5592,50 @@ mod looks_like_partial_table_tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct FolioFilteredItems {
|
||||
items: Vec<types::TextItem>,
|
||||
layout_items: Vec<types::TextItem>,
|
||||
removal_mask: Vec<bool>,
|
||||
removed_pages: HashSet<u32>,
|
||||
}
|
||||
|
||||
/// Resolve folios with complete document context, then select the caller's
|
||||
/// requested pages without losing those decisions.
|
||||
fn select_items_with_document_folio_context(
|
||||
all_items: Vec<types::TextItem>,
|
||||
page_count: u32,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> FolioFilteredItems {
|
||||
let (all_layout_items, all_removed_pages, all_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
let selected_page = |page: u32| page_filter.is_none_or(|filter| filter.contains(&page));
|
||||
|
||||
let (items, removal_mask) = all_items
|
||||
.into_iter()
|
||||
.zip(all_removal_mask)
|
||||
.filter(|(item, _)| selected_page(item.page))
|
||||
.unzip();
|
||||
let layout_items = all_layout_items
|
||||
.into_iter()
|
||||
.filter(|item| selected_page(item.page))
|
||||
.collect();
|
||||
let removed_pages = all_removed_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
|
||||
FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
}
|
||||
}
|
||||
|
||||
/// Analyse extracted items and rects for layout complexity.
|
||||
fn compute_layout_complexity(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
) -> LayoutComplexity {
|
||||
@@ -5494,6 +5657,7 @@ fn compute_layout_complexity(
|
||||
|
||||
// Check for side-by-side layout
|
||||
let owned_items: Vec<types::TextItem> = page_items.iter().map(|i| (*i).clone()).collect();
|
||||
let page_content_width = tables::content_width(&owned_items);
|
||||
let bands = markdown::split_side_by_side(&owned_items);
|
||||
|
||||
let band_ranges: Vec<(f32, f32)> = if bands.is_empty() {
|
||||
@@ -5544,7 +5708,12 @@ fn compute_layout_complexity(
|
||||
break;
|
||||
}
|
||||
// Heuristic fallback for borderless tables
|
||||
let heuristic_tables = tables::detect_tables(&band_items, base_size, false);
|
||||
let heuristic_tables = tables::detect_tables_with_page_width(
|
||||
&band_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
if has_data_table(&heuristic_tables) {
|
||||
found_table = true;
|
||||
break;
|
||||
@@ -5557,7 +5726,7 @@ fn compute_layout_complexity(
|
||||
|
||||
let mut pages_with_columns: Vec<u32> = Vec::new();
|
||||
for page in seen_pages {
|
||||
let cols = extractor::detect_columns(items, page, pages_with_tables.contains(&page));
|
||||
let cols = extractor::detect_columns(column_items, page, pages_with_tables.contains(&page));
|
||||
if cols.len() >= 2 {
|
||||
pages_with_columns.push(page);
|
||||
}
|
||||
@@ -5769,6 +5938,68 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sparse_folios_leave_no_layout_evidence() {
|
||||
let items = vec![
|
||||
test_item("1", 25.0, 20.0, 12.0, 10.0),
|
||||
test_item("2", 520.0, 60.0, 12.0, 10.0),
|
||||
];
|
||||
let (filtered, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(items.clone(), 1);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let filtered = compute_layout_complexity(&items, &filtered, &[], &[]);
|
||||
|
||||
assert!(!filtered.is_complex);
|
||||
assert!(filtered.pages_with_tables.is_empty());
|
||||
assert!(filtered.pages_with_columns.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_selection_keeps_document_wide_folio_layout_decisions() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
for row in 0..8 {
|
||||
let y = 20.0 + row as f32 * 8.0;
|
||||
let mut folio = test_item(&(row * 10 + page).to_string(), 25.0, y, 12.0, 10.0);
|
||||
folio.page = page;
|
||||
let mut footer =
|
||||
test_item(&format!("Footer row {row} summary"), 43.0, y, 470.0, 10.0);
|
||||
footer.page = page;
|
||||
let mut body = test_item(&format!("Body{row}"), 530.0, y, 55.0, 10.0);
|
||||
body.page = page;
|
||||
items.extend([folio, footer, body]);
|
||||
}
|
||||
}
|
||||
|
||||
let page_one_items: Vec<_> = items
|
||||
.iter()
|
||||
.filter(|item| item.page == 1)
|
||||
.cloned()
|
||||
.collect();
|
||||
let (page_local_layout, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(page_one_items.clone(), 4);
|
||||
let page_local = compute_layout_complexity(&page_one_items, &page_local_layout, &[], &[]);
|
||||
assert!(
|
||||
page_local.pages_with_columns.contains(&1),
|
||||
"fixture must reproduce page-local folio column evidence"
|
||||
);
|
||||
|
||||
let selected =
|
||||
select_items_with_document_folio_context(items, 4, Some(&HashSet::from([1])));
|
||||
assert_eq!(
|
||||
selected
|
||||
.removal_mask
|
||||
.iter()
|
||||
.filter(|remove| **remove)
|
||||
.count(),
|
||||
8
|
||||
);
|
||||
let document_wide =
|
||||
compute_layout_complexity(&selected.items, &selected.layout_items, &[], &[]);
|
||||
assert!(!document_wide.pages_with_columns.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_detect_encoding_issues_fffd() {
|
||||
assert!(detect_encoding_issues(
|
||||
|
||||
+165
-28
@@ -975,18 +975,54 @@ pub fn to_markdown_from_items_with_rects(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
) -> String {
|
||||
let document_page_count = items.iter().map(|item| item.page).max().unwrap_or(0);
|
||||
to_markdown_from_items_with_rects_and_page_count(items, options, rects, document_page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to Markdown with an authoritative PDF page count.
|
||||
///
|
||||
/// Use this overload when the owning PDF is available so trailing blank or
|
||||
/// unextracted pages are included in document-level header and folio coverage.
|
||||
/// Item-only callers can continue using [`to_markdown_from_items_with_rects`],
|
||||
/// which falls back to the highest observed item page.
|
||||
pub fn to_markdown_from_items_with_rects_and_page_count(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
document_page_count: u32,
|
||||
) -> String {
|
||||
to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
options,
|
||||
rects,
|
||||
&[],
|
||||
&HashMap::new(),
|
||||
None,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages: None,
|
||||
prefiltered_page_number_mask: None,
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) struct MarkdownDocumentContext<'a> {
|
||||
pub(crate) page_thresholds: &'a HashMap<u32, f32>,
|
||||
pub(crate) struct_roles:
|
||||
Option<&'a HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
pub(crate) struct_tables: &'a [crate::structure_tree::StructTable],
|
||||
pub(crate) page_count: u32,
|
||||
/// Pages where an upstream document-level pass removed folios. This keeps
|
||||
/// table-continuation classification consistent after masked items drop.
|
||||
pub(crate) prefiltered_page_number_pages: Option<&'a HashSet<u32>>,
|
||||
/// Document-level removal decisions aligned with this call's input items.
|
||||
/// Table detection consumes the original items; the mask is applied only
|
||||
/// after table claims have been established.
|
||||
pub(crate) prefiltered_page_number_mask: Option<&'a [bool]>,
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown, using rectangles and line segments for table detection.
|
||||
///
|
||||
/// Line-based detection runs first (strongest structural evidence), then rect-based,
|
||||
@@ -996,27 +1032,43 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
pdf_lines: &[crate::types::PdfLine],
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
struct_roles: Option<&HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
struct_tables: &[crate::structure_tree::StructTable],
|
||||
context: MarkdownDocumentContext<'_>,
|
||||
) -> String {
|
||||
use crate::tables::{
|
||||
detect_tables, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, try_build_rect_guided_table,
|
||||
content_width, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, detect_tables_with_page_width, try_build_rect_guided_table,
|
||||
};
|
||||
use crate::types::ItemType;
|
||||
|
||||
let MarkdownDocumentContext {
|
||||
page_thresholds,
|
||||
struct_roles,
|
||||
struct_tables,
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages,
|
||||
prefiltered_page_number_mask,
|
||||
} = context;
|
||||
|
||||
if items.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Table detection must retain the original collection because short
|
||||
// numeric table cells can be indistinguishable from folios until
|
||||
// structural context is available. A precomputed mask carries the
|
||||
// document-wide decision without removing items before table claims.
|
||||
debug_assert!(prefiltered_page_number_mask.is_none_or(|mask| mask.len() == items.len()));
|
||||
let has_precomputed_page_number_mask = prefiltered_page_number_mask.is_some();
|
||||
let removed_page_number_pages = prefiltered_page_number_pages.cloned().unwrap_or_default();
|
||||
|
||||
// Separate images and links from text items
|
||||
let mut images: Vec<TextItem> = Vec::new();
|
||||
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
let mut links: Vec<TextItem> = Vec::new();
|
||||
let mut text_items: Vec<TextItem> = Vec::new();
|
||||
let mut text_item_page_number_mask: Vec<bool> = Vec::new();
|
||||
|
||||
for item in items {
|
||||
for (input_index, item) in items.into_iter().enumerate() {
|
||||
match &item.item_type {
|
||||
ItemType::Image => {
|
||||
page_image_regions.entry(item.page).or_default().push((
|
||||
@@ -1035,6 +1087,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
ItemType::Text | ItemType::FormField => {
|
||||
text_item_page_number_mask.push(
|
||||
prefiltered_page_number_mask
|
||||
.and_then(|mask| mask.get(input_index))
|
||||
.copied()
|
||||
.unwrap_or(false),
|
||||
);
|
||||
text_items.push(item);
|
||||
}
|
||||
}
|
||||
@@ -1075,7 +1133,6 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
let mut pages: Vec<u32> = page_groups.keys().copied().collect();
|
||||
pages.sort();
|
||||
let page_count = pages.last().copied().unwrap_or(0) + 1;
|
||||
|
||||
// Track band splits per page so we can split non-table items later
|
||||
let mut page_band_splits: HashMap<u32, Vec<(f32, f32)>> = HashMap::new();
|
||||
@@ -1087,6 +1144,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
for page in pages {
|
||||
let group = page_groups.get(&page).unwrap();
|
||||
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
|
||||
let page_content_width = content_width(&page_items);
|
||||
|
||||
// Chart-bar regions: bar charts drawn as filled rects read as cell
|
||||
// rects or aligned text and get gridded into phantom tables. Their
|
||||
@@ -1128,7 +1186,10 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
});
|
||||
let chart_prose_columns = chart_prose_split.is_some();
|
||||
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
// Check for side-by-side table layout using the original items. Sparse
|
||||
// numeric cells need table context before they can be distinguished
|
||||
// safely from folios; cleaned evidence is reserved for column and
|
||||
// final non-table layout decisions.
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// A rect table crossing a proposed split boundary means the "gutter"
|
||||
// is really the gap between ruled and borderless table columns —
|
||||
@@ -1368,7 +1429,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// table can share the prose anchors. Reject only candidates
|
||||
// whose cells prove they are parallel prose fragments.
|
||||
let reject_parallel_prose = chart_prose_columns && !was_split;
|
||||
let tables = detect_tables(subset_items, base_size, false);
|
||||
let tables = detect_tables_with_page_width(
|
||||
subset_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
for table in tables {
|
||||
if reject_parallel_prose && is_parallel_prose_table(&table) {
|
||||
log::debug!(
|
||||
@@ -1549,7 +1615,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// and reject chart-page prose candidates individually below.
|
||||
let skip_body_font =
|
||||
merged_retry_skips_body_font(detected_columns, !chart_regions.is_empty());
|
||||
let heuristic_tables = detect_tables(&chart_free, base_size, skip_body_font);
|
||||
let heuristic_tables = detect_tables_with_page_width(
|
||||
&chart_free,
|
||||
base_size,
|
||||
skip_body_font,
|
||||
page_content_width,
|
||||
);
|
||||
for table in &heuristic_tables {
|
||||
if !chart_regions.is_empty() && is_parallel_prose_table(table) {
|
||||
log::debug!(
|
||||
@@ -1634,16 +1705,20 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
};
|
||||
|
||||
// Filter out table items and process the rest
|
||||
let non_table_items: Vec<TextItem> = text_items
|
||||
let non_table_items: Vec<(usize, TextItem)> = text_items
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.filter(|(idx, _)| !table_items.contains(idx))
|
||||
.map(|(_, item)| item)
|
||||
.collect();
|
||||
|
||||
// Find pages that are table-only (no remaining non-table text)
|
||||
let table_only_pages: HashSet<u32> = {
|
||||
let pages_with_text: HashSet<u32> = non_table_items.iter().map(|i| i.page).collect();
|
||||
let mut pages_with_text: HashSet<u32> =
|
||||
non_table_items.iter().map(|(_, item)| item.page).collect();
|
||||
// Preserve the pre-filter continuation classification: a page that
|
||||
// originally also contained a folio does not become table-only merely
|
||||
// because an upstream document-level pass removed it.
|
||||
pages_with_text.extend(removed_page_number_pages);
|
||||
page_tables
|
||||
.keys()
|
||||
.filter(|p| !pages_with_text.contains(p))
|
||||
@@ -1658,11 +1733,25 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// column detection on pages where table column gaps would be misidentified.
|
||||
let table_page_set: HashSet<u32> = page_tables.keys().copied().collect();
|
||||
|
||||
let non_table_items = if has_precomputed_page_number_mask {
|
||||
non_table_items
|
||||
.into_iter()
|
||||
.filter(|(index, _)| !text_item_page_number_mask[*index])
|
||||
.map(|(_, item)| item)
|
||||
.collect()
|
||||
} else {
|
||||
crate::extractor::filter_markdown_page_numbers_with_removed_pages(
|
||||
non_table_items.into_iter().map(|(_, item)| item).collect(),
|
||||
document_page_count,
|
||||
)
|
||||
.0
|
||||
};
|
||||
|
||||
// Split non-table items by band boundaries before line grouping so that
|
||||
// items from different side-by-side zones (e.g. left/right month columns
|
||||
// in a calendar) don't merge into the same line.
|
||||
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
|
||||
crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
non_table_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1690,13 +1779,14 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
// Process unsplit pages normally
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
let mut all_lines =
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
// Process each split page's bands independently, then interleave
|
||||
// by Y position so paired zones (e.g. left/right months) appear together.
|
||||
let mut split_pages: Vec<u32> = split_page_items.keys().copied().collect();
|
||||
@@ -1714,7 +1804,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !band_items.is_empty() {
|
||||
page_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
band_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1748,7 +1838,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !column_items.is_empty() {
|
||||
zone_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
column_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1789,7 +1879,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
item.y >= low || item_is_in_chart_region(item, chart_regions)
|
||||
});
|
||||
all_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
chart_zone,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1808,7 +1898,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Strip repeated headers/footers before conversion
|
||||
let lines = if options.strip_headers_footers {
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
preprocess::strip_repeated_lines(lines, document_page_count)
|
||||
} else {
|
||||
lines
|
||||
};
|
||||
@@ -1921,6 +2011,53 @@ mod tests {
|
||||
it
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn precomputed_folio_mask_preserves_numeric_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for column in 0..2 {
|
||||
let mut item = make_item_w(
|
||||
110.0 + column as f32 * 100.0,
|
||||
30.0 + row as f32 * 20.0,
|
||||
20.0,
|
||||
1,
|
||||
);
|
||||
item.text = (row * 2 + column + 1).to_string();
|
||||
items.push(item);
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + column as f32 * 100.0,
|
||||
y: 20.0 + row as f32 * 20.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Simulate document-level folio decisions that would remove every
|
||||
// short numeric item if applied before structural table detection.
|
||||
let removal_mask = vec![true; items.len()];
|
||||
let removed_pages = HashSet::from([1]);
|
||||
let markdown = to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
MarkdownOptions::default(),
|
||||
&rects,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: 1,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(&removal_mask),
|
||||
},
|
||||
);
|
||||
|
||||
assert!(markdown.contains("|1|2|"), "{markdown}");
|
||||
assert!(markdown.contains("|7|8|"), "{markdown}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn early_layout_excludes_chart_items_before_column_detection() {
|
||||
let mut items = Vec::new();
|
||||
|
||||
+20
-64
@@ -3,6 +3,7 @@
|
||||
use regex::Regex;
|
||||
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
use crate::text_utils::is_page_number_line;
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
@@ -145,7 +146,7 @@ fn fix_hyphenation(text: &str) -> String {
|
||||
result
|
||||
}
|
||||
|
||||
/// Remove standalone page numbers (lines that are just 1-4 digit numbers)
|
||||
/// Remove isolated page-number expressions from Markdown.
|
||||
fn remove_page_numbers(text: &str) -> String {
|
||||
let mut result = Vec::new();
|
||||
let lines: Vec<&str> = text.lines().collect();
|
||||
@@ -183,69 +184,6 @@ fn remove_page_numbers(text: &str) -> String {
|
||||
result.join("\n")
|
||||
}
|
||||
|
||||
/// Check if a line looks like a page number
|
||||
fn is_page_number_line(trimmed: &str) -> bool {
|
||||
// Empty lines are not page numbers
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Pattern 1: Just a number (1-4 digits)
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Pattern 2: "Page X of Y" or "Page X" or "Page of" (placeholder)
|
||||
let lower = trimmed.to_lowercase();
|
||||
if let Some(rest) = lower.strip_prefix("page") {
|
||||
let rest = rest.trim();
|
||||
// "Page of" (empty page numbers)
|
||||
if rest == "of" || rest.starts_with("of ") {
|
||||
return true;
|
||||
}
|
||||
// "Page X" or "Page X of Y"
|
||||
if rest
|
||||
.chars()
|
||||
.next()
|
||||
.map(|c| c.is_ascii_digit())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Just "Page" followed by whitespace and maybe "of"
|
||||
if rest.is_empty()
|
||||
|| rest
|
||||
.split_whitespace()
|
||||
.all(|w| w == "of" || w.chars().all(|c| c.is_ascii_digit()))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 3: "X of Y" where X and Y are numbers
|
||||
if let Some(of_idx) = trimmed.find(" of ") {
|
||||
let before = trimmed[..of_idx].trim();
|
||||
let after = trimmed[of_idx + 4..].trim();
|
||||
if before.chars().all(|c| c.is_ascii_digit())
|
||||
&& after.chars().all(|c| c.is_ascii_digit())
|
||||
&& !before.is_empty()
|
||||
&& !after.is_empty()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 4: "- X -" centered page number
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if inner.chars().all(|c| c.is_ascii_digit()) && !inner.is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Convert URLs to markdown links
|
||||
fn format_urls(text: &str) -> String {
|
||||
use once_cell::sync::Lazy;
|
||||
@@ -489,12 +427,14 @@ mod tests {
|
||||
fn test_is_page_number_page_x() {
|
||||
assert!(is_page_number_line("Page 5"));
|
||||
assert!(is_page_number_line("page 12"));
|
||||
assert!(is_page_number_line("Page123"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_page_x_of_y() {
|
||||
assert!(is_page_number_line("Page 3 of 10"));
|
||||
assert!(is_page_number_line("page 1 of 5"));
|
||||
assert!(is_page_number_line("Page 3 of 10 Report header"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -526,6 +466,12 @@ mod tests {
|
||||
assert!(!is_page_number_line("Total: 500"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_labeled_running_header() {
|
||||
assert!(is_page_number_line("Page 42 Chapter 5"));
|
||||
assert!(is_page_number_line("Page 42 explains the result"));
|
||||
}
|
||||
|
||||
// --- remove_page_numbers ---
|
||||
|
||||
#[test]
|
||||
@@ -551,6 +497,16 @@ mod tests {
|
||||
assert!(result.contains("42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_labeled_header_with_content() {
|
||||
let input = "Content\n\nPage 42 explains the result\n---\nEnd";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert!(!result.contains("Page 42 explains the result"));
|
||||
assert!(result.contains("Content"));
|
||||
assert!(result.contains("End"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_multiple_patterns() {
|
||||
let input = "\n1\n\nContent\n\n2\n\n---\nMore\n\n3\n";
|
||||
|
||||
@@ -18,7 +18,15 @@ use super::{Table, TableDetectionMode};
|
||||
///
|
||||
/// Returns `(merged_items, index_map)` where `index_map[merged_idx]` contains
|
||||
/// the original item indices that were merged into that item.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Vec<usize>>) {
|
||||
merge_adjacent_items_preserving(items, &std::collections::HashSet::new())
|
||||
}
|
||||
|
||||
fn merge_adjacent_items_preserving(
|
||||
items: &[TextItem],
|
||||
preserved_indices: &std::collections::HashSet<usize>,
|
||||
) -> (Vec<TextItem>, Vec<Vec<usize>>) {
|
||||
if items.is_empty() {
|
||||
return (vec![], vec![]);
|
||||
}
|
||||
@@ -72,6 +80,23 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
break;
|
||||
}
|
||||
|
||||
// Proven replacement cells must retain their own decoration.
|
||||
// Otherwise an adjacent old/new pair inherits only the first
|
||||
// fragment's flags and can lose the live table evidence.
|
||||
let decoration_changes = indices.iter().any(|index| {
|
||||
let merged_item = &items[*index];
|
||||
next_item.is_underline != merged_item.is_underline
|
||||
|| next_item.is_strikeout != merged_item.is_strikeout
|
||||
});
|
||||
if decoration_changes
|
||||
&& (indices
|
||||
.iter()
|
||||
.any(|index| preserved_indices.contains(index))
|
||||
|| preserved_indices.contains(&next_idx))
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
let gap = next_item.x - end_x;
|
||||
// Stop if gap exceeds threshold (inter-column gap)
|
||||
if gap > x_gap_max {
|
||||
@@ -137,17 +162,319 @@ fn expand_consolidated_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<usize>)
|
||||
(expanded, index_map)
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct RedlineEditRegion {
|
||||
x_ranges: Vec<(f32, f32)>,
|
||||
y_min: f32,
|
||||
y_max: f32,
|
||||
spans_page_width: bool,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct UnderlinedTableColumn {
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
}
|
||||
|
||||
pub(crate) fn content_width(items: &[TextItem]) -> f32 {
|
||||
let x_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + item.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
(x_max - x_min).max(1.0)
|
||||
}
|
||||
|
||||
/// Spatial regions where multiple strikeout rows indicate a redline edit block.
|
||||
///
|
||||
/// A lone deletion can occur inside or beside a real table, so it must not
|
||||
/// globally suppress underlined table cells. Closely spaced strikeout rows are
|
||||
/// different: together with nearby underlines they form the overlapping
|
||||
/// old/new text layers used by legislative redlines, and those decorations
|
||||
/// must not become heuristic column evidence.
|
||||
fn redline_edit_regions(items: &[TextItem], page_width: f32) -> Vec<RedlineEditRegion> {
|
||||
const ROW_DEDUP_TOLERANCE: f32 = 8.0;
|
||||
const MAX_CLUSTER_GAP: f32 = 64.0;
|
||||
const Y_PADDING: f32 = 36.0;
|
||||
const X_PADDING: f32 = 12.0;
|
||||
const PAGE_WIDTH_RATIO: f32 = 0.35;
|
||||
const MAX_HORIZONTAL_GAP_RATIO: f32 = 0.20;
|
||||
|
||||
let mut strikeouts: Vec<&TextItem> = items.iter().filter(|item| item.is_strikeout).collect();
|
||||
strikeouts.sort_by(|a, b| a.y.total_cmp(&b.y));
|
||||
|
||||
let mut rows: Vec<(f32, Vec<(f32, f32)>)> = Vec::new();
|
||||
for item in strikeouts {
|
||||
if let Some((_, x_ranges)) = rows
|
||||
.last_mut()
|
||||
.filter(|(y, _)| (item.y - *y).abs() <= ROW_DEDUP_TOLERANCE)
|
||||
{
|
||||
x_ranges.push((item.x, item.x + item.width));
|
||||
} else {
|
||||
rows.push((item.y, vec![(item.x, item.x + item.width)]));
|
||||
}
|
||||
}
|
||||
|
||||
let mut regions = Vec::new();
|
||||
let mut start = 0;
|
||||
while start < rows.len() {
|
||||
let mut end = start + 1;
|
||||
while end < rows.len() && rows[end].0 - rows[end - 1].0 <= MAX_CLUSTER_GAP {
|
||||
end += 1;
|
||||
}
|
||||
if end - start >= 2 {
|
||||
let mut x_ranges: Vec<(f32, f32)> = rows[start..end]
|
||||
.iter()
|
||||
.flat_map(|(_, x_ranges)| x_ranges.iter().copied())
|
||||
.collect();
|
||||
x_ranges.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
let mut merged_ranges: Vec<(f32, f32)> = Vec::new();
|
||||
for (x_min, x_max) in x_ranges {
|
||||
// Modest inline gaps can separate fragments of one flowing
|
||||
// edit; column-scale gaps must remain distinct spatial masks.
|
||||
if let Some((_, merged_max)) = merged_ranges.last_mut().filter(|(_, merged_max)| {
|
||||
x_min - *merged_max <= page_width * MAX_HORIZONTAL_GAP_RATIO
|
||||
}) {
|
||||
*merged_max = merged_max.max(x_max);
|
||||
} else {
|
||||
merged_ranges.push((x_min, x_max));
|
||||
}
|
||||
}
|
||||
let covered_width: f32 = merged_ranges
|
||||
.iter()
|
||||
.map(|(x_min, x_max)| x_max - x_min)
|
||||
.sum();
|
||||
for (x_min, x_max) in &mut merged_ranges {
|
||||
*x_min -= X_PADDING;
|
||||
*x_max += X_PADDING;
|
||||
}
|
||||
// Redlines spread across much of the text width are flowing prose,
|
||||
// so their whole Y-band is ambiguous. Compact edits can be scoped
|
||||
// to their actual horizontal spans without hiding content between
|
||||
// unrelated edits in separate columns.
|
||||
regions.push(RedlineEditRegion {
|
||||
x_ranges: merged_ranges,
|
||||
y_min: rows[start].0 - Y_PADDING,
|
||||
y_max: rows[end - 1].0 + Y_PADDING,
|
||||
spans_page_width: covered_width >= page_width * PAGE_WIDTH_RATIO,
|
||||
});
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
// Padding can make neighboring clusters overlap. Partition that overlap at
|
||||
// its midpoint so each Y position maps to one region without combining
|
||||
// horizontally unrelated edits.
|
||||
for index in 1..regions.len() {
|
||||
let (previous, current) = regions.split_at_mut(index);
|
||||
let previous = &mut previous[index - 1];
|
||||
let current = &mut current[0];
|
||||
if previous.y_max >= current.y_min {
|
||||
let boundary = (previous.y_max + current.y_min) / 2.0;
|
||||
previous.y_max = boundary;
|
||||
current.y_min = boundary;
|
||||
}
|
||||
}
|
||||
regions
|
||||
}
|
||||
|
||||
fn overlaps_redline_x(item: &TextItem, region: &RedlineEditRegion) -> bool {
|
||||
let range_index = region
|
||||
.x_ranges
|
||||
.partition_point(|(_, x_max)| *x_max < item.x);
|
||||
region
|
||||
.x_ranges
|
||||
.get(range_index)
|
||||
.is_some_and(|(x_min, _)| item.x + item.width >= *x_min)
|
||||
}
|
||||
|
||||
fn has_distinct_rows(
|
||||
items: &[&TextItem],
|
||||
underlined_only: bool,
|
||||
required: usize,
|
||||
tolerance: f32,
|
||||
) -> bool {
|
||||
debug_assert!(required <= 3);
|
||||
let mut rows = [0.0; 3];
|
||||
let mut row_count = 0;
|
||||
for item in items {
|
||||
if underlined_only && !item.is_underline {
|
||||
continue;
|
||||
}
|
||||
if rows[..row_count]
|
||||
.iter()
|
||||
.all(|row| (item.y - row).abs() > tolerance)
|
||||
{
|
||||
rows[row_count] = item.y;
|
||||
row_count += 1;
|
||||
if row_count == required {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Aligned live items seeded by underline evidence form revised table columns.
|
||||
/// Compact edits accept one replacement backed by surrounding live rows; wide
|
||||
/// prose-like edits require replacements on at least two distinct rows.
|
||||
fn underlined_table_columns(
|
||||
items: &[TextItem],
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
) -> Vec<Vec<UnderlinedTableColumn>> {
|
||||
const X_ALIGNMENT_TOLERANCE: f32 = 4.0;
|
||||
const ROW_DEDUP_TOLERANCE: f32 = 8.0;
|
||||
|
||||
// Sort once globally by X, then partition candidates into their unique Y
|
||||
// regions. Each regional vector remains X-sorted without another sort.
|
||||
let mut live_items: Vec<&TextItem> = items.iter().filter(|item| !item.is_strikeout).collect();
|
||||
live_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let mut candidates_by_region: Vec<Vec<&TextItem>> = vec![Vec::new(); redline_regions.len()];
|
||||
for item in live_items {
|
||||
if let Some(region_index) = redline_region_at_y(redline_regions, item.y) {
|
||||
if overlaps_redline_x(item, &redline_regions[region_index]) {
|
||||
candidates_by_region[region_index].push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let mut columns_by_region = Vec::with_capacity(redline_regions.len());
|
||||
for (region, candidates) in redline_regions.iter().zip(candidates_by_region) {
|
||||
let mut columns: Vec<UnderlinedTableColumn> = Vec::new();
|
||||
let mut start = 0;
|
||||
while start < candidates.len() {
|
||||
let mut end = start + 1;
|
||||
while end < candidates.len()
|
||||
&& candidates[end].x - candidates[start].x <= X_ALIGNMENT_TOLERANCE
|
||||
{
|
||||
end += 1;
|
||||
}
|
||||
let aligned_items = &candidates[start..end];
|
||||
let enough_live_rows = has_distinct_rows(aligned_items, false, 3, ROW_DEDUP_TOLERANCE);
|
||||
let required_underlined_rows = if region.spans_page_width { 2 } else { 1 };
|
||||
let enough_underlined_rows = has_distinct_rows(
|
||||
aligned_items,
|
||||
true,
|
||||
required_underlined_rows,
|
||||
ROW_DEDUP_TOLERANCE,
|
||||
);
|
||||
if enough_live_rows && enough_underlined_rows {
|
||||
let x_min = candidates[start].x - X_ALIGNMENT_TOLERANCE;
|
||||
let x_max = candidates[end - 1].x + X_ALIGNMENT_TOLERANCE;
|
||||
if let Some(column) = columns.last_mut().filter(|column| column.x_max >= x_min) {
|
||||
column.x_max = column.x_max.max(x_max);
|
||||
} else {
|
||||
columns.push(UnderlinedTableColumn { x_min, x_max });
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
columns_by_region.push(columns);
|
||||
}
|
||||
columns_by_region
|
||||
}
|
||||
|
||||
fn redline_region_at_y(redline_regions: &[RedlineEditRegion], y: f32) -> Option<usize> {
|
||||
let region_index = redline_regions.partition_point(|region| region.y_max < y);
|
||||
redline_regions
|
||||
.get(region_index)
|
||||
.filter(|region| y >= region.y_min)
|
||||
.map(|_| region_index)
|
||||
}
|
||||
|
||||
fn is_heuristic_table_evidence(
|
||||
item: &TextItem,
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
underlined_table_columns: &[Vec<UnderlinedTableColumn>],
|
||||
) -> bool {
|
||||
if item.is_strikeout {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(region_index) = redline_region_at_y(redline_regions, item.y) else {
|
||||
return true;
|
||||
};
|
||||
let region = &redline_regions[region_index];
|
||||
let columns = &underlined_table_columns[region_index];
|
||||
if region.spans_page_width && columns.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlaps_x = overlaps_redline_x(item, region);
|
||||
!overlaps_x || is_revised_table_cell(item, columns)
|
||||
}
|
||||
|
||||
fn is_revised_table_cell(
|
||||
item: &TextItem,
|
||||
underlined_table_columns: &[UnderlinedTableColumn],
|
||||
) -> bool {
|
||||
let column_index = underlined_table_columns.partition_point(|column| column.x_max < item.x);
|
||||
underlined_table_columns
|
||||
.get(column_index)
|
||||
.is_some_and(|column| item.x >= column.x_min)
|
||||
}
|
||||
|
||||
fn revised_table_cell_indices(
|
||||
items: &[TextItem],
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
underlined_table_columns: &[Vec<UnderlinedTableColumn>],
|
||||
) -> std::collections::HashSet<usize> {
|
||||
items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(item_index, item)| {
|
||||
let region_index = redline_region_at_y(redline_regions, item.y)?;
|
||||
let region = &redline_regions[region_index];
|
||||
(item.is_underline
|
||||
&& overlaps_redline_x(item, region)
|
||||
&& is_revised_table_cell(item, &underlined_table_columns[region_index]))
|
||||
.then_some(item_index)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Detect tables in a set of text items from a single page
|
||||
pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bool) -> Vec<Table> {
|
||||
detect_tables_with_page_width(items, base_font_size, skip_body_font, content_width(items))
|
||||
}
|
||||
|
||||
/// Detect tables in a subset while using the full page's text width for
|
||||
/// page-spanning redline classification.
|
||||
pub(crate) fn detect_tables_with_page_width(
|
||||
items: &[TextItem],
|
||||
base_font_size: f32,
|
||||
skip_body_font: bool,
|
||||
page_width: f32,
|
||||
) -> Vec<Table> {
|
||||
if items.len() < 6 {
|
||||
return vec![];
|
||||
}
|
||||
// Compute these before consolidation: adjacent old/new text can merge and
|
||||
// inherit only the first fragment's decoration flags.
|
||||
let redline_regions = redline_edit_regions(items, page_width);
|
||||
let underlined_table_columns = underlined_table_columns(items, &redline_regions);
|
||||
let source_evidence: Vec<bool> = items
|
||||
.iter()
|
||||
.map(|item| is_heuristic_table_evidence(item, &redline_regions, &underlined_table_columns))
|
||||
.collect();
|
||||
let revised_table_cells =
|
||||
revised_table_cell_indices(items, &redline_regions, &underlined_table_columns);
|
||||
|
||||
// Step 1: Merge adjacent single-char items into words (handles per-character PDFs)
|
||||
let (merged_items, merge_map) = merge_adjacent_items(items);
|
||||
let (merged_items, merge_map) = merge_adjacent_items_preserving(items, &revised_table_cells);
|
||||
|
||||
// Step 2: Expand consolidated financial items (e.g. "$ 1,234 $ 5,678" → sub-items)
|
||||
let (expanded_items, expand_map) = expand_consolidated_items(&merged_items);
|
||||
let expanded_evidence: Vec<bool> = expand_map
|
||||
.iter()
|
||||
.map(|&merged_index| {
|
||||
merge_map[merged_index]
|
||||
.iter()
|
||||
.all(|&source_index| source_evidence[source_index])
|
||||
})
|
||||
.collect();
|
||||
let items = &expanded_items[..]; // shadow parameter — all detection uses processed items
|
||||
|
||||
let mut tables = Vec::new();
|
||||
@@ -159,7 +486,11 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
let table_candidates: Vec<(usize, &TextItem)> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| item.font_size <= table_font_threshold && item.font_size >= 6.0)
|
||||
.filter(|(index, item)| {
|
||||
expanded_evidence[*index]
|
||||
&& item.font_size <= table_font_threshold
|
||||
&& item.font_size >= 6.0
|
||||
})
|
||||
.collect();
|
||||
|
||||
if table_candidates.len() >= 6 {
|
||||
@@ -208,6 +539,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
.enumerate()
|
||||
.filter(|(idx, item)| {
|
||||
!claimed_indices.contains(idx)
|
||||
&& expanded_evidence[*idx]
|
||||
&& item.font_size >= body_font_low
|
||||
&& item.font_size <= body_font_high
|
||||
&& item.font_size >= 6.0
|
||||
@@ -1646,6 +1978,362 @@ fn try_add_label_column(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn body_item(text: &str, x: f32, y: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y,
|
||||
width: 90.0,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_preserves_redline_boundaries() {
|
||||
let old = body_item("old value", 220.0, 700.0, true);
|
||||
let mut replacement = body_item("new value", 312.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let preserved_indices = std::collections::HashSet::from([1]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[old, replacement], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0], vec![1]]);
|
||||
assert!(merged[0].is_strikeout);
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_keeps_boundary_after_preserved_fragment() {
|
||||
let mut prefix = body_item("prefix", 20.0, 700.0, false);
|
||||
prefix.is_underline = true;
|
||||
let mut replacement = body_item("replacement", 111.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let deleted = body_item("deleted", 202.0, 700.0, true);
|
||||
let preserved_indices = std::collections::HashSet::from([1]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[prefix, replacement, deleted], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0, 1], vec![2]]);
|
||||
assert!(merged[0].is_underline);
|
||||
assert!(merged[1].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_keeps_preserved_fragment_out_of_mixed_run() {
|
||||
let mut prefix = body_item("prefix", 20.0, 700.0, false);
|
||||
prefix.is_underline = true;
|
||||
let deleted = body_item("deleted", 111.0, 700.0, true);
|
||||
let mut replacement = body_item("replacement", 202.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let preserved_indices = std::collections::HashSet::from([2]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[prefix, deleted, replacement], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0, 1], vec![2]]);
|
||||
assert!(merged[0].is_underline);
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn body_font_redline_deletions_do_not_create_a_table() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("live paragraph text", 50.0, y, false));
|
||||
items.push(body_item("deleted wording", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"aligned strikeout overlays are source edits, not table columns"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn underlined_body_font_table_without_deletions_is_still_detected() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"underline-only tables must keep their heuristic evidence"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn body_font_table_with_one_deletion_is_still_detected() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("row value", 220.0, y, row == 0));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"one revised cell must not suppress an otherwise complete table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unrelated_strikeout_does_not_remove_underlined_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
items.push(body_item("deleted prose", 50.0, 300.0, true));
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"a distant deletion must not suppress underlined table columns"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nearby_redline_rows_do_not_remove_plain_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("row value", 220.0, y, false));
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("deleted prose", 400.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"redline rows must suppress decorations, not nearby live table cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separate_strikeout_columns_do_not_suppress_content_between_them() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 160.0, y, false));
|
||||
items.push(body_item("row value", 280.0, y, false));
|
||||
}
|
||||
for (row, y) in [700.0, 684.0, 668.0, 652.0].into_iter().enumerate() {
|
||||
let x = if row % 2 == 0 { 50.0 } else { 420.0 };
|
||||
let mut deletion = body_item("old", x, y, true);
|
||||
deletion.width = 30.0;
|
||||
items.push(deletion);
|
||||
}
|
||||
|
||||
let regions = redline_edit_regions(&items, content_width(&items));
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert_eq!(regions[0].x_ranges.len(), 2);
|
||||
assert!(!regions[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"separate edit columns must not create a suppression bridge across the page"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_redline_rows_do_not_turn_fragmented_prose_into_a_table() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("line number", 50.0, y, false));
|
||||
items.push(body_item("live prose fragment", 160.0, y, false));
|
||||
let strikeout_x = if row % 2 == 0 { 280.0 } else { 430.0 };
|
||||
items.push(body_item("deleted prose", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"full-width redline prose must not retain table-shaped fragments"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_redline_prose_ignores_underlines_outside_edit_span() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
let mut line_number = body_item("line number", 50.0, y, false);
|
||||
line_number.is_underline = row < 2;
|
||||
items.push(line_number);
|
||||
items.push(body_item("live prose fragment", 160.0, y, false));
|
||||
let strikeout_x = if row % 2 == 0 { 280.0 } else { 430.0 };
|
||||
items.push(body_item("deleted prose", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"underlines outside the edit span must not disable the wide-prose veto"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nearby_redline_rows_do_not_remove_separate_underlined_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("deleted prose", 400.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"redline rows must not suppress a horizontally separate underlined table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_revised_rows_do_not_remove_underlined_table_column() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let replacement_x = 220.0 + (row % 2) as f32 * 2.0;
|
||||
let mut value = body_item("new value", replacement_x, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"aligned replacement cells must preserve a partially revised table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn single_replacement_uses_aligned_live_table_rows() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = row == 0;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"one replacement cell must retain its aligned live table column"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_revised_table_keeps_repeated_replacement_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = row < 2;
|
||||
items.push(value);
|
||||
let strikeout_x = if row % 2 == 0 { 220.0 } else { 400.0 };
|
||||
items.push(body_item("old value", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"wide edits must keep a table column with repeated replacements"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn revised_financial_columns_survive_item_expansion() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..4 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("old value", 180.0, y, true));
|
||||
let mut values = body_item("$ 100 $ 200 $ 300", 180.0, y, false);
|
||||
values.width = 300.0;
|
||||
values.is_underline = true;
|
||||
items.push(values);
|
||||
}
|
||||
|
||||
let tables = detect_tables(&items, 12.0, false);
|
||||
assert!(
|
||||
tables.iter().any(|table| table.columns.len() >= 4),
|
||||
"all expanded financial columns must inherit their source evidence"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_layout_band_uses_full_page_width_for_redline_scope() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 190.0, y, false));
|
||||
items.push(body_item("old value", 300.0, y, true));
|
||||
let mut replacement = body_item("new value", 300.0, y, false);
|
||||
replacement.is_underline = true;
|
||||
items.push(replacement);
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(!redline_edit_regions(&items, 500.0)[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables_with_page_width(&items, 12.0, false, 500.0).is_empty(),
|
||||
"a narrow band must use full-page context to preserve revised table cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adjacent_revised_fragments_preserve_live_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..4 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
let mut replacement = body_item("new value", 312.0, y, false);
|
||||
replacement.is_underline = true;
|
||||
items.push(replacement);
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"adjacent old/new fragments must retain the live revised cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_rejects_toc() {
|
||||
|
||||
+27
-1
@@ -436,7 +436,8 @@ pub(crate) fn recover_header_row(
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
item.font_size > small_font_threshold
|
||||
!item.is_strikeout
|
||||
&& item.font_size > small_font_threshold
|
||||
&& item.y > first_row_y
|
||||
&& item.y <= first_row_y + row_gap_limit
|
||||
})
|
||||
@@ -804,6 +805,31 @@ mod tests {
|
||||
assert_eq!(table.rows.len(), rows_before);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_skips_strikeout_candidates() {
|
||||
let mut old_col1 = make_item("Old Col1", 100.0, 520.0, 12.0);
|
||||
old_col1.is_strikeout = true;
|
||||
let mut old_col2 = make_item("Old Col2", 200.0, 520.0, 12.0);
|
||||
old_col2.is_strikeout = true;
|
||||
let all_items = vec![
|
||||
old_col1,
|
||||
old_col2,
|
||||
make_item("A", 100.0, 500.0, 8.0),
|
||||
make_item("B", 200.0, 500.0, 8.0),
|
||||
];
|
||||
let mut table = Table {
|
||||
columns: vec![100.0, 200.0],
|
||||
rows: vec![500.0, 480.0],
|
||||
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||
item_indices: vec![2, 3],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
recover_header_row(&mut table, &all_items, 9.0);
|
||||
assert_eq!(table.rows.len(), 2);
|
||||
assert_eq!(table.cells[0], vec!["A", "B"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_too_far_above() {
|
||||
let all_items = vec![
|
||||
|
||||
+3
-1
@@ -12,7 +12,9 @@ mod grid;
|
||||
pub mod structured;
|
||||
|
||||
pub use detect_heuristic::detect_tables;
|
||||
pub(crate) use detect_heuristic::is_table_of_contents;
|
||||
pub(crate) use detect_heuristic::{
|
||||
content_width, detect_tables_with_page_width, is_table_of_contents,
|
||||
};
|
||||
pub use detect_lines::detect_tables_from_lines;
|
||||
pub(crate) use detect_lines::detect_vector_grid_tables_from_lines;
|
||||
pub(crate) use detect_rects::cluster_rects;
|
||||
|
||||
@@ -7,6 +7,76 @@
|
||||
use crate::types::TextItem;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
/// Return whether text is an explicit page-number expression.
|
||||
///
|
||||
/// This strict form is suitable before layout, where removing one numeric item
|
||||
/// from substantive text such as `Page 42 explains the result` would lose data.
|
||||
pub(crate) fn is_explicit_page_number_expression(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let is_number = |value: &str| {
|
||||
!value.is_empty() && value.chars().all(|character| character.is_ascii_digit())
|
||||
};
|
||||
|
||||
if trimmed.len() <= 4 && is_number(trimmed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if is_number(inner) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
let lowercase = trimmed.to_ascii_lowercase();
|
||||
if let Some(rest) = lowercase.strip_prefix("page") {
|
||||
let words: Vec<&str> = rest.split_whitespace().collect();
|
||||
if words.len() >= 3 && is_number(words[0]) && words[1] == "of" && is_number(words[2]) {
|
||||
return true;
|
||||
}
|
||||
if words.len() >= 2 && words[0] == "of" && is_number(words[1]) {
|
||||
return true;
|
||||
}
|
||||
return match words.as_slice() {
|
||||
[] | ["of"] => true,
|
||||
[number] => is_number(number),
|
||||
["of", total] => is_number(total),
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
};
|
||||
}
|
||||
|
||||
let words: Vec<&str> = lowercase.split_whitespace().collect();
|
||||
match words.as_slice() {
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return whether a completed Markdown line looks like a page number or a
|
||||
/// labeled running header.
|
||||
///
|
||||
/// At this stage the complete line and surrounding breaks are available, so a
|
||||
/// leading `Page N` remains compatible with the existing header cleanup even
|
||||
/// when the PDF appends a chapter or document title.
|
||||
pub(crate) fn is_page_number_line(text: &str) -> bool {
|
||||
if is_explicit_page_number_expression(text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let lowercase = text.trim().to_ascii_lowercase();
|
||||
lowercase.strip_prefix("page").is_some_and(|rest| {
|
||||
rest.trim_start()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
/// Check if a character is CJK (Chinese, Japanese, Korean).
|
||||
/// CJK languages don't use spaces between words, so word-boundary
|
||||
/// heuristics should not apply when CJK characters are involved.
|
||||
|
||||
+23
-10
@@ -4,11 +4,17 @@
|
||||
|
||||
use log::{debug, warn};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::borrow::Cow;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::glyph_names::glyph_to_char;
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
static BUILTIN_CMAPS: include_dir::Dir<'_> =
|
||||
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
|
||||
|
||||
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ToUnicodeCMap {
|
||||
@@ -1149,9 +1155,7 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
|
||||
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
|
||||
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(name);
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&name)?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -1159,13 +1163,14 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
cmap.code_byte_length = 2;
|
||||
debug!(
|
||||
"Built-in CMap {}: char_map={} ranges={}",
|
||||
path.display(),
|
||||
name,
|
||||
cmap.char_map.len(),
|
||||
cmap.ranges.len()
|
||||
);
|
||||
Some(cmap)
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
|
||||
let p = PathBuf::from(dir);
|
||||
@@ -1182,6 +1187,18 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let path = find_bcmaps_dir()?.join(name);
|
||||
std::fs::read(path).ok().map(Cow::Owned)
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let file = BUILTIN_CMAPS.get_file(name)?;
|
||||
Some(Cow::Borrowed(file.contents()))
|
||||
}
|
||||
|
||||
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
|
||||
let mut stream = BinaryCMapStream::new(data);
|
||||
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
|
||||
@@ -1492,9 +1509,7 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
|
||||
}
|
||||
|
||||
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
parse_binary_cmap_encoding(&data).ok()
|
||||
}
|
||||
|
||||
@@ -1779,9 +1794,7 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
|
||||
if !name.ends_with("UCS2") {
|
||||
return None;
|
||||
}
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
|
||||
+113
-10
@@ -148,14 +148,17 @@ impl TextLine {
|
||||
self.text_with_formatting(false, false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
/// Get text with optional bold/italic/decorative markdown formatting.
|
||||
///
|
||||
/// `format_decorations` enables both geometrically detected source
|
||||
/// decorations: underline (`<u>`) and strikeout (`<s>`).
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_underline: bool,
|
||||
format_decorations: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
if !format_bold && !format_italic && !format_decorations {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -165,6 +168,7 @@ impl TextLine {
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
let mut current_strikeout = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -190,13 +194,16 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
// Check for style changes. Source decorations are exclusive:
|
||||
// `<u>`/`<s>` content stays free of `**`/`*` markers — consumers
|
||||
// (and the eval harnesses this feeds) match tag content literally,
|
||||
// and mixed nesting breaks that. A struck-and-underlined item is
|
||||
// emitted as struck text because deletion is the stronger semantic
|
||||
// distinction in redline documents.
|
||||
let item_strikeout = format_decorations && item.is_strikeout;
|
||||
let item_underline = format_decorations && item.is_underline && !item_strikeout;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline && !item_strikeout;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline && !item_strikeout;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -211,6 +218,10 @@ impl TextLine {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
if current_strikeout && !item_strikeout {
|
||||
result.push_str("</s>");
|
||||
current_strikeout = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -222,6 +233,10 @@ impl TextLine {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_strikeout && !current_strikeout {
|
||||
result.push_str("<s>");
|
||||
current_strikeout = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -244,6 +259,9 @@ impl TextLine {
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
if current_strikeout {
|
||||
result.push_str("</s>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
@@ -309,3 +327,88 @@ impl TextLine {
|
||||
|| space_already_exists)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod formatting_tests {
|
||||
use super::{ItemType, TextItem, TextLine};
|
||||
|
||||
fn item(text: &str, x: f32, width: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y: 100.0,
|
||||
width,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn line(items: Vec<TextItem>) -> TextLine {
|
||||
TextLine {
|
||||
items,
|
||||
y: 100.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_emits_semantic_strikeout() {
|
||||
let line = line(vec![item("deleted", 10.0, 42.0, true)]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_closes_strikeout_before_live_text() {
|
||||
let line = line(vec![
|
||||
item("keep", 10.0, 24.0, false),
|
||||
item("remove", 40.0, 42.0, true),
|
||||
item("keep", 88.0, 24.0, false),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"keep <s>remove</s> keep"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_coalesces_adjacent_struck_items() {
|
||||
let line = line(vec![
|
||||
item("deleted", 10.0, 42.0, true),
|
||||
item("words", 58.0, 30.0, true),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted words</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strikeout_takes_precedence_over_other_styles() {
|
||||
let mut decorated = item("deleted", 10.0, 42.0, true);
|
||||
decorated.is_bold = true;
|
||||
decorated.is_italic = true;
|
||||
decorated.is_underline = true;
|
||||
let line = line(vec![decorated]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
assert_eq!(line.text(), "deleted");
|
||||
}
|
||||
}
|
||||
|
||||
BIN
Binary file not shown.
+288
-4
@@ -8,12 +8,13 @@ use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
process_pdf_mem, process_pdf_mem_with_options, process_pdf_with_options, to_markdown,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
@@ -40,10 +41,11 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [{media_box}] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>"
|
||||
),
|
||||
);
|
||||
|
||||
let content = "BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
@@ -79,6 +81,186 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_recurring_contextual_folio_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R 7 0 R 9 0 R] /Count 4 >>",
|
||||
);
|
||||
for page_index in 0..4 {
|
||||
let page_id = 3 + page_index * 2;
|
||||
let content_id = page_id + 1;
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
page_id,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 11 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
),
|
||||
);
|
||||
let page_number = page_index + 1;
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 1 0 0 1 25 30 Tm ({page_number}) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Body page {page_number}) Tj ET"
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
content_id,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
}
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
11,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_malformed_unselected_page() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R] /Count 2 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
let content = "BT /F1 12 Tf 1 0 0 1 25 30 Tm (1) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Selected page text) Tj 0 -16 Td (More selected text) Tj 0 -16 Td (Still selected text) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 6 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Length 3 >>\nstream\nBI \nendstream",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
make_text_pdf(
|
||||
"BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET",
|
||||
"0 0 612 792",
|
||||
)
|
||||
}
|
||||
|
||||
fn make_digit_run_repro_pdf() -> Vec<u8> {
|
||||
let content = r#"BT
|
||||
/F1 12 Tf
|
||||
1 0 0 1 72 780 Tm (A\)) Tj
|
||||
1 0 0 1 96 780 Tm (The) Tj
|
||||
1 0 0 1 126 780 Tm (total) Tj
|
||||
1 0 0 1 166 780 Tm (of) Tj
|
||||
1 0 0 1 186 780 Tm (730) Tj
|
||||
1 0 0 1 220 780 Tm (seats) Tj
|
||||
1 0 0 1 262 780 Tm (was) Tj
|
||||
1 0 0 1 296 780 Tm (approved.) Tj
|
||||
1 0 0 1 72 755 Tm (B\)) Tj
|
||||
1 0 0 1 96 755 Tm (let) Tj
|
||||
1 0 0 1 120 755 Tm (log) Tj
|
||||
1 0 0 1 150 755 Tm (2) Tj
|
||||
1 0 0 1 164 755 Tm (=) Tj
|
||||
1 0 0 1 180 755 Tm (a) Tj
|
||||
1 0 0 1 72 720 Tm (C\) Control: The total of 730 seats was approved. let log 2 = a) Tj
|
||||
ET"#;
|
||||
make_text_pdf(content, "0 0 595 842")
|
||||
}
|
||||
|
||||
fn truncate_eof_marker(mut pdf: Vec<u8>) -> Vec<u8> {
|
||||
assert!(pdf.ends_with(b"%%EOF"));
|
||||
pdf.pop();
|
||||
@@ -330,6 +512,21 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
assert_eq!(lines[0].text(), "First Second Third");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_digit_only_text_runs_are_preserved_in_markdown() {
|
||||
let pdf = make_digit_run_repro_pdf();
|
||||
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract positioned text");
|
||||
assert!(items.iter().any(|item| item.text == "730"));
|
||||
assert!(items.iter().any(|item| item.text == "2"));
|
||||
|
||||
let result = process_pdf_mem(&pdf).expect("convert PDF to markdown");
|
||||
assert_eq!(
|
||||
result.markdown.expect("markdown output").trim(),
|
||||
"A) The total of 730 seats was approved.\nB) let log 2 = a\nC) Control: The total of 730 seats was approved. let log 2 = a"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MarkdownOptions Tests
|
||||
// ============================================================================
|
||||
@@ -547,6 +744,30 @@ fn test_markdown_from_items_page_breaks() {
|
||||
assert!(md.contains("Content on second page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_page_count_overload_includes_trailing_blank_pages_in_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
items.push(make_text_item(value, 25.0, 30.0, 12.0, page));
|
||||
items.push(make_text_item(
|
||||
"Company report footer",
|
||||
41.0,
|
||||
30.0,
|
||||
12.0,
|
||||
page,
|
||||
));
|
||||
}
|
||||
let options = MarkdownOptions {
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = to_markdown_from_items_with_rects_and_page_count(items, options, &[], 20);
|
||||
|
||||
assert!(md.contains("1 Company report footer"));
|
||||
assert!(md.contains("4 Company report footer"));
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Markdown From Lines Tests
|
||||
// ============================================================================
|
||||
@@ -1089,6 +1310,18 @@ fn test_snapshot_2013_app2() {
|
||||
assert_snapshot("2013-app2");
|
||||
}
|
||||
|
||||
/// First two pages of Shannon's "A Mathematical Theory of Communication"
|
||||
/// (1998 dvips 5.58 → Distiller 3 retypesetting). Canonical legacy-TeX PDF:
|
||||
/// non-embedded base-14 fonts with no /Widths (exercises the built-in AFM
|
||||
/// metrics fallback), Type3 PK bitmap math fonts with FontMatrix
|
||||
/// [1 0 0 -1 0 0] (exercises visual-size scaling), a two-line embedded drop
|
||||
/// cap, indent-only paragraph breaks, and display math that must not be
|
||||
/// detected as tables or headings.
|
||||
#[test]
|
||||
fn test_snapshot_shannon_entropy() {
|
||||
assert_snapshot("shannon-entropy-p1-2");
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Pages Needing OCR Tests
|
||||
// ============================================================================
|
||||
@@ -2915,6 +3148,57 @@ fn test_extract_pages_markdown_basic() {
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = extract_pages_markdown_mem(&pdf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 4);
|
||||
for (index, page) in result.pages.iter().enumerate() {
|
||||
assert!(page.markdown.contains("Company report footer"));
|
||||
assert!(
|
||||
!page
|
||||
.markdown
|
||||
.contains(&format!("{} Company report footer", index + 1)),
|
||||
"recurring contextual folio survived on page {}: {}",
|
||||
index + 1,
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_process_pdf_page_filter_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
|
||||
assert!(markdown.contains("Company report footer"));
|
||||
assert!(!markdown.contains("1 Company report footer"), "{markdown}");
|
||||
assert!(markdown.contains("Body page 1"));
|
||||
assert!(!markdown.contains("Body page 2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_selected_page_ignores_context_only_extraction_failure() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
let pages = extract_pages_markdown_mem(&pdf, Some(&[0])).unwrap();
|
||||
assert_eq!(pages.pages.len(), 1);
|
||||
assert!(pages.pages[0].markdown.contains("Selected page text"));
|
||||
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
assert!(markdown.contains("Selected page text"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_requested_page_extraction_failure_remains_fatal() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
assert!(extract_pages_markdown_mem(&pdf, Some(&[1])).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
@@ -56,7 +56,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
Form **4070** Employee’s Report (Rev. July 1996)
|
||||
|
||||
## of Tips to EmployerOMB No. 1545-0065
|
||||
|
||||
@@ -81,4 +81,3 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
Reprinted with corrections from *The Bell System Technical Journal,* Vol. 27, pp. 379–423, 623–656, July, October, 1948.
|
||||
|
||||
## A Mathematical Theory of Communication
|
||||
|
||||
### By C. E. SHANNON
|
||||
|
||||
INTRODUCTION
|
||||
|
||||
HE recent development of various methods of modulation such as PCM and PPM which exchange
|
||||
|
||||
# Tbandwidth for signal-to-noise ratio has intensified the interest in a general theory of communication. A
|
||||
|
||||
basis for such a theory is contained in the important papers of Nyquist¹ and Hartley² on this subject. In the present paper we will extend the theory to include a number of new factors, in particular the effect of noise in the channel, and the savings possible due to the statistical structure of the original message and due to the nature of the final destination of the information. The fundamental problem of communication is that of reproducing at one point either exactly or ap- proximately a message selected at another point. Frequently the messages have *meaning*; that is they refer to or are correlated according to some system with certain physical or conceptual entities. These semantic aspects of communication are irrelevant to the engineering problem. The significant aspect is that the actual message is one *selected from a set* of possible messages. The system must be designed to operate for each possible selection, not just the one which will actually be chosen since this is unknown at the time of design. If the number of messages in the set is finite then this number or any monotonic function of this number can be regarded as a measure of the information produced when one message is chosen from the set, all choices being equally likely. As was pointed out by Hartley the most natural choice is the logarithmic function. Although this definition must be generalized considerably when we consider the influence of the statistics of the message and when we have a continuous range of messages, we will in all cases use an essentially logarithmic measure. The logarithmic measure is more convenient for various reasons:
|
||||
|
||||
1. It is practically more useful. Parameters of engineering importance such as time, bandwidth, number of relays, etc., tend to vary linearly with the logarithm of the number of possibilities. For example, adding one relay to a group doubles the number of possible states of the relays. It adds 1 to the base 2 logarithm of this number. Doubling the time roughly squares the number of possible messages, or doubles the logarithm, etc.
|
||||
2. It is nearer to our intuitive feeling as to the proper measure. This is closely related to (1) since we in- tuitively measures entities by linear comparison with common standards. One feels, for example, that two punched cards should have twice the capacity of one for information storage, and two identical channels twice the capacity of one for transmitting information.
|
||||
3. It is mathematically more suitable. Many of the limiting operations are simple in terms of the loga- rithm but would require clumsy restatement in terms of the number of possibilities. The choice of a logarithmic base corresponds to the choice of a unit for measuring information. If the
|
||||
base 2 is used the resulting units may be called binary digits, or more briefly *bits,* a word suggested by
|
||||
|
||||
J. W. Tukey. A device with two stable positions, such as a relay or a flip-flop circuit, can store one bit of information. *N* such devices can store*N* bits, since the total number of possible states is 2
|
||||
*N* and log₂2 *N* = *N*. If the base 10 is used the units may be called decimal digits. Since
|
||||
|
||||
log₂*M* = log₁₀*M*= log₁₀2 = 3:32 log₁₀*M*;
|
||||
|
||||
1 Nyquist, H., “Certain Factors Affecting Telegraph Speed,” *Bell System Technical Journal,* April 1924, p. 324; “Certain Topics in Telegraph Transmission Theory,” *A.I.E.E. Trans.,* v. 47, April 1928, p. 617. 2 Hartley, R. V. L., “Transmission of Information,” *Bell System Technical Journal,* July 1928, p. 535.
|
||||
|
||||
INFORMATION SOURCE TRANSMITTER RECEIVER DESTINATION
|
||||
|
||||
SIGNAL RECEIVED SIGNAL MESSAGE MESSAGE
|
||||
|
||||
NOISE SOURCE
|
||||
|
||||
Fig. 1 — Schematic diagram of a general communication system.
|
||||
|
||||
a decimal digit is about 3 13 bits. A digit wheel on a desk computing machine has ten stable positions and therefore has a storage capacity of one decimal digit. In analytical work where integration and differentiation are involved the base *e* is sometimes useful. The resulting units of information will be called natural units. Change from the base *a* to base *b* merely requires multiplication by log*ba*. By a communication system we will mean a system of the type indicated schematically in Fig. 1. It consists of essentially five parts:
|
||||
|
||||
1. An *information source* which produces a message or sequence of messages to be communicated to the receiving terminal. The message may be of various types: (a) A sequence of letters as in a telegraph of teletype system; (b) A single function of time *f* (*t*) as in radio or telephony; (c) A function of time and other variables as in black and white television — here the message may be thought of as a function *f* (*x*; *y*;*t*) of two space coordinates and time, the light intensity at point (*x*; *y*) and time *t* on a pickup tube plate; (d) Two or more functions of time, say *f* (*t*), *g*(*t*), *h*(*t*) — this is the case in “three- dimensional” sound transmission or if the system is intended to service several individual channels in multiplex; (e) Several functions of several variables — in color television the message consists of three functions *f* (*x*; *y*;*t*), *g*(*x*; *y*;*t*), *h*(*x*; *y*;*t*) defined in a three-dimensional continuum — we may also think of these three functions as components of a vector field defined in the region — similarly, several black and white television sources would produce “messages” consisting of a number of functions of three variables; (f) Various combinations also occur, for example in television with an associated audio channel.
|
||||
2. A *transmitter* which operates on the message in some way to produce a signal suitable for trans- mission over the channel. In telephony this operation consists merely of changing sound pressure into a proportional electrical current. In telegraphy we have an encoding operation which produces a sequence of dots, dashes and spaces on the channel corresponding to the message. In a multiplex PCM system the different speech functions must be sampled, compressed, quantized and encoded, and finally interleaved properly to construct the signal. Vocoder systems, television and frequency modulation are other examples of complex operations applied to the message to obtain the signal.
|
||||
3. The *channel* is merely the medium used to transmit the signal from transmitter to receiver. It may be a pair of wires, a coaxial cable, a band of radio frequencies, a beam of light, etc.
|
||||
4. The *receiver* ordinarily performs the inverse operation of that done by the transmitter, reconstructing the message from the signal.
|
||||
5. The *destination* is the person (or thing) for whom the message is intended. We wish to consider certain general problems involving communication systems. To do this it is first
|
||||
necessary to represent the various elements involved as mathematical entities, suitably idealized from their
|
||||
|
||||
Generated
+1304
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,37 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.3"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "README.md"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[dependencies]
|
||||
console_error_panic_hook = "0.1"
|
||||
js-sys = "0.3"
|
||||
pdf-inspector = { path = ".." }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde-wasm-bindgen = "0.6"
|
||||
wasm-bindgen = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
wasm-bindgen-test = "0.3"
|
||||
|
||||
[profile.release]
|
||||
codegen-units = 1
|
||||
lto = true
|
||||
opt-level = "s"
|
||||
strip = true
|
||||
|
||||
[package.metadata.wasm-pack.profile.release]
|
||||
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
|
||||
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
|
||||
# size, and LTO optimizations above.
|
||||
wasm-opt = false
|
||||
@@ -0,0 +1,58 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
Third-party notices
|
||||
===================
|
||||
|
||||
Adobe CMaps
|
||||
-----------
|
||||
|
||||
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
|
||||
|
||||
Copyright 1990-2009 Adobe Systems Incorporated.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
Neither the name of Adobe Systems Incorporated nor the names of its
|
||||
contributors may be used to endorse or promote products derived from this
|
||||
software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -0,0 +1,60 @@
|
||||
# @firecrawl/pdf-inspector-wasm
|
||||
|
||||
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```ts
|
||||
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
|
||||
|
||||
await init();
|
||||
|
||||
const response = await fetch("/annual-report.pdf");
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
Pass options when you need selected pages or compact Markdown:
|
||||
|
||||
```ts
|
||||
const result = processPdf(pdf, {
|
||||
pages: [1, 3, 5],
|
||||
profile: "compact",
|
||||
includePageMarkers: true,
|
||||
});
|
||||
```
|
||||
|
||||
The package also exports:
|
||||
|
||||
- `detectPdf(pdf, options?)` for detection without extraction.
|
||||
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
|
||||
- `extractText(pdf)` for plain text.
|
||||
- `version()` for the WASM package version.
|
||||
|
||||
## Browser behavior
|
||||
|
||||
- Parsing runs locally. PDF bytes are not uploaded anywhere.
|
||||
- The build is single-threaded and does not require cross-origin isolation.
|
||||
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
|
||||
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
|
||||
- Image-only documents still require a separate OCR step.
|
||||
|
||||
## Build from source
|
||||
|
||||
```bash
|
||||
cargo install wasm-pack --version 0.15.0 --locked
|
||||
wasm-pack build wasm --target web --scope firecrawl --release
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
+440
@@ -0,0 +1,440 @@
|
||||
use pdf_inspector::{
|
||||
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
|
||||
ProcessMode,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wasm_bindgen::prelude::*;
|
||||
|
||||
#[wasm_bindgen(typescript_custom_section)]
|
||||
const TYPESCRIPT_TYPES: &str = r#"
|
||||
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
|
||||
export type MarkdownProfile = "fidelity" | "compact";
|
||||
|
||||
export interface ProcessOptions {
|
||||
/** Restrict extraction to these 1-indexed page numbers. */
|
||||
pages?: number[];
|
||||
/** Password for an encrypted PDF. */
|
||||
password?: string;
|
||||
/** Source-faithful output by default, or compact output for fewer tokens. */
|
||||
profile?: MarkdownProfile;
|
||||
/** Insert `<!-- Page N -->` markers between pages. */
|
||||
includePageMarkers?: boolean;
|
||||
/** Include image placeholders in Markdown output. */
|
||||
includeImages?: boolean;
|
||||
}
|
||||
|
||||
export interface PageOcrReasons {
|
||||
/** 1-indexed page number. */
|
||||
page: number;
|
||||
reasons: string[];
|
||||
}
|
||||
|
||||
export interface LayoutComplexity {
|
||||
isComplex: boolean;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithTables: number[];
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithColumns: number[];
|
||||
}
|
||||
|
||||
export interface PdfProcessResult {
|
||||
pdfType: PdfType;
|
||||
markdown?: string;
|
||||
pageCount: number;
|
||||
processingTimeMs: number;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesNeedingOcr: number[];
|
||||
ocrReasonsByPage: PageOcrReasons[];
|
||||
title?: string;
|
||||
confidence: number;
|
||||
layout: LayoutComplexity;
|
||||
hasEncodingIssues: boolean;
|
||||
}
|
||||
|
||||
export interface PdfClassification {
|
||||
pdfType: PdfType;
|
||||
pageCount: number;
|
||||
/** 0-indexed page numbers, matching the native Node.js API. */
|
||||
pagesNeedingOcr: number[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
|
||||
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
|
||||
export function classifyPdf(data: Uint8Array): PdfClassification;
|
||||
export function extractText(data: Uint8Array): string;
|
||||
export function version(): string;
|
||||
"#;
|
||||
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
|
||||
struct WasmProcessOptions {
|
||||
pages: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
profile: Option<WasmMarkdownProfile>,
|
||||
include_page_markers: Option<bool>,
|
||||
include_images: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
enum WasmMarkdownProfile {
|
||||
Fidelity,
|
||||
Compact,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPageOcrReasons {
|
||||
page: u32,
|
||||
reasons: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<PageOcrReasons> for WasmPageOcrReasons {
|
||||
fn from(value: PageOcrReasons) -> Self {
|
||||
Self {
|
||||
page: value.page,
|
||||
reasons: value.reasons,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmLayoutComplexity {
|
||||
is_complex: bool,
|
||||
pages_with_tables: Vec<u32>,
|
||||
pages_with_columns: Vec<u32>,
|
||||
}
|
||||
|
||||
impl From<LayoutComplexity> for WasmLayoutComplexity {
|
||||
fn from(value: LayoutComplexity) -> Self {
|
||||
Self {
|
||||
is_complex: value.is_complex,
|
||||
pages_with_tables: value.pages_with_tables,
|
||||
pages_with_columns: value.pages_with_columns,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfProcessResult {
|
||||
pdf_type: &'static str,
|
||||
markdown: Option<String>,
|
||||
page_count: u32,
|
||||
processing_time_ms: f64,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
|
||||
title: Option<String>,
|
||||
confidence: f64,
|
||||
layout: WasmLayoutComplexity,
|
||||
has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
impl From<PdfProcessResult> for WasmPdfProcessResult {
|
||||
fn from(value: PdfProcessResult) -> Self {
|
||||
Self {
|
||||
pdf_type: pdf_type_name(value.pdf_type),
|
||||
markdown: value.markdown,
|
||||
page_count: value.page_count,
|
||||
processing_time_ms: value.processing_time_ms as f64,
|
||||
pages_needing_ocr: value.pages_needing_ocr,
|
||||
ocr_reasons_by_page: value
|
||||
.ocr_reasons_by_page
|
||||
.into_iter()
|
||||
.map(Into::into)
|
||||
.collect(),
|
||||
title: value.title,
|
||||
confidence: value.confidence as f64,
|
||||
layout: value.layout.into(),
|
||||
has_encoding_issues: value.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfClassification {
|
||||
pdf_type: &'static str,
|
||||
page_count: u32,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
confidence: f64,
|
||||
}
|
||||
|
||||
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
|
||||
match pdf_type {
|
||||
PdfType::TextBased => "TextBased",
|
||||
PdfType::Scanned => "Scanned",
|
||||
PdfType::ImageBased => "ImageBased",
|
||||
PdfType::Mixed => "Mixed",
|
||||
}
|
||||
}
|
||||
|
||||
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
|
||||
js_sys::Error::new(&format!("{context}: {error}")).into()
|
||||
}
|
||||
|
||||
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
|
||||
if value.is_undefined() || value.is_null() {
|
||||
return Ok(WasmProcessOptions::default());
|
||||
}
|
||||
|
||||
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
|
||||
}
|
||||
|
||||
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
|
||||
let options = deserialize_options(value)?;
|
||||
if options
|
||||
.pages
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(js_error(
|
||||
"invalid options",
|
||||
"pages are 1-indexed; page 0 is invalid",
|
||||
));
|
||||
}
|
||||
|
||||
let mut result = PdfOptions::new().mode(mode);
|
||||
if let Some(pages) = options.pages {
|
||||
result = result.pages(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(profile) = options.profile {
|
||||
result.markdown.profile = match profile {
|
||||
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
|
||||
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
|
||||
};
|
||||
}
|
||||
if let Some(include_page_markers) = options.include_page_markers {
|
||||
result.markdown.include_page_numbers = include_page_markers;
|
||||
}
|
||||
if let Some(include_images) = options.include_images {
|
||||
result.markdown.include_images = include_images;
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
|
||||
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
|
||||
}
|
||||
|
||||
fn initialize() {
|
||||
console_error_panic_hook::set_once();
|
||||
}
|
||||
|
||||
/// Process PDF bytes entirely inside WebAssembly.
|
||||
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
|
||||
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::Full)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("process PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Classify PDF bytes without extracting text or producing Markdown.
|
||||
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
|
||||
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::DetectOnly)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("detect PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Return the lightweight classification shape used by the native Node API.
|
||||
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
|
||||
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
|
||||
serialize(&WasmPdfClassification {
|
||||
pdf_type: pdf_type_name(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes without Markdown conversion.
|
||||
#[wasm_bindgen(js_name = extractText, skip_typescript)]
|
||||
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
|
||||
initialize();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
|
||||
.map_err(|error| js_error("extract text", error))?;
|
||||
Ok(
|
||||
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
|
||||
.into_iter()
|
||||
.map(|line| line.text())
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n"),
|
||||
)
|
||||
}
|
||||
|
||||
/// Return the WebAssembly package version.
|
||||
#[wasm_bindgen(skip_typescript)]
|
||||
pub fn version() -> String {
|
||||
env!("CARGO_PKG_VERSION").to_string()
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_arch = "wasm32"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use js_sys::Reflect;
|
||||
use wasm_bindgen_test::*;
|
||||
|
||||
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
|
||||
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
|
||||
|
||||
fn synthetic_korea1_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
|
||||
// There is deliberately no ToUnicode stream: decoding must use the
|
||||
// embedded predefined CMap rather than lopdf's plain-text fallback.
|
||||
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
|
||||
// the bottom edge so they look exactly like a numeric page footer.
|
||||
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn processes_pdf_to_markdown() {
|
||||
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
|
||||
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn rejects_non_pdf_bytes() {
|
||||
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn classifies_and_extracts_plain_text() {
|
||||
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
|
||||
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let text = extract_text(TEXT_PDF).expect("extract text");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!text.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn extracts_cjk_and_preserves_numeric_page_footer() {
|
||||
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
|
||||
|
||||
assert_eq!(text, "가\n42");
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn opens_encrypted_pdf_with_password() {
|
||||
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
|
||||
|
||||
let options = js_sys::Object::new();
|
||||
Reflect::set(
|
||||
&options,
|
||||
&JsValue::from_str("password"),
|
||||
&JsValue::from_str("secret123"),
|
||||
)
|
||||
.expect("set password");
|
||||
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user