Compare commits
19
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f729f92ddf | ||
|
|
874919a3ea | ||
|
|
1843a29819 | ||
|
|
b129d3bf62 | ||
|
|
ae6246ba0c | ||
|
|
3fb545284b | ||
|
|
1c32e4bd69 | ||
|
|
8121ae97ce | ||
|
|
98990cc550 | ||
|
|
a15ec2d68d | ||
|
|
7b7960ee73 | ||
|
|
7188667045 | ||
|
|
6ff104409a | ||
|
|
5e8f1570f6 | ||
|
|
b31e4b1727 | ||
|
|
3c6eb8bf6b | ||
|
|
5b287341a0 | ||
|
|
55d50ad1c4 | ||
|
|
a910b7df1d |
+55
-11
@@ -14,13 +14,15 @@ jobs:
|
||||
name: Test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
@@ -32,29 +34,34 @@ jobs:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: rustfmt
|
||||
|
||||
- name: Check formatting
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Check WASM formatting
|
||||
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
|
||||
|
||||
clippy:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: clippy
|
||||
|
||||
@@ -68,15 +75,52 @@ jobs:
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: build
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
key: wasm
|
||||
|
||||
- name: Check WebAssembly bindings
|
||||
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
|
||||
|
||||
- name: Check root package for WebAssembly
|
||||
run: cargo check --target wasm32-unknown-unknown
|
||||
|
||||
- name: Lint WebAssembly bindings
|
||||
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Test WebAssembly package
|
||||
run: wasm-pack test --node --release wasm
|
||||
|
||||
@@ -24,13 +24,13 @@ jobs:
|
||||
name: github-pages
|
||||
url: ${{ steps.deploy.outputs.page_url }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Upload site artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
|
||||
with:
|
||||
path: site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deploy
|
||||
uses: actions/deploy-pages@v4
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5.0.0
|
||||
|
||||
@@ -27,7 +27,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -85,17 +85,19 @@ jobs:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
uses: rust-lang/crates-io-auth-action@c6f97d42243bad5fab37ca0427f495c86d5b1a18 # v1.0.5
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -98,21 +98,21 @@ jobs:
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Build wheel
|
||||
uses: PyO3/maturin-action@v1
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: wheels-${{ matrix.target }}
|
||||
path: dist/*.whl
|
||||
@@ -124,16 +124,16 @@ jobs:
|
||||
name: Build sdist
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Build sdist
|
||||
uses: PyO3/maturin-action@v1
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
command: sdist
|
||||
args: --out dist
|
||||
|
||||
- name: Upload sdist
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: sdist
|
||||
path: dist/*.tar.gz
|
||||
@@ -149,7 +149,7 @@ jobs:
|
||||
id-token: write
|
||||
steps:
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
@@ -158,7 +158,7 @@ jobs:
|
||||
run: ls -la dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
with:
|
||||
packages-dir: dist
|
||||
# Tolerate already-uploaded files so a manual re-run can complete
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
name: Publish WebAssembly package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['wasm/Cargo.toml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
package_exists: ${{ steps.check.outputs.package_exists }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
|
||||
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
|
||||
echo "package_exists=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
echo "The initial package must be published once before trusted publishing can be configured."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "package_exists=true" >> "$GITHUB_OUTPUT"
|
||||
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
publish:
|
||||
name: Build and publish
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Build browser package
|
||||
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
|
||||
- name: Prepare package metadata
|
||||
run: |
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const path = "wasm/pkg/package.json"
|
||||
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
|
||||
pkg.name = "@firecrawl/pdf-inspector-wasm"
|
||||
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
|
||||
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
|
||||
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
|
||||
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
|
||||
pkg.publishConfig = { access: "public" }
|
||||
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
- name: Inspect package contents
|
||||
run: npm pack --dry-run ./wasm/pkg
|
||||
|
||||
- name: Publish package
|
||||
run: npm publish ./wasm/pkg --provenance --access public
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -61,31 +61,61 @@ jobs:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
# napi-cross builds gnu targets against an old glibc sysroot for
|
||||
# broad distro compatibility; musl targets cross-compile with
|
||||
# zig via cargo-zigbuild (napi's -x flag).
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
build-flags: --use-napi-cross
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: ${{ matrix.target }}
|
||||
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install zig
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: mlugg/setup-zig@d1434d08867e3ee9daa34448df10607b98908d29 # v2.2.1
|
||||
with:
|
||||
version: 0.14.1
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: taiki-e/install-action@67729d5c413db75907f0ad1e39bb04b9c868ff60 # v2.85.7
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
tool: cargo-zigbuild
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
~/.napi-rs
|
||||
napi/target/
|
||||
key: ${{ runner.os }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
key: ${{ matrix.target }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-napi-
|
||||
${{ matrix.target }}-cargo-napi-
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: napi
|
||||
@@ -93,10 +123,10 @@ jobs:
|
||||
|
||||
- name: Build native addon
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release
|
||||
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
|
||||
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
@@ -104,7 +134,7 @@ jobs:
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
@@ -112,23 +142,68 @@ jobs:
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
smoke-test:
|
||||
name: Smoke test ${{ matrix.target }}
|
||||
needs: [check-version, build]
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-gnu
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-musl
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Download native binary
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi
|
||||
|
||||
- name: Download generated JS bindings
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: napi
|
||||
|
||||
# musl binaries must load under a real musl libc, so run inside Alpine.
|
||||
- name: Run smoke test (Alpine)
|
||||
if: contains(matrix.target, 'musl')
|
||||
run: docker run --rm -v "$PWD:/repo" -w /repo/napi node:24-alpine node test.mjs
|
||||
|
||||
- name: Run smoke test
|
||||
if: ${{ !contains(matrix.target, 'musl') }}
|
||||
working-directory: napi
|
||||
run: node test.mjs
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: [check-version, build]
|
||||
needs: [check-version, build, smoke-test]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
@@ -154,9 +229,12 @@ jobs:
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"linux-x64-musl": { os: ["linux"], cpu: ["x64"], libc: ["musl"] },
|
||||
"linux-arm64-gnu": { os: ["linux"], cpu: ["arm64"], libc: ["glibc"] },
|
||||
"linux-arm64-musl": { os: ["linux"], cpu: ["arm64"], libc: ["musl"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
|
||||
+3
-1
@@ -1,10 +1,13 @@
|
||||
# Rust build artifacts
|
||||
/target/
|
||||
/wasm/target/
|
||||
/wasm/pkg/
|
||||
debug/
|
||||
*.pdb
|
||||
|
||||
# Cargo lock (optional for libraries)
|
||||
Cargo.lock
|
||||
!/wasm/Cargo.lock
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
@@ -39,4 +42,3 @@ test_output/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
+19
-13
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.6"
|
||||
version = "0.1.7"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
@@ -12,13 +12,13 @@ readme = "docs/rust-api.md"
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"src/**",
|
||||
"external/bcmaps/**",
|
||||
"docs/rust-api.md",
|
||||
"LICENSE",
|
||||
"/src/**",
|
||||
"/external/bcmaps/**",
|
||||
"/docs/rust-api.md",
|
||||
"/LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"pdf_inspector.pyi",
|
||||
"/pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
@@ -29,18 +29,11 @@ crate-type = ["lib", "cdylib"]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
# Parallel processing
|
||||
rayon = "1.10"
|
||||
|
||||
# Logging
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Text processing
|
||||
regex = "1.10"
|
||||
@@ -50,6 +43,19 @@ unicode-normalization = "0.1"
|
||||
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
||||
ttf-parser = "0.25"
|
||||
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.42.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.42.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.3"
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -19,24 +19,26 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only local engines without model-based PDF parsing are shown; OCR was disabled. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | 3.3s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 3.0s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.59 | 0.84 | 0.27 | 0.00 | 23s |
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the top of that range without any OCR, in 3.3 seconds.
|
||||
Results were refreshed on July 31, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Speed is the median of five alternating or rotating complete corpus runs after an excluded warm-up run, with each parser processing documents sequentially in a single process.
|
||||
|
||||
**Where we do well:** The best overall, reading-order, and table scores among the direct extraction engines shown.
|
||||
The complete parser configuration, per-document predictions, evaluator output, and generated charts are available in the [reproducible results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
**Where we lag:** Some direct engines remain slightly faster, and OCR-based engines can recover text that has no usable PDF text layer.
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. In this comparison, pdf-inspector delivered the higher overall, reading-order, and table scores, along with the fastest complete run. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
|
||||
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
|
||||
|
||||
@@ -76,6 +78,26 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Browser WebAssembly
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
```javascript
|
||||
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
|
||||
|
||||
await init();
|
||||
const response = await fetch('/document.pdf');
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
> Full API reference: [wasm/README.md](wasm/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
@@ -187,6 +209,7 @@ src/
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
wasm/ — Browser bindings (wasm-bindgen)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
|
||||
@@ -28,6 +28,17 @@ The OpenDataLoader repository is external and keeps its normal
|
||||
temporary directory before evaluating it, so the baseline and candidate cannot
|
||||
overwrite one another.
|
||||
|
||||
## Published comparison protocol
|
||||
|
||||
The public benchmark table was refreshed on July 31, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Every engine processed the same 200
|
||||
PDFs sequentially in a single process with OCR disabled. Reported speed is the
|
||||
median of five alternating or rotating complete corpus runs after an excluded
|
||||
warm-up run; quality scores come from the benchmark evaluator over all 200
|
||||
outputs. Raw timings, predictions, evaluations, and charts are available in the
|
||||
[results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Optional backend evidence probe
|
||||
|
||||
The evidence probe compares positioned `pdf2md` items with MuPDF structured
|
||||
|
||||
@@ -19,3 +19,20 @@ The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
|
||||
## Browser WebAssembly package
|
||||
|
||||
The browser package is published as `@firecrawl/pdf-inspector-wasm`. Its version lives in `wasm/Cargo.toml`, and `.github/workflows/publish-wasm.yml` builds the `web` target with `wasm-pack` before publishing the generated package.
|
||||
|
||||
The npm package must exist before a trusted publisher can be configured. For the first release only:
|
||||
|
||||
1. Build with `wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release`.
|
||||
2. Inspect with `npm pack --dry-run ./wasm/pkg`.
|
||||
3. Publish with `npm publish ./wasm/pkg --access public` from an authorized maintainer session.
|
||||
4. In the package settings on npm, configure the GitHub Actions trusted publisher:
|
||||
- Organization: `firecrawl`
|
||||
- Repository: `pdf-inspector`
|
||||
- Workflow: `publish-wasm.yml`
|
||||
- Allowed action: `npm publish`
|
||||
|
||||
After that one-time bootstrap, bumping the version in `wasm/Cargo.toml` and merging it to `main` publishes through OIDC. Until the package exists, the workflow exits cleanly without attempting an unauthenticated first publish. See npm's [trusted publishing documentation](https://docs.npmjs.com/trusted-publishers/) for the registry-side setup.
|
||||
|
||||
+7
-5
@@ -14,15 +14,17 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | 3.3s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 3.0s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
|
||||
+7
-5
@@ -14,15 +14,17 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | 3.3s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 3.0s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
|
||||
Generated
+25
-3
@@ -499,11 +499,13 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"js-sys",
|
||||
"libc",
|
||||
"r-efi",
|
||||
"rand_core",
|
||||
"wasip2",
|
||||
"wasip3",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -557,6 +559,25 @@ version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
|
||||
|
||||
[[package]]
|
||||
name = "include_dir"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||
dependencies = [
|
||||
"include_dir_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir_macros"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.0"
|
||||
@@ -672,9 +693,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.41.0"
|
||||
version = "0.42.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -830,9 +851,10 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.6"
|
||||
version = "0.1.7"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
"log",
|
||||
"lopdf",
|
||||
"once_cell",
|
||||
|
||||
+14
-9
@@ -14,15 +14,17 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
@@ -32,7 +34,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries for **Linux x64**, **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -114,9 +116,12 @@ Prebuilt binaries ship as platform-specific packages installed automatically via
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| Linux | x64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-x64-musl` |
|
||||
| Linux | ARM64 (glibc) | `@firecrawl/pdf-inspector-linux-arm64-gnu` |
|
||||
| Linux | ARM64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-arm64-musl` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
+6
-3
@@ -8,9 +8,12 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+10
-4
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.11.1",
|
||||
"version": "1.12.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -37,6 +37,9 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-unknown-linux-musl",
|
||||
"aarch64-unknown-linux-gnu",
|
||||
"aarch64-unknown-linux-musl",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
]
|
||||
@@ -49,8 +52,11 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.1"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ build-backend = "maturin"
|
||||
name = "pdf-inspector"
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.5"
|
||||
version = "0.2.6"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
+1100
-356
File diff suppressed because it is too large
Load Diff
@@ -63,6 +63,7 @@ fn json_escape(s: &str) -> String {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
|
||||
@@ -190,6 +190,7 @@ fn print_layout_info(layout: &LayoutComplexity) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
|
||||
@@ -162,7 +162,7 @@ pub(crate) fn extract_page_text_items(
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
|
||||
// Build font encoding maps from Differences arrays
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
|
||||
|
||||
// Build font width info for accurate text positioning
|
||||
let font_widths = build_font_widths(doc, &fonts);
|
||||
|
||||
+154
-12
@@ -497,9 +497,13 @@ pub(crate) fn get_operand_bytes(obj: &Object) -> Option<&[u8]> {
|
||||
/// Build encoding maps for all fonts on a page.
|
||||
/// Returns `(encodings, has_gid_fonts)` where `has_gid_fonts` is true when
|
||||
/// any font uses raw glyph ID names (gidNNNNN) that can't be decoded.
|
||||
/// Gid names whose codes the font's own ToUnicode CMap maps are decodable
|
||||
/// and do not set the flag (LibreOffice subsets write /gidNNNN Differences
|
||||
/// names alongside a complete ToUnicode CMap).
|
||||
pub(crate) fn build_font_encodings(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
cmaps: &FontCMaps,
|
||||
) -> (PageFontEncodings, bool) {
|
||||
let mut encodings = PageFontEncodings::new();
|
||||
let mut has_gid_fonts = false;
|
||||
@@ -508,7 +512,9 @@ pub(crate) fn build_font_encodings(
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
|
||||
if let Some(result) = parse_font_encoding(doc, font_dict) {
|
||||
if result.gid_glyph_count > 0 {
|
||||
if !result.gid_codes.is_empty()
|
||||
&& !tounicode_maps_codes(font_dict, cmaps, &result.gid_codes)
|
||||
{
|
||||
has_gid_fonts = true;
|
||||
}
|
||||
if !result.map.is_empty() {
|
||||
@@ -520,6 +526,34 @@ pub(crate) fn build_font_encodings(
|
||||
(encodings, has_gid_fonts)
|
||||
}
|
||||
|
||||
/// True when the font's ToUnicode CMap maps the gid-named character codes,
|
||||
/// so the Differences entries still decode through the CMap.
|
||||
fn tounicode_maps_codes(font_dict: &lopdf::Dictionary, cmaps: &FontCMaps, codes: &[u8]) -> bool {
|
||||
let Some(obj_ref) = font_dict
|
||||
.get(b"ToUnicode")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
else {
|
||||
return false;
|
||||
};
|
||||
let Some(entry) = cmaps.get_by_obj(obj_ref.0) else {
|
||||
return false;
|
||||
};
|
||||
// At least one gid code usably mapped means the CMap addresses these
|
||||
// codes; remaining unmapped codes are subset leftovers (e.g. the
|
||||
// component glyphs of an emoji ZWJ sequence mapped whole on its first
|
||||
// code). A mapping is usable only when extraction would accept it —
|
||||
// empty or U+FFFD results are rejected there as invalid. Fonts whose
|
||||
// CMap ignores the gid codes entirely stay flagged, and the downstream
|
||||
// garbage/encoding checks still catch partial damage.
|
||||
codes.iter().any(|&code| {
|
||||
entry
|
||||
.primary
|
||||
.lookup(code as u16)
|
||||
.is_some_and(|s| !s.is_empty() && !s.contains('\u{FFFD}'))
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse font encoding from a font dictionary
|
||||
pub(crate) fn parse_font_encoding(
|
||||
doc: &Document,
|
||||
@@ -558,11 +592,10 @@ pub(crate) fn parse_font_encoding(
|
||||
/// Result of parsing an encoding dictionary's Differences array.
|
||||
pub(crate) struct EncodingResult {
|
||||
pub map: FontEncodingMap,
|
||||
/// Number of glyph names matching the `gidNNNNN` pattern (raw glyph IDs).
|
||||
/// These indicate a font with unresolvable encoding — the glyph IDs
|
||||
/// reference the original font's glyph table, but without the original
|
||||
/// font's cmap there is no way to map them to Unicode.
|
||||
pub gid_glyph_count: u32,
|
||||
/// Character codes whose glyph names match the `gidNNNNN` pattern (raw
|
||||
/// glyph IDs). These reference the original font's glyph table and are
|
||||
/// only decodable when the font's ToUnicode CMap maps the code.
|
||||
pub gid_codes: Vec<u8>,
|
||||
}
|
||||
|
||||
/// Parse an encoding dictionary with Differences array
|
||||
@@ -588,7 +621,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
let mut encoding_map = FontEncodingMap::new();
|
||||
let mut current_code: u8 = 0;
|
||||
let mut ligature_count = 0u32;
|
||||
let mut gid_glyph_count = 0u32;
|
||||
let mut gid_codes: Vec<u8> = Vec::new();
|
||||
|
||||
for item in diff_array {
|
||||
match item {
|
||||
@@ -614,7 +647,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
&& glyph_name.len() >= 4
|
||||
&& glyph_name[3..].chars().all(|c| c.is_ascii_digit())
|
||||
{
|
||||
gid_glyph_count += 1;
|
||||
gid_codes.push(current_code);
|
||||
}
|
||||
if let Some(ch) = mapped_char {
|
||||
encoding_map.insert(current_code, ch);
|
||||
@@ -638,16 +671,16 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
);
|
||||
}
|
||||
|
||||
if gid_glyph_count > 0 {
|
||||
if !gid_codes.is_empty() {
|
||||
debug!(
|
||||
" Differences: {} gid-encoded glyphs (unresolvable without original font)",
|
||||
gid_glyph_count
|
||||
" Differences: {} gid-encoded glyphs (decodable only via ToUnicode)",
|
||||
gid_codes.len()
|
||||
);
|
||||
}
|
||||
|
||||
Some(EncodingResult {
|
||||
map: encoding_map,
|
||||
gid_glyph_count,
|
||||
gid_codes,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1938,4 +1971,113 @@ mod tests {
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
fn gid_font_doc(bfchar: Option<&str>) -> (Document, lopdf::ObjectId) {
|
||||
use lopdf::Stream;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let cmap = format!(
|
||||
"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
{}
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end",
|
||||
bfchar.unwrap_or_default()
|
||||
);
|
||||
let tounicode_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
cmap.into_bytes(),
|
||||
)));
|
||||
let enc_id = doc.add_object(dictionary! {
|
||||
"Type" => "Encoding",
|
||||
"Differences" => vec![
|
||||
1.into(),
|
||||
Object::Name(b"gid1283".to_vec()),
|
||||
Object::Name(b"gid1464".to_vec()),
|
||||
],
|
||||
});
|
||||
let mut font = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "ABCDEF+OpenSymbol",
|
||||
"Encoding" => Object::Reference(enc_id),
|
||||
};
|
||||
if bfchar.is_some() {
|
||||
font.set("ToUnicode", Object::Reference(tounicode_id));
|
||||
}
|
||||
let font_id = doc.add_object(font);
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn gid_flagged(bfchar: Option<&str>) -> bool {
|
||||
let (doc, page_id) = gid_font_doc(bfchar);
|
||||
let cmaps = FontCMaps::from_doc(&doc);
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap();
|
||||
let (_, has_gid_fonts) = build_font_encodings(&doc, &fonts, &cmaps);
|
||||
has_gid_fonts
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_covering_tounicode_are_not_flagged() {
|
||||
// LibreOffice subsets write /gidNNNN Differences names alongside a
|
||||
// ToUnicode CMap that decodes those codes; the page must not be
|
||||
// flagged as unresolvable (which would suppress the whole document's
|
||||
// markdown when every page carries such a font).
|
||||
assert!(!gid_flagged(Some("<01> <2022>\n<02> <25E6>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_partial_tounicode_are_not_flagged() {
|
||||
// An emoji ZWJ sequence maps whole on its first code; the remaining
|
||||
// component-glyph codes are subset leftovers, not damage.
|
||||
assert!(!gid_flagged(Some(
|
||||
"<01> <D83DDC68200DD83DDC69200DD83DDC67>"
|
||||
)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_without_tounicode_are_flagged() {
|
||||
assert!(
|
||||
gid_flagged(None),
|
||||
"gid glyphs without ToUnicode are unresolvable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_disjoint_tounicode_are_flagged() {
|
||||
// A ToUnicode that never addresses the gid codes leaves them
|
||||
// unresolvable.
|
||||
assert!(gid_flagged(Some("<10> <0041>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_replacement_char_tounicode_are_flagged() {
|
||||
// A mapping to U+FFFD is not usable — extraction rejects it as an
|
||||
// invalid CMap result — so it must not clear the gid flag.
|
||||
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
|
||||
}
|
||||
}
|
||||
|
||||
+810
-17
@@ -960,22 +960,713 @@ fn spans_multiple_columns(item: &TextItem, columns: &[ColumnRegion]) -> bool {
|
||||
overlap_count >= 2
|
||||
}
|
||||
|
||||
/// Check if a text item is likely a page number
|
||||
fn is_page_number(item: &TextItem) -> bool {
|
||||
const PAGE_NUMBER_Y_TOLERANCE: f32 = 3.0;
|
||||
const PAGE_NUMBER_CONTEXT_GAP_EM: f32 = 1.5;
|
||||
const PAGE_NUMBER_BOTTOM_Y: f32 = 100.0;
|
||||
const PAGE_NUMBER_TOP_Y: f32 = 720.0;
|
||||
const SPREAD_MIN_CONTENT_WIDTH_EM: f32 = 40.0;
|
||||
const SPREAD_EDGE_FRACTION: f32 = 0.25;
|
||||
const ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM: f32 = 26.0;
|
||||
|
||||
type ContextualCandidateOccurrence = (u32, f32, Vec<(usize, u32)>);
|
||||
|
||||
fn page_number_value(item: &TextItem) -> Option<u32> {
|
||||
if !matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let text = item.text.trim();
|
||||
|
||||
// Must be 1-4 digits only
|
||||
if text.is_empty() || text.len() > 4 {
|
||||
return false;
|
||||
}
|
||||
if !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return false;
|
||||
if text.is_empty() || text.len() > 4 || !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Must be at top or bottom of page.
|
||||
// US Letter = 792pt, A4 = 841pt. Page numbers are typically in the
|
||||
// top ~5% or bottom ~12% of the page.
|
||||
item.y > 720.0 || item.y < 100.0
|
||||
if item.y <= PAGE_NUMBER_TOP_Y && item.y >= PAGE_NUMBER_BOTTOM_Y {
|
||||
return None;
|
||||
}
|
||||
|
||||
text.parse().ok()
|
||||
}
|
||||
|
||||
/// Mark numeric slots that advance inside a repeated deep-margin line.
|
||||
///
|
||||
/// A folio can be emitted as part of a footer text run (for example,
|
||||
/// `42 Company report`) and therefore look contextual on a single page. Across
|
||||
/// the document, however, the surrounding text and Y position repeat while the
|
||||
/// numeric slot advances. Require that full signal before treating the slot as
|
||||
/// a folio so constant metadata and substantive rows near the page edge remain
|
||||
/// untouched.
|
||||
fn mark_repeated_folio_candidates(
|
||||
occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>>,
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
// Both thresholds are evidence floors: short documents still need four
|
||||
// occurrences, while long documents also need meaningful coverage. Using
|
||||
// `min` here would make one occurrence sufficient in a one-page document.
|
||||
let min_pages = 4usize.max((document_page_count * 30).div_ceil(100));
|
||||
|
||||
for occurrences in occurrences_by_signature.into_values() {
|
||||
if occurrences.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let distinct_pages: HashSet<u32> = occurrences.iter().map(|(page, _, _)| *page).collect();
|
||||
// Repeated table rows or duplicated drawing labels can share a
|
||||
// signature multiple times on one page. They are not running folios.
|
||||
if distinct_pages.len() != occurrences.len() || distinct_pages.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let min_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if max_y - min_y >= PAGE_NUMBER_Y_TOLERANCE {
|
||||
continue;
|
||||
}
|
||||
|
||||
let slot_count = occurrences[0].2.len();
|
||||
if slot_count == 0
|
||||
|| occurrences
|
||||
.iter()
|
||||
.any(|(_, _, candidates)| candidates.len() != slot_count)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
for slot in 0..slot_count {
|
||||
let mut values: Vec<(u32, u32, usize)> = occurrences
|
||||
.iter()
|
||||
.map(|(page, _, candidates)| {
|
||||
let (index, value) = candidates[slot];
|
||||
(*page, value, index)
|
||||
})
|
||||
.collect();
|
||||
values.sort_by_key(|(page, _, _)| *page);
|
||||
|
||||
let unique_values: HashSet<u32> = values.iter().map(|(_, value, _)| *value).collect();
|
||||
let mostly_unique = unique_values.len() * 5 >= values.len() * 4;
|
||||
let page_tracking_pairs = values
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let page_delta = pair[1].0 - pair[0].0;
|
||||
let value_delta = pair[1].1.saturating_sub(pair[0].1);
|
||||
value_delta == page_delta || value_delta == page_delta.saturating_mul(2)
|
||||
})
|
||||
.count();
|
||||
let mostly_tracks_page_order = page_tracking_pairs * 5 >= (values.len() - 1) * 4;
|
||||
// A running folio can be offset by front matter or advance twice per
|
||||
// PDF page in a two-page spread, but its magnitude should still be
|
||||
// plausible for the document. This keeps changing metadata such as
|
||||
// a sequence of years from becoming a deletion signal.
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
let plausible_magnitude = values
|
||||
.iter()
|
||||
.all(|(_, value, _)| *value <= max_plausible_folio);
|
||||
|
||||
if mostly_unique && mostly_tracks_page_order && plausible_magnitude {
|
||||
for (_, _, index) in values {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark the contextual half of a facing-page folio pair.
|
||||
///
|
||||
/// A landscape PDF can contain two printed pages per PDF page. One folio may be
|
||||
/// isolated while the other touches footer text; they remain a pair because
|
||||
/// they are consecutive, share a deep-margin baseline, and sit on opposite
|
||||
/// sides of the spread. The isolated half is strong evidence that the touching
|
||||
/// half is also a folio.
|
||||
fn mark_spread_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let mut candidates_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
if value.is_some() {
|
||||
candidates_by_page
|
||||
.entry(items[index].page)
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
}
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
for (page, page_candidates) in candidates_by_page {
|
||||
let Some(&(page_left, page_right)) = page_bounds.get(&page) else {
|
||||
continue;
|
||||
};
|
||||
let page_width = page_right - page_left;
|
||||
if page_width <= 0.0 {
|
||||
continue;
|
||||
}
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
let max_pair_font_size = page_width / SPREAD_MIN_CONTENT_WIDTH_EM;
|
||||
let edge_side = |index: usize| {
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Index strong folio evidence by value and spread edge. Sorted
|
||||
// baselines let each contextual candidate query only the two adjacent
|
||||
// values on the opposite edge in O(log n), rather than comparing every
|
||||
// candidate pair on numeric-heavy pages.
|
||||
let mut known_baselines: HashMap<(u32, bool), Vec<f32>> = HashMap::new();
|
||||
for &index in &page_candidates {
|
||||
if (contextual[index] && !explicit_folio[index])
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let value = candidate_values[index].unwrap();
|
||||
known_baselines
|
||||
.entry((value, side))
|
||||
.or_default()
|
||||
.push(items[index].y);
|
||||
}
|
||||
for baselines in known_baselines.values_mut() {
|
||||
baselines.sort_by(f32::total_cmp);
|
||||
}
|
||||
|
||||
for index in page_candidates {
|
||||
if !contextual[index]
|
||||
|| explicit_folio[index]
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(value) = candidate_values[index] else {
|
||||
continue;
|
||||
};
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let y = items[index].y;
|
||||
let paired = [value.checked_sub(1), value.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|other_value| known_baselines.get(&(other_value, !side)))
|
||||
.any(|baselines| {
|
||||
let first = baselines
|
||||
.partition_point(|baseline| *baseline <= y - PAGE_NUMBER_Y_TOLERANCE);
|
||||
baselines
|
||||
.get(first)
|
||||
.is_some_and(|baseline| *baseline < y + PAGE_NUMBER_Y_TOLERANCE)
|
||||
});
|
||||
if paired {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark a contextual folio that alternates with an isolated folio on the
|
||||
/// neighboring PDF page.
|
||||
///
|
||||
/// Facing pages commonly put folios on opposite outer edges. A running header
|
||||
/// can touch the right-hand folio while the next left-hand folio is isolated.
|
||||
/// A single isolated candidate is not enough to remove nearby contextual text.
|
||||
/// Require a second pre-existing anchor in the same advancing sequence, along
|
||||
/// with a genuinely wide content span, stable baselines/font sizes, and
|
||||
/// alternating outer edges.
|
||||
fn mark_adjacent_page_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Do not let newly inferred candidates recursively become evidence for
|
||||
// later candidates; every match must be anchored by evidence established
|
||||
// before this cross-page pass.
|
||||
let strong_folio_evidence = explicit_folio.to_vec();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
let edge_side = |index: usize| {
|
||||
let &(page_left, page_right) = page_bounds.get(&items[index].page)?;
|
||||
let page_width = page_right - page_left;
|
||||
if page_width < items[index].font_size * ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM {
|
||||
return None;
|
||||
}
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Anchor sequences by outer edge and the value/page offset. This enforces
|
||||
// forward page tracking and lets candidates query adjacent pages directly,
|
||||
// while a baseline-sorted index finds a second independent anchor without
|
||||
// a document-wide quadratic scan.
|
||||
let mut anchors_by_page: HashMap<(u32, bool, i64), Vec<usize>> = HashMap::new();
|
||||
let mut anchors_by_sequence: HashMap<(bool, i64), Vec<usize>> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if *value > max_plausible_folio || (contextual[index] && !strong_folio_evidence[index]) {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
anchors_by_page
|
||||
.entry((page, side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
anchors_by_sequence
|
||||
.entry((side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
for anchors in anchors_by_sequence.values_mut() {
|
||||
anchors.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
}
|
||||
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if !contextual[index] || explicit_folio[index] || *value > max_plausible_folio {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
let Some(sequence_anchors) = anchors_by_sequence.get(&(!side, offset)) else {
|
||||
continue;
|
||||
};
|
||||
let neighbor = [page.checked_sub(1), page.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|neighbor_page| anchors_by_page.get(&(neighbor_page, !side, offset)))
|
||||
.flatten()
|
||||
.copied()
|
||||
.find(|&neighbor_index| {
|
||||
(items[index].y - items[neighbor_index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
&& (items[index].font_size - items[neighbor_index].font_size).abs() < 1.0
|
||||
});
|
||||
let Some(neighbor_index) = neighbor else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let first = sequence_anchors.partition_point(|&anchor_index| {
|
||||
items[anchor_index].y <= items[index].y - PAGE_NUMBER_Y_TOLERANCE
|
||||
});
|
||||
let has_second_anchor = sequence_anchors[first..]
|
||||
.iter()
|
||||
.take_while(|&&anchor_index| {
|
||||
items[anchor_index].y < items[index].y + PAGE_NUMBER_Y_TOLERANCE
|
||||
})
|
||||
.any(|&anchor_index| {
|
||||
items[anchor_index].page != page
|
||||
&& items[anchor_index].page != items[neighbor_index].page
|
||||
&& (items[index].font_size - items[anchor_index].font_size).abs() < 1.0
|
||||
});
|
||||
if has_second_anchor {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Identify page-edge numeric items that belong to a nearby content run.
|
||||
///
|
||||
/// Numeric candidates on their own do not establish context for one another.
|
||||
/// A connected same-baseline run is contextual only when it also contains a
|
||||
/// non-candidate item, preserving lines such as `Chapter 1 2026` while still
|
||||
/// removing isolated numeric footer clusters.
|
||||
fn page_number_context_masks(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
document_page_count: usize,
|
||||
) -> (Vec<bool>, Vec<bool>) {
|
||||
let mut contextual = vec![false; items.len()];
|
||||
let mut explicit_folio = vec![false; items.len()];
|
||||
let mut occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>> =
|
||||
HashMap::new();
|
||||
let mut indices_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
for (index, item) in items.iter().enumerate() {
|
||||
if matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) && !item.text.trim().is_empty()
|
||||
{
|
||||
indices_by_page.entry(item.page).or_default().push(index);
|
||||
}
|
||||
}
|
||||
for mut page_indices in indices_by_page.into_values() {
|
||||
page_indices.sort_by(|&left, &right| {
|
||||
items[right]
|
||||
.y
|
||||
.total_cmp(&items[left].y)
|
||||
.then(items[left].x.total_cmp(&items[right].x))
|
||||
});
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in page_indices {
|
||||
if rows.last().is_some_and(|row| {
|
||||
(items[row[0]].y - items[index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
}) {
|
||||
rows.last_mut().unwrap().push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
let mut start = 0;
|
||||
while start < row.len() {
|
||||
let mut end = start + 1;
|
||||
let first = &items[row[start]];
|
||||
let mut group_right = first.x + effective_width(first);
|
||||
let mut group_font_size = first.font_size;
|
||||
|
||||
while end < row.len() {
|
||||
let item = &items[row[end]];
|
||||
let gap = item.x - group_right;
|
||||
if gap > group_font_size.max(item.font_size) * PAGE_NUMBER_CONTEXT_GAP_EM {
|
||||
break;
|
||||
}
|
||||
group_right = group_right.max(item.x + effective_width(item));
|
||||
group_font_size = group_font_size.max(item.font_size);
|
||||
end += 1;
|
||||
}
|
||||
|
||||
let group = &row[start..end];
|
||||
let has_lexical_context = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.chars()
|
||||
.any(|character| character.is_alphabetic())
|
||||
});
|
||||
// Numeric data near a page edge also needs protection, but a
|
||||
// lone long integer beside a short candidate is not enough to
|
||||
// establish context. Preserve explicit numeric structures
|
||||
// (list markers, ranges, comma-formatted values, dotted index
|
||||
// entries) and dense runs with at least one long integer.
|
||||
let numeric_like = |text: &str| {
|
||||
text.chars().any(|character| character.is_numeric())
|
||||
&& !text.chars().any(|character| character.is_alphabetic())
|
||||
};
|
||||
let is_structured_numeric_context = |index: usize| {
|
||||
if candidate_values[index].is_some() {
|
||||
return false;
|
||||
}
|
||||
let text = items[index].text.trim();
|
||||
numeric_like(text)
|
||||
&& text
|
||||
.chars()
|
||||
.any(|character| !character.is_numeric() && !character.is_whitespace())
|
||||
};
|
||||
let has_structured_numeric_context = group
|
||||
.iter()
|
||||
.any(|&index| is_structured_numeric_context(index));
|
||||
let numeric_item_count = group
|
||||
.iter()
|
||||
.filter(|&&index| numeric_like(items[index].text.trim()))
|
||||
.count();
|
||||
let has_long_integer = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_numeric())
|
||||
});
|
||||
let has_dense_numeric_context = numeric_item_count >= 3 && has_long_integer;
|
||||
let has_context = has_lexical_context
|
||||
|| has_structured_numeric_context
|
||||
|| has_dense_numeric_context;
|
||||
let has_candidate = row[start..end]
|
||||
.iter()
|
||||
.any(|&index| candidate_values[index].is_some());
|
||||
// Decorative centered folios have no lexical context, so
|
||||
// recognize the complete delimiter-number-delimiter triplet
|
||||
// before the contextual-content gate. This prevents `- 42 -`
|
||||
// from leaving a malformed `- -` line.
|
||||
if group.len() == 3
|
||||
&& items[group[0]].text.trim() == "-"
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim() == "-"
|
||||
{
|
||||
for &index in group {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
if has_context && has_candidate {
|
||||
let group = &row[start..end];
|
||||
let group_text = group
|
||||
.iter()
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let group_is_folio =
|
||||
crate::text_utils::is_explicit_page_number_expression(&group_text);
|
||||
let context_text = group
|
||||
.iter()
|
||||
.filter(|&&index| candidate_values[index].is_none())
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let candidates: Vec<(usize, u32)> = group
|
||||
.iter()
|
||||
.filter_map(|&index| candidate_values[index].map(|value| (index, value)))
|
||||
.collect();
|
||||
// Recurrence is only evidence for numeric slots at the
|
||||
// outer boundary of a contextual run. An embedded number
|
||||
// in repeated prose such as `Page 42 explains the result`
|
||||
// is substantive content, not a running folio.
|
||||
let recurrence_candidates: Vec<(usize, u32)> = candidates
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(index, _)| {
|
||||
group.first() == Some(index) || group.last() == Some(index)
|
||||
})
|
||||
.collect();
|
||||
let in_deep_margin = candidates.iter().all(|(index, _)| {
|
||||
items[*index].y < PAGE_NUMBER_BOTTOM_Y
|
||||
|| items[*index].y > PAGE_NUMBER_TOP_Y
|
||||
});
|
||||
if in_deep_margin
|
||||
&& !recurrence_candidates.is_empty()
|
||||
&& context_text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphanumeric())
|
||||
.count()
|
||||
>= 8
|
||||
{
|
||||
let signature = group
|
||||
.iter()
|
||||
.map(|&index| {
|
||||
if candidate_values[index].is_some() {
|
||||
"{number}".to_string()
|
||||
} else {
|
||||
items[index]
|
||||
.text
|
||||
.split_whitespace()
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ")
|
||||
.to_lowercase()
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
occurrences_by_signature
|
||||
.entry(signature)
|
||||
.or_default()
|
||||
.push((
|
||||
items[group[0]].page,
|
||||
items[group[0]].y,
|
||||
recurrence_candidates,
|
||||
));
|
||||
}
|
||||
for (position, &index) in group.iter().enumerate() {
|
||||
if let Some(value) = candidate_values[index] {
|
||||
let adjacent_context = [position.checked_sub(1), Some(position + 1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|position| group.get(position).copied())
|
||||
.any(|adjacent| {
|
||||
candidate_values[adjacent].is_none()
|
||||
&& items[adjacent].text.chars().any(|character| {
|
||||
!character.is_numeric() && !character.is_whitespace()
|
||||
})
|
||||
});
|
||||
let max_plausible_folio =
|
||||
(document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Large year/identifier-like values stay attached
|
||||
// to their lexical run even when a smaller numeric
|
||||
// candidate sits between them and the text.
|
||||
let implausible_folio_with_lexical_context =
|
||||
value > max_plausible_folio && has_lexical_context;
|
||||
contextual[index] = adjacent_context
|
||||
|| has_dense_numeric_context
|
||||
|| implausible_folio_with_lexical_context;
|
||||
let previous = position
|
||||
.checked_sub(1)
|
||||
.map(|position| items[group[position]].text.trim());
|
||||
let next = group
|
||||
.get(position + 1)
|
||||
.map(|&index| items[index].text.trim());
|
||||
let follows_page_label =
|
||||
previous.is_some_and(|text| text.eq_ignore_ascii_case("page"));
|
||||
let starts_of_expression =
|
||||
next.is_some_and(|text| text.eq_ignore_ascii_case("of"));
|
||||
let is_centered_folio = previous == Some("-") && next == Some("-");
|
||||
explicit_folio[index] |= group_is_folio
|
||||
&& (follows_page_label
|
||||
|| starts_of_expression
|
||||
|| is_centered_folio);
|
||||
}
|
||||
}
|
||||
// Remove the complete labeled expression rather than
|
||||
// leaving fragments such as `Page of 15`. A trailing
|
||||
// running-header suffix remains untouched.
|
||||
if group_is_folio
|
||||
&& group.len() >= 4
|
||||
&& items[group[0]].text.trim().eq_ignore_ascii_case("page")
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim().eq_ignore_ascii_case("of")
|
||||
&& items[group[3]]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_ascii_digit())
|
||||
{
|
||||
for &index in &group[..4] {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mark_repeated_folio_candidates(
|
||||
occurrences_by_signature,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
mark_spread_folio_pairs(items, candidate_values, &contextual, &mut explicit_folio);
|
||||
mark_adjacent_page_folio_pairs(
|
||||
items,
|
||||
candidate_values,
|
||||
&contextual,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
|
||||
(contextual, explicit_folio)
|
||||
}
|
||||
|
||||
/// Decide which digit-only page-edge items can be removed before layout.
|
||||
///
|
||||
/// PDF producers commonly emit one text-showing operation per word. A numeric
|
||||
/// item attached to neighboring content on the same baseline is therefore kept.
|
||||
/// Complete page-number expressions such as `Page 42` remain removable even
|
||||
/// though their numeric item has lexical context.
|
||||
fn page_number_removal_mask(items: &[TextItem], document_page_count: usize) -> Vec<bool> {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, value)| explicit_folio[index] || (value.is_some() && !contextual[index]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return whether selected-page extraction contains a page-edge number whose
|
||||
/// folio status depends on evidence from other pages. Isolated and explicitly
|
||||
/// labeled folios can be decided locally; only contextual candidates require
|
||||
/// a document-wide extraction pass.
|
||||
pub(super) fn needs_document_page_number_context(
|
||||
items: &[TextItem],
|
||||
document_page_count: usize,
|
||||
) -> bool {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.any(|(index, value)| value.is_some() && contextual[index] && !explicit_folio[index])
|
||||
}
|
||||
|
||||
/// Remove numeric folios using complete document context before downstream
|
||||
/// non-table layout partitions could separate the evidence needed to recognize
|
||||
/// them.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn filter_markdown_page_numbers(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> Vec<TextItem> {
|
||||
filter_markdown_page_numbers_with_removed_pages(items, document_page_count).0
|
||||
}
|
||||
|
||||
/// Filter Markdown folios while retaining the pages where items were removed.
|
||||
///
|
||||
/// The page set lets downstream table-continuation classification preserve its
|
||||
/// pre-filter semantics even though structural layout consumes the cleaned
|
||||
/// item collection.
|
||||
pub(crate) fn filter_markdown_page_numbers_with_removed_pages(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> (Vec<TextItem>, HashSet<u32>, Vec<bool>) {
|
||||
let remove = page_number_removal_mask(&items, document_page_count as usize);
|
||||
let mut removed_pages = HashSet::new();
|
||||
let items = items
|
||||
.into_iter()
|
||||
.zip(remove.iter().copied())
|
||||
.filter_map(|(item, remove)| {
|
||||
if remove {
|
||||
removed_pages.insert(item.page);
|
||||
None
|
||||
} else {
|
||||
Some(item)
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(items, removed_pages, remove)
|
||||
}
|
||||
|
||||
/// Group text items into lines, with multi-column support
|
||||
@@ -1153,6 +1844,22 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds(items, &HashMap::new(), &HashSet::new())
|
||||
}
|
||||
|
||||
/// Group text items into lines without removing numeric page headers or footers.
|
||||
///
|
||||
/// Plain-text extraction uses this path because every extracted item is part of
|
||||
/// the API result. Markdown conversion keeps using [`group_into_lines`], where
|
||||
/// page-number suppression is an intentional presentation cleanup.
|
||||
pub fn group_into_lines_preserving_all_text(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
/// Group text items into lines, using pre-computed per-page adaptive thresholds
|
||||
/// from Canva-style letter-spacing detection. Falls back to computing the
|
||||
/// threshold from item gaps when no pre-computed value is available.
|
||||
@@ -1189,22 +1896,96 @@ pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
)
|
||||
}
|
||||
|
||||
/// Group items after document-level page-number filtering has already run.
|
||||
///
|
||||
/// Partitioned Markdown layout uses this path so a contextual candidate that
|
||||
/// was preserved with its complete baseline context is not reconsidered after
|
||||
/// its neighboring text lands in another band or chart/prose zone.
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
true,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
filter_page_numbers: bool,
|
||||
) -> Vec<TextLine> {
|
||||
if items.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Filter out page numbers (standalone numbers at top/bottom of page)
|
||||
let items: Vec<TextItem> = items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.collect();
|
||||
// Markdown output omits standalone numeric headers/footers. Determine
|
||||
// standalone status from rough baseline context before layout analysis so
|
||||
// removed page numbers cannot affect column detection. Plain-text callers
|
||||
// opt out because dropping extracted text violates that API.
|
||||
let items = if filter_page_numbers {
|
||||
// Item-only grouping has no document metadata, so use the highest
|
||||
// observed 1-based page as its best available coverage denominator.
|
||||
// The Markdown document path passes the authoritative PDF page count
|
||||
// through `filter_markdown_page_numbers` before reaching this helper.
|
||||
let observed_page_count = items
|
||||
.iter()
|
||||
.map(|item| item.page as usize)
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
let remove = page_number_removal_mask(&items, observed_page_count);
|
||||
items
|
||||
.into_iter()
|
||||
.zip(remove)
|
||||
.filter_map(|(item, remove)| (!remove).then_some(item))
|
||||
.collect()
|
||||
} else {
|
||||
items
|
||||
};
|
||||
|
||||
// Get unique pages
|
||||
let mut pages: Vec<u32> = items.iter().map(|i| i.page).collect();
|
||||
@@ -1215,6 +1996,15 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
|
||||
for page in pages {
|
||||
let page_items: Vec<TextItem> = items.iter().filter(|i| i.page == page).cloned().collect();
|
||||
// Page-edge numeric runs are weak evidence for column geometry. Keep
|
||||
// contextual values for line assembly, but prevent their preservation
|
||||
// from changing the page's inferred layout.
|
||||
let column_detection_items: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|item| page_number_value(item).is_none())
|
||||
.cloned()
|
||||
.collect();
|
||||
let column_detection_items = column_detection_items.as_slice();
|
||||
|
||||
// Use pre-computed threshold from fix_letterspaced_items if available
|
||||
// (computed before embedded-space removal, with full signal).
|
||||
@@ -1226,9 +2016,9 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
// their own positioned-region ordering and therefore stay on that path.
|
||||
if !chart_regions.contains_key(&page) {
|
||||
let preliminary_columns =
|
||||
detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
detect_columns(column_detection_items, page, table_pages.contains(&page));
|
||||
let detected_split =
|
||||
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
|
||||
(preliminary_columns.len() == 2).then(|| preliminary_columns[0].x_max);
|
||||
if let Some(band) = image_regions.get(&page).and_then(|regions| {
|
||||
super::reading_order::infer_image_anchored_flow(
|
||||
&page_items,
|
||||
@@ -1268,6 +2058,9 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
let col_input: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
if page_number_value(it).is_some() {
|
||||
return false;
|
||||
}
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Tight bounds: this only blinds the histogram to
|
||||
// chart-internal text; rows adjacent to the chart
|
||||
@@ -1280,7 +2073,7 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
.collect();
|
||||
detect_columns(&col_input, page, table_pages.contains(&page))
|
||||
}
|
||||
None => detect_columns(&page_items, page, table_pages.contains(&page)),
|
||||
None => detect_columns(column_detection_items, page, table_pages.contains(&page)),
|
||||
};
|
||||
|
||||
if columns.len() <= 1 {
|
||||
|
||||
+90
-2
@@ -153,16 +153,54 @@ pub(crate) fn extract_form_fields(
|
||||
},
|
||||
Err(_) => return items,
|
||||
};
|
||||
if fields.is_empty() {
|
||||
return items;
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(doc, field_ref, None, "", page_map, &mut items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
field_ref,
|
||||
None,
|
||||
"",
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
items
|
||||
}
|
||||
|
||||
/// Map widget annotation objects back to the page whose `/Annots` array owns
|
||||
/// them. Some valid widgets omit `/P`, so the page tree is the only reliable
|
||||
/// ownership signal available for page-filtered extraction.
|
||||
fn annotation_page_map(
|
||||
doc: &Document,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
) -> HashMap<ObjectId, u32> {
|
||||
let mut annotation_pages = HashMap::new();
|
||||
for (&page_id, &page_num) in page_map {
|
||||
let Some(annotations) = doc
|
||||
.get_dictionary(page_id)
|
||||
.ok()
|
||||
.and_then(|page| page.get(b"Annots").ok())
|
||||
.and_then(|annotations| resolve_array(doc, annotations))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
for annotation in annotations {
|
||||
if let Ok(annotation_id) = annotation.as_reference() {
|
||||
annotation_pages.insert(annotation_id, page_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
annotation_pages
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
@@ -170,6 +208,7 @@ pub(crate) fn walk_form_fields(
|
||||
parent_ft: Option<&[u8]>,
|
||||
parent_name: &str,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
) {
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
@@ -206,7 +245,15 @@ pub(crate) fn walk_form_fields(
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(doc, kid_ref, ft, &full_name, page_map, items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
kid_ref,
|
||||
ft,
|
||||
&full_name,
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -299,6 +346,7 @@ pub(crate) fn walk_form_fields(
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.and_then(|p| page_map.get(&p).copied())
|
||||
.or_else(|| annotation_pages.get(&field_id).copied())
|
||||
.unwrap_or(1);
|
||||
|
||||
let text = if full_name.is_empty() {
|
||||
@@ -324,3 +372,43 @@ pub(crate) fn walk_form_fields(
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::{dictionary, Object};
|
||||
|
||||
#[test]
|
||||
fn widget_without_page_reference_uses_owning_page_annotation() {
|
||||
let mut doc = Document::new();
|
||||
let widget_id = doc.add_object(dictionary! {
|
||||
"Type" => "Annot",
|
||||
"Subtype" => "Widget",
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("customer"),
|
||||
"V" => Object::string_literal("Alice"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
});
|
||||
let page_one_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
});
|
||||
let page_two_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Annots" => vec![Object::Reference(widget_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(widget_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::from([(page_one_id, 1), (page_two_id, 2)]);
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
}
|
||||
|
||||
+839
-18
@@ -27,12 +27,15 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
pub use layout::group_into_lines;
|
||||
#[cfg(test)]
|
||||
use layout::filter_markdown_page_numbers;
|
||||
pub(crate) use layout::filter_markdown_page_numbers_with_removed_pages;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
@@ -140,17 +143,92 @@ pub(crate) fn extract_positioned_text_from_doc(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false)
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false, None)
|
||||
}
|
||||
|
||||
/// Extract with option to include invisible (Tr=3) text.
|
||||
/// Used for Mixed/template PDFs where the OCR text layer is invisible.
|
||||
pub(crate) fn extract_positioned_text_include_invisible(
|
||||
/// Extract selected pages and gather document-wide folio evidence only when a
|
||||
/// selected page contains an ambiguous contextual page-edge number. Errors on
|
||||
/// selected pages remain fatal; errors on context-only pages are skipped.
|
||||
pub(crate) fn extract_positioned_text_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, true)
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, false)
|
||||
}
|
||||
|
||||
/// Invisible-text variant of [`extract_positioned_text_with_folio_context`].
|
||||
pub(crate) fn extract_positioned_text_include_invisible_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, true)
|
||||
}
|
||||
|
||||
fn extract_positioned_text_with_folio_context_impl(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let Some(required_pages) = page_filter else {
|
||||
return extract_positioned_text_impl(doc, font_cmaps, None, include_invisible, None);
|
||||
};
|
||||
|
||||
let (
|
||||
(mut selected_items, mut selected_rects, mut selected_lines),
|
||||
mut page_thresholds,
|
||||
mut gid_encoded_pages,
|
||||
) = extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(required_pages),
|
||||
include_invisible,
|
||||
None,
|
||||
)?;
|
||||
if !layout::needs_document_page_number_context(&selected_items, doc.get_pages().len()) {
|
||||
return Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
));
|
||||
}
|
||||
|
||||
let context_pages: HashSet<u32> = doc
|
||||
.get_pages()
|
||||
.keys()
|
||||
.copied()
|
||||
.filter(|page| !required_pages.contains(page))
|
||||
.collect();
|
||||
let ((context_items, context_rects, context_lines), context_thresholds, context_gid_pages) =
|
||||
extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(&context_pages),
|
||||
include_invisible,
|
||||
Some(required_pages),
|
||||
)?;
|
||||
selected_items.extend(context_items);
|
||||
selected_rects.extend(context_rects);
|
||||
selected_lines.extend(context_lines);
|
||||
page_thresholds.extend(context_thresholds);
|
||||
gid_encoded_pages.extend(context_gid_pages);
|
||||
Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
))
|
||||
}
|
||||
|
||||
/// Extract all pages for document-wide analysis while allowing malformed
|
||||
/// unselected pages to be skipped. Any requested page still fails normally.
|
||||
pub(crate) fn extract_positioned_text_for_document_analysis(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
required_pages: &HashSet<u32>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, None, false, Some(required_pages))
|
||||
}
|
||||
|
||||
fn extract_positioned_text_impl(
|
||||
@@ -158,6 +236,7 @@ fn extract_positioned_text_impl(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
required_pages: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let pages = doc.get_pages();
|
||||
let mut all_items = Vec::new();
|
||||
@@ -179,15 +258,25 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) =
|
||||
extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
let page_result = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
);
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) = match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error) if required_pages.is_some_and(|required| !required.contains(page_num)) => {
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
@@ -317,7 +406,9 @@ fn extract_positioned_text_impl(
|
||||
}
|
||||
|
||||
// Extract AcroForm field values
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num);
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num)
|
||||
.into_iter()
|
||||
.filter(|item| page_filter.is_none_or(|filter| filter.contains(&item.page)));
|
||||
all_items.extend(form_items);
|
||||
|
||||
Ok((
|
||||
@@ -1519,6 +1610,736 @@ mod tests {
|
||||
assert_eq!(lines[1].text(), "Next line");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn preserving_all_text_keeps_numeric_page_footer() {
|
||||
let mut page_number = make_merge_item("42", 100.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
assert!(group_into_lines(vec![page_number.clone()]).is_empty());
|
||||
|
||||
let lines = group_into_lines_preserving_all_text(vec![page_number]);
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_numeric_run_near_page_edge_is_not_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Total 730 seats");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_page_footer_separated_from_label_is_removed() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut footer_label = make_merge_item("DOCUMENT FOOTER", 60.0, 100.0);
|
||||
footer_label.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, footer_label]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "DOCUMENT FOOTER");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decorative_marker_does_not_contextualize_numeric_page_footer() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut page_number = make_merge_item("42", 37.0, 10.0);
|
||||
page_number.y = 30.0;
|
||||
let mut footer_label = make_merge_item("Company report footer", 68.0, 120.0);
|
||||
footer_label.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, page_number, footer_label]);
|
||||
|
||||
assert!(lines.iter().all(|line| !line.text().contains("42")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_is_removed_in_a_short_document() {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item("42", 57.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![label, page_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_with_running_header_suffix_is_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("of", 73.0, 12.0),
|
||||
make_merge_item("100", 89.0, 18.0),
|
||||
make_merge_item("Report header", 111.0, 78.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report header");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_of_total_expression_is_removed_without_leaving_fragments() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 482.0, 27.0),
|
||||
make_merge_item("1", 513.0, 6.0),
|
||||
make_merge_item("of", 523.0, 10.0),
|
||||
make_merge_item("15", 537.0, 12.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 46.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn document_folio_filter_survives_per_page_layout_splitting() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
items.extend([label, page_number]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 3);
|
||||
assert!(filtered
|
||||
.iter()
|
||||
.all(|item| !matches!(item.text.as_str(), "42" | "43" | "44")));
|
||||
let mut lines = Vec::new();
|
||||
for page in 1..=3 {
|
||||
let page_items = filtered
|
||||
.iter()
|
||||
.filter(|item| item.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
lines.extend(
|
||||
group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
page_items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert!(lines.iter().all(|line| line.text() == "Page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_page_edge_runs_do_not_contextualize_folios() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut long_number = make_merge_item("12345", 43.0, 30.0);
|
||||
long_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, long_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12345");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structured_and_dense_numeric_page_edge_runs_are_preserved() {
|
||||
let mut list_marker = make_merge_item("11)", 25.0, 18.0);
|
||||
list_marker.y = 50.0;
|
||||
let mut chapter = make_merge_item("13", 47.0, 12.0);
|
||||
chapter.y = 50.0;
|
||||
|
||||
let mut isbn_prefix = make_merge_item("9", 25.0, 6.0);
|
||||
isbn_prefix.page = 2;
|
||||
isbn_prefix.y = 50.0;
|
||||
let mut isbn_mid = make_merge_item("780113", 35.0, 36.0);
|
||||
isbn_mid.page = 2;
|
||||
isbn_mid.y = 50.0;
|
||||
let mut isbn_end = make_merge_item("227426", 75.0, 36.0);
|
||||
isbn_end.page = 2;
|
||||
isbn_end.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![list_marker, chapter, isbn_prefix, isbn_mid, isbn_end]);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "11) 13");
|
||||
assert_eq!(lines[1].text(), "9 780113 227426");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn incrementing_numeric_body_column_is_not_treated_as_a_folio() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "13"), (2, "14"), (3, "15")] {
|
||||
let mut row_number = make_merge_item(value, 72.0, 12.0);
|
||||
row_number.page = page;
|
||||
row_number.y = 730.0;
|
||||
let mut name = make_merge_item("Person", 90.0, 42.0);
|
||||
name.page = page;
|
||||
name.y = 730.0;
|
||||
items.extend([row_number, name]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert_eq!(lines[0].text(), "13 Person");
|
||||
assert_eq!(lines[1].text(), "14 Person");
|
||||
assert_eq!(lines[2].text(), "15 Person");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn advancing_number_in_repeated_deep_margin_footer_is_removed() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_substantive_page_number_prose_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44"), (4, "45")] {
|
||||
let mut page_label = make_merge_item("Page", 25.0, 28.0);
|
||||
page_label.page = page;
|
||||
page_label.y = 30.0;
|
||||
let mut number = make_merge_item(value, 57.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut explanation = make_merge_item("explains the result", 73.0, 108.0);
|
||||
explanation.page = page;
|
||||
explanation.y = 30.0;
|
||||
items.extend([page_label, number, explanation]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
for (line, value) in lines.iter().zip(["42", "43", "44", "45"]) {
|
||||
assert_eq!(line.text(), format!("Page {value} explains the result"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_candidates_do_not_bridge_lexical_context() {
|
||||
let mut report = make_merge_item("Report", 25.0, 40.0);
|
||||
report.y = 30.0;
|
||||
let mut year = make_merge_item("2026", 69.0, 24.0);
|
||||
year.y = 30.0;
|
||||
let mut folio = make_merge_item("42", 97.0, 12.0);
|
||||
folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![report, year, folio]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_folio_delimiters_are_removed_with_the_number() {
|
||||
let mut left = make_merge_item("-", 270.0, 6.0);
|
||||
left.y = 30.0;
|
||||
let mut number = make_merge_item("42", 280.0, 12.0);
|
||||
number.y = 30.0;
|
||||
let mut right = make_merge_item("-", 296.0, 6.0);
|
||||
right.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![left, number, right]);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_delimiters_inside_substantive_text_are_preserved() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Result", 240.0, 36.0),
|
||||
make_merge_item("-", 280.0, 6.0),
|
||||
make_merge_item("42", 290.0, 12.0),
|
||||
make_merge_item("-", 306.0, 6.0),
|
||||
make_merge_item("approved", 316.0, 48.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 30.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Result-42-approved");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn changing_year_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, year) in [(1, "2020"), (2, "2021"), (3, "2022"), (4, "2023")] {
|
||||
let mut year = make_merge_item(year, 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert_eq!(lines[0].text(), "2020 Annual report");
|
||||
assert_eq!(lines[3].text(), "2023 Annual report");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_repeated_margin_numbers_do_not_meet_the_folio_evidence_floor() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
if page <= 2 {
|
||||
let value = if page == 1 { "2" } else { "4" };
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
} else {
|
||||
let mut body = make_merge_item("Body text", 72.0, 54.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
}
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "2 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "4 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_document_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (10, "10"), (19, "19"), (28, "28")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "1 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "28 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_blank_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 20);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().any(|item| item.text == "4"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefiltered_contextual_number_survives_layout_partitioning() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 1);
|
||||
let partitioned_number: Vec<TextItem> = filtered
|
||||
.into_iter()
|
||||
.filter(|item| item.text == "730")
|
||||
.collect();
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
partitioned_number,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "730");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_partition_does_not_define_columns() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..20 {
|
||||
let y = 90.0 - row as f32 * 4.0;
|
||||
let mut left = make_merge_item(&(row + 1).to_string(), 50.0, 20.0);
|
||||
left.y = y;
|
||||
let mut right = make_merge_item(&(row + 101).to_string(), 350.0, 20.0);
|
||||
right.y = y;
|
||||
items.extend([left, right]);
|
||||
}
|
||||
assert_eq!(detect_columns(&items, 1, false).len(), 2);
|
||||
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 20);
|
||||
assert!(lines.iter().all(|line| line.items.len() == 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separated_content_is_not_treated_as_a_spread_folio_pair() {
|
||||
let mut value = make_merge_item("12", 100.0, 12.0);
|
||||
value.y = 30.0;
|
||||
let mut label = make_merge_item("Total", 116.0, 30.0);
|
||||
label.y = 30.0;
|
||||
let mut unrelated_number = make_merge_item("13", 300.0, 12.0);
|
||||
unrelated_number.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![value, label, unrelated_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12 Total");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_folio_uses_the_full_page_edge_band() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 80.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 80.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folio_on_facing_page_spread_is_removed() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut left_folio = make_merge_item("326", 35.0, 17.0);
|
||||
left_folio.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 61.0, 120.0);
|
||||
footer.y = 30.0;
|
||||
let mut right_folio = make_merge_item("327", 1148.0, 17.0);
|
||||
right_folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, left_folio, footer, right_folio]);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| !line.text().contains("326") && !line.text().contains("327")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folios_alternating_across_pages_are_removed() {
|
||||
let headers = [
|
||||
"Letter to shareholders",
|
||||
"Corporate governance report",
|
||||
"Business environment overview",
|
||||
"Consolidated financial statements",
|
||||
];
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=8 {
|
||||
let mut body = make_merge_item("Body text", 50.0, 500.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
|
||||
let mut folio = make_merge_item(&(page + 22).to_string(), 0.0, 14.0);
|
||||
folio.page = page;
|
||||
folio.y = 780.0;
|
||||
if page % 2 == 0 {
|
||||
folio.x = 50.0;
|
||||
items.push(folio);
|
||||
} else {
|
||||
let mut header = make_merge_item(headers[(page / 2) as usize], 350.0, 180.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
folio.x = 536.0;
|
||||
items.extend([header, folio]);
|
||||
}
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 8);
|
||||
|
||||
assert!(filtered.iter().all(|item| {
|
||||
!matches!(
|
||||
item.text.as_str(),
|
||||
"23" | "24" | "25" | "26" | "27" | "28" | "29" | "30"
|
||||
)
|
||||
}));
|
||||
assert!(headers
|
||||
.iter()
|
||||
.all(|header| filtered.iter().any(|item| item.text == *header)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_isolated_neighbor_does_not_remove_contextual_number() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 450.0, 70.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 526.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated = make_merge_item("2", 50.0, 7.0);
|
||||
isolated.page = 2;
|
||||
isolated.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, label, contextual, body_two, isolated], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().all(|item| item.text != "2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_content_span_does_not_establish_adjacent_page_edges() {
|
||||
let mut body_one = make_merge_item("Body text", 100.0, 120.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 170.0, 60.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 235.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated_two = make_merge_item("2", 100.0, 7.0);
|
||||
isolated_two.page = 2;
|
||||
isolated_two.y = 780.0;
|
||||
|
||||
let mut body_four = body_one.clone();
|
||||
body_four.page = 4;
|
||||
let mut isolated_four = make_merge_item("4", 100.0, 7.0);
|
||||
isolated_four.page = 4;
|
||||
isolated_four.y = 780.0;
|
||||
|
||||
let filtered = filter_markdown_page_numbers(
|
||||
vec![
|
||||
body_one,
|
||||
label,
|
||||
contextual,
|
||||
body_two,
|
||||
isolated_two,
|
||||
body_four,
|
||||
isolated_four,
|
||||
],
|
||||
4,
|
||||
);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_edge_number_on_an_adjacent_page_is_not_folio_evidence() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut isolated = make_merge_item("42", 50.0, 14.0);
|
||||
isolated.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut contextual = make_merge_item("43", 50.0, 14.0);
|
||||
contextual.page = 2;
|
||||
contextual.y = 780.0;
|
||||
let mut label = make_merge_item("cases reviewed", 70.0, 90.0);
|
||||
label.page = 2;
|
||||
label.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, isolated, body_two, contextual, label], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "43"));
|
||||
assert!(filtered.iter().any(|item| item.text == "cases reviewed"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_number_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
let mut year = make_merge_item("2026", 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines.iter().all(|line| line.text() == "2026 Annual report"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_prefix_does_not_remove_substantive_text_during_layout() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("explains", 73.0, 44.0),
|
||||
make_merge_item("the result", 121.0, 55.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42 explains the result");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_page_number_prefix_with_substantive_text_is_preserved_during_layout() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value, chapter) in [(1, "42", "Chapter 1"), (2, "43", "Chapter 2")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
let mut suffix = make_merge_item(chapter, 73.0, 58.0);
|
||||
suffix.page = page;
|
||||
suffix.y = 50.0;
|
||||
items.extend([label, page_number, suffix]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "Page 42 Chapter 1");
|
||||
assert_eq!(lines[1].text(), "Page 43 Chapter 2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_phrase_in_the_page_body_is_preserved() {
|
||||
let items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
];
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_numeric_context_near_page_edge_is_preserved() {
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
|
||||
let chapter_lines = group_into_lines(vec![chapter, chapter_number]);
|
||||
assert_eq!(chapter_lines.len(), 1);
|
||||
assert_eq!(chapter_lines[0].text(), "Chapter 1");
|
||||
|
||||
let mut year = make_merge_item("2026", 100.0, 24.0);
|
||||
year.y = 760.0;
|
||||
let mut report = make_merge_item("Report", 130.0, 36.0);
|
||||
report.y = 760.0;
|
||||
|
||||
let report_lines = group_into_lines(vec![year, report]);
|
||||
assert_eq!(report_lines.len(), 1);
|
||||
assert_eq!(report_lines[0].text(), "2026 Report");
|
||||
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
let mut edition_year = make_merge_item("2026", 163.0, 24.0);
|
||||
edition_year.y = 760.0;
|
||||
|
||||
let chained_lines = group_into_lines(vec![chapter, chapter_number, edition_year]);
|
||||
assert_eq!(chained_lines.len(), 1);
|
||||
assert_eq!(chained_lines[0].text(), "Chapter 1 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bold_italic_detection() {
|
||||
// Test bold detection
|
||||
|
||||
+385
-17
@@ -44,6 +44,16 @@ const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
/// Strikeout decorations are text-sized. Diagram connectors, signature
|
||||
/// lines, and chart rules often cross glyphs too, but extend well beyond the
|
||||
/// text they happen to intersect.
|
||||
const STRIKE_OWNER_PAD_EM: f32 = 0.75;
|
||||
const STRIKE_OWNER_MIN_PAD: f32 = 4.0;
|
||||
const STRIKE_ROW_Y_TOLERANCE_EM: f32 = 0.15;
|
||||
const STRIKE_ROW_Y_TOLERANCE_MIN: f32 = 5.0;
|
||||
const GRAPHIC_CONNECTION_EPS: f32 = 2.0;
|
||||
const GRAPHIC_CONNECTOR_MAX_THICKNESS: f32 = 4.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
@@ -387,6 +397,206 @@ fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
fn is_bare_list_marker(text: &str) -> bool {
|
||||
matches!(
|
||||
text.trim(),
|
||||
"•" | "◦" | "▪" | "▫" | "‣" | "⁃" | "●" | "○" | "■" | "□" | "-" | "*"
|
||||
)
|
||||
}
|
||||
|
||||
fn same_strike_row(left: &TextItem, right: &TextItem) -> bool {
|
||||
let font_size = left.font_size.max(right.font_size);
|
||||
let tolerance = (font_size * STRIKE_ROW_Y_TOLERANCE_EM).max(STRIKE_ROW_Y_TOLERANCE_MIN);
|
||||
(left.y - right.y).abs() <= tolerance
|
||||
}
|
||||
|
||||
fn is_inline_script(rule: &Rule, candidate: &TextItem, parent: &TextItem) -> bool {
|
||||
if !is_underline_candidate(candidate)
|
||||
|| is_bare_list_marker(&candidate.text)
|
||||
|| candidate.font_size <= 0.0
|
||||
|| candidate.font_size >= parent.font_size * 0.75
|
||||
|| candidate.text.len() > 4
|
||||
|| !candidate.text.chars().all(|c| c.is_ascii_digit())
|
||||
|| (candidate.y - parent.y).abs() > 5.0
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_ends_with_letter = parent.text.chars().last().is_some_and(char::is_alphabetic);
|
||||
if !parent_ends_with_letter {
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_right = parent.x + parent.width;
|
||||
let gap = candidate.x - parent_right;
|
||||
if gap >= parent.font_size * 0.2 || gap <= -parent.font_size * 0.3 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = rule.x2.min(candidate.x + candidate.width) - rule.x1.max(candidate.x);
|
||||
overlap >= candidate.width * MIN_X_OVERLAP
|
||||
}
|
||||
|
||||
/// Return the items owned by a snug mid-glyph rule.
|
||||
///
|
||||
/// Real strikeout decorations track the width of the deleted text, including
|
||||
/// runs split by font/style changes and adjacent numeric super/subscripts.
|
||||
/// Non-text graphics can cross the same vertical window, but arrow shafts,
|
||||
/// signature lines, fraction bars, and chart rules extend materially beyond
|
||||
/// the intersected glyphs. Requiring the rule to stay within a small em-sized
|
||||
/// pad of a contiguous matched row separates those cases without relying on
|
||||
/// document-specific fonts or coordinates.
|
||||
///
|
||||
/// Ownership is computed once per rule. This keeps the strikeout pass at the
|
||||
/// same rule-by-item scale as underline detection instead of rescanning the
|
||||
/// whole page for every matching item.
|
||||
fn snug_strike_owner_indices(rule: &Rule, items: &[TextItem]) -> Vec<usize> {
|
||||
let mut struck_indices: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
(is_underline_candidate(candidate)
|
||||
&& !is_bare_list_marker(&candidate.text)
|
||||
&& rule_strikes_item(rule, candidate))
|
||||
.then_some(index)
|
||||
})
|
||||
.collect();
|
||||
if struck_indices.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
struck_indices.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in struck_indices {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| same_strike_row(&items[row[0]], &items[index]))
|
||||
{
|
||||
row.push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
let mut owned_indices = Vec::new();
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
|
||||
// Underline detection runs before the extractor's script-merging
|
||||
// pass. Include the same tightly adjacent numeric script shape here
|
||||
// when the rule spans it, so the owner width and semantic mark both
|
||||
// survive that later merge.
|
||||
let scripts: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
let parent_pos =
|
||||
row.partition_point(|&row_index| items[row_index].x <= candidate.x);
|
||||
let parent_index = parent_pos.checked_sub(1).map(|pos| row[pos])?;
|
||||
is_inline_script(rule, candidate, &items[parent_index]).then_some(index)
|
||||
})
|
||||
.collect();
|
||||
row.extend(scripts);
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
row.dedup();
|
||||
|
||||
let x1 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x2 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x + items[index].width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let max_font_size = row
|
||||
.iter()
|
||||
.map(|&index| items[index].font_size)
|
||||
.fold(0.0, f32::max);
|
||||
let pad = (max_font_size * STRIKE_OWNER_PAD_EM).max(STRIKE_OWNER_MIN_PAD);
|
||||
|
||||
if rule.x1 < x1 - pad || rule.x2 > x2 + pad {
|
||||
continue;
|
||||
}
|
||||
|
||||
let contiguous = row.windows(2).all(|pair| {
|
||||
let gap = items[pair[1]].x - (items[pair[0]].x + items[pair[0]].width);
|
||||
gap <= (max_font_size * 2.0).max(12.0)
|
||||
});
|
||||
if contiguous {
|
||||
owned_indices.extend(row);
|
||||
}
|
||||
}
|
||||
|
||||
owned_indices.sort_unstable();
|
||||
owned_indices.dedup();
|
||||
owned_indices
|
||||
}
|
||||
|
||||
/// Diagram and table rules participate in larger path geometry. A vertical
|
||||
/// or diagonal segment meeting the candidate rule is strong evidence that
|
||||
/// the horizontal segment is a connector, border, arrow, or symbol rather
|
||||
/// than an isolated text decoration.
|
||||
fn has_connected_nonhorizontal_segment(rule: &Rule, lines: &[UnderlineLine], page: u32) -> bool {
|
||||
lines.iter().any(|line| {
|
||||
if line.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let dx = line.x2 - line.x1;
|
||||
let dy = line.y2 - line.y1;
|
||||
if dy.abs() <= MAX_RULE_THICKNESS {
|
||||
return false;
|
||||
}
|
||||
|
||||
let y_min = line.y1.min(line.y2) - GRAPHIC_CONNECTION_EPS;
|
||||
let y_max = line.y1.max(line.y2) + GRAPHIC_CONNECTION_EPS;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let t = (rule.y - line.y1) / dy;
|
||||
if !(-0.05..=1.05).contains(&t) {
|
||||
return false;
|
||||
}
|
||||
let intersection_x = line.x1 + t * dx;
|
||||
intersection_x >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& intersection_x <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Filled diagrams often build connectors from intersecting thin rectangles
|
||||
/// instead of stroked path segments. Treat only narrow, vertically elongated
|
||||
/// rectangles as connector geometry; broad fills can legitimately sit behind
|
||||
/// struck text and must not veto its decoration.
|
||||
fn has_connected_nonhorizontal_rect(rule: &Rule, rects: &[PdfRect], page: u32) -> bool {
|
||||
rects.iter().any(|rect| {
|
||||
if rect.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let (x1, x2) = if rect.width >= 0.0 {
|
||||
(rect.x, rect.x + rect.width)
|
||||
} else {
|
||||
(rect.x + rect.width, rect.x)
|
||||
};
|
||||
let (y1, y2) = if rect.height >= 0.0 {
|
||||
(rect.y, rect.y + rect.height)
|
||||
} else {
|
||||
(rect.y + rect.height, rect.y)
|
||||
};
|
||||
let width = x2 - x1;
|
||||
let height = y2 - y1;
|
||||
|
||||
width > 0.0
|
||||
&& width <= GRAPHIC_CONNECTOR_MAX_THICKNESS
|
||||
&& height > width * 2.0
|
||||
&& rule.y >= y1 - GRAPHIC_CONNECTION_EPS
|
||||
&& rule.y <= y2 + GRAPHIC_CONNECTION_EPS
|
||||
&& x2 >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& x1 <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
@@ -440,27 +650,43 @@ pub(crate) fn mark_underlined_items(
|
||||
.map(|(i, _)| i)
|
||||
.collect();
|
||||
|
||||
for item in items.iter_mut() {
|
||||
let mut strikeout_items = vec![false; items.len()];
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx)
|
||||
|| has_connected_nonhorizontal_segment(rule, lines, page)
|
||||
|| has_connected_nonhorizontal_rect(rule, rects, page)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
for item_idx in snug_strike_owner_indices(rule, items) {
|
||||
strikeout_items[item_idx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
let underlined_items: HashSet<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
is_underline_candidate(item)
|
||||
&& rules.iter().enumerate().any(|(rule_idx, rule)| {
|
||||
!tabular_rules.contains(&rule_idx)
|
||||
&& !fraction_rules.contains(&rule_idx)
|
||||
&& rule_matches_item(rule, item)
|
||||
})
|
||||
})
|
||||
.map(|(item_idx, _)| item_idx)
|
||||
.collect();
|
||||
|
||||
for (item_idx, item) in items.iter_mut().enumerate() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
// The fraction guard only gates UNDERLINE marking — a rule that
|
||||
// reads as a fraction bar from below can still legitimately
|
||||
// strike through a line above it.
|
||||
if !fraction_rules.contains(&rule_idx) && rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
if strikeout_items[item_idx] {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if underlined_items.contains(&item_idx) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -580,6 +806,148 @@ mod tests {
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_connector_crossing_text_is_not_a_strikeout() {
|
||||
let mut items = vec![item("diagram label", 160.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 280.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_rule_ending_inside_short_label_is_not_a_strikeout() {
|
||||
let mut items = vec![item("T 18", 300.0, 500.0, 20.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 315.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_diagram_segment_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let lines = vec![
|
||||
hline(99.0, 113.0, 503.0),
|
||||
UnderlineLine {
|
||||
x1: 106.0,
|
||||
y1: 496.0,
|
||||
x2: 109.0,
|
||||
y2: 510.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_filled_rect_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 14.0),
|
||||
PdfRect {
|
||||
x: 106.0,
|
||||
y: 496.0,
|
||||
width: 2.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn broad_fill_behind_text_does_not_block_strikeout() {
|
||||
let mut items = vec![item("deleted", 100.0, 500.0, 40.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 42.0),
|
||||
PdfRect {
|
||||
x: 90.0,
|
||||
y: 490.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_bullet_is_not_a_strikeout() {
|
||||
for marker in ["•", "-", "*"] {
|
||||
let mut items = vec![item(marker, 100.0, 500.0, 6.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 107.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout, "marker {marker:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_marks_adjacent_split_runs_as_strikeout() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 500.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_groups_split_runs_with_baseline_drift() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 498.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_superscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 503.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_subscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 497.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
|
||||
@@ -162,7 +162,7 @@ fn extract_form_xobject_text_inner(
|
||||
|
||||
// Get fonts from the Form's Resources
|
||||
let form_fonts = get_form_fonts(doc, &stream.dict);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
|
||||
|
||||
// Build font width info for the form
|
||||
let font_widths = build_font_widths(doc, &form_fonts);
|
||||
|
||||
+263
-32
@@ -53,8 +53,8 @@ pub use extractor::{
|
||||
extract_text_with_positions_pages,
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
MarkdownProfile,
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -68,6 +68,40 @@ use text_quality::{
|
||||
};
|
||||
use tounicode::FontCMaps;
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
struct ProcessingTimer(std::time::Instant);
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
struct ProcessingTimer;
|
||||
|
||||
impl ProcessingTimer {
|
||||
fn start() -> Self {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
Self(std::time::Instant::now())
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
Self
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(&self) -> u64 {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
self.0.elapsed().as_millis() as u64
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
// The wasm32-unknown-unknown standard library has no clock.
|
||||
// Browser bindings measure with JavaScript's host clock.
|
||||
0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reason emitted when the extracted text layer appears garbled due to
|
||||
/// broken font decoding or mojibake.
|
||||
pub const OCR_REASON_SUSPECTED_GARBLED_TEXT: &str = "suspected_garbled_text";
|
||||
@@ -250,7 +284,7 @@ pub fn process_pdf_with_options<P: AsRef<Path>>(
|
||||
path: P,
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = std::time::Instant::now();
|
||||
let start = ProcessingTimer::start();
|
||||
validate_pdf_file(&path)?;
|
||||
|
||||
// Load the document once — shared by detection AND extraction.
|
||||
@@ -277,7 +311,7 @@ pub fn process_pdf_mem_with_options(
|
||||
buffer: &[u8],
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = std::time::Instant::now();
|
||||
let start = ProcessingTimer::start();
|
||||
validate_pdf_bytes(buffer)?;
|
||||
|
||||
let (doc, page_count) =
|
||||
@@ -428,16 +462,39 @@ pub fn extract_pages_markdown_mem(
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats.
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
// unselected page cannot make a valid requested page fail, but errors on a
|
||||
// requested page retain the normal extraction semantics.
|
||||
let required_pages: Option<HashSet<u32>> = pages.map(|pages| {
|
||||
pages
|
||||
.iter()
|
||||
.filter_map(|page| page.checked_add(1))
|
||||
.collect()
|
||||
});
|
||||
let ((all_items, all_rects, all_lines), page_thresholds, gid_pages) =
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||
if let Some(required_pages) = required_pages.as_ref() {
|
||||
extractor::extract_positioned_text_for_document_analysis(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
required_pages,
|
||||
)?
|
||||
} else {
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?
|
||||
};
|
||||
let text_quality = analyze_text_quality(&all_items);
|
||||
|
||||
// Compute layout complexity from full document (near-zero cost).
|
||||
let complexity = compute_layout_complexity(&all_items, &all_rects, &all_lines);
|
||||
// Resolve page numbers with full-document context before partitioning.
|
||||
// Per-page Markdown receives the original items plus these decisions so
|
||||
// table detection can retain legitimate numeric cells.
|
||||
let (filtered_items, removed_page_number_pages, page_number_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
|
||||
// Tables need the original numeric cells; columns use folio-cleaned
|
||||
// evidence so removed page numbers cannot create false layout metadata.
|
||||
let complexity = compute_layout_complexity(&all_items, &filtered_items, &all_rects, &all_lines);
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -468,12 +525,13 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
let page_1idx = page_0idx + 1;
|
||||
|
||||
// Filter items/rects for this page only
|
||||
let page_items: Vec<TextItem> = all_items
|
||||
// Partition items, removal decisions, and rects for this page only.
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page_1idx)
|
||||
.cloned()
|
||||
.collect();
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
let page_rects: Vec<PdfRect> = all_rects
|
||||
.iter()
|
||||
@@ -500,9 +558,14 @@ pub fn extract_pages_markdown_mem(
|
||||
options,
|
||||
&page_rects,
|
||||
&[],
|
||||
&page_thresholds,
|
||||
None,
|
||||
&[],
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_page_number_pages),
|
||||
prefiltered_page_number_mask: Some(&page_number_removal_mask),
|
||||
},
|
||||
)
|
||||
};
|
||||
|
||||
@@ -1027,7 +1090,12 @@ pub fn extract_tables_in_regions_mem(
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
let detected = tables::detect_tables(&matched, base_font_size, false);
|
||||
let detected = tables::detect_tables_with_page_width(
|
||||
&matched,
|
||||
base_font_size,
|
||||
false,
|
||||
items.map_or(1.0, |items| tables::content_width(items)),
|
||||
);
|
||||
if let Some(candidate) = detected
|
||||
.iter()
|
||||
.find_map(|t| evaluate(TableCandidateSource::Heuristic, t))
|
||||
@@ -3526,7 +3594,7 @@ fn process_document(
|
||||
doc: Document,
|
||||
page_count: u32,
|
||||
options: PdfOptions,
|
||||
start: std::time::Instant,
|
||||
start: ProcessingTimer,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
// Step 1 — Detection (cheap: scans content streams for text operators)
|
||||
let detection = detector::detect_from_document(&doc, page_count, &options.detection)?;
|
||||
@@ -3542,7 +3610,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3558,7 +3626,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
@@ -3571,7 +3639,10 @@ fn process_document(
|
||||
// Step 2 — Extraction (reuses the already-loaded document)
|
||||
let extracted = {
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extractor::extract_positioned_text_from_doc(
|
||||
// Most page-filtered requests extract only the selected pages. Gather
|
||||
// other pages only when a selected contextual folio needs cross-page
|
||||
// evidence; failures on those context-only pages are non-fatal.
|
||||
let result = extractor::extract_positioned_text_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3582,9 +3653,19 @@ fn process_document(
|
||||
// This unlocks OCR text layers behind scanned images.
|
||||
if pdf_type == PdfType::Mixed {
|
||||
if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) {
|
||||
let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect();
|
||||
let sample: String = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&item.page))
|
||||
})
|
||||
.take(200)
|
||||
.map(|item| item.text.as_str())
|
||||
.collect();
|
||||
if is_garbage_text(&sample) || sample.trim().is_empty() {
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3594,7 +3675,7 @@ fn process_document(
|
||||
}
|
||||
} else {
|
||||
// Normal extraction failed — try invisible as fallback
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3658,6 +3739,13 @@ fn process_document(
|
||||
let mut garbage_pages: std::collections::HashSet<u32> =
|
||||
std::collections::HashSet::new();
|
||||
for &pg in &ocr_set {
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|filter| !filter.contains(&pg))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let page_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.page == pg)
|
||||
@@ -3697,9 +3785,38 @@ fn process_document(
|
||||
}
|
||||
};
|
||||
|
||||
let selected_page = |page: u32| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&page))
|
||||
};
|
||||
let rects: Vec<_> = rects
|
||||
.into_iter()
|
||||
.filter(|rect| selected_page(rect.page))
|
||||
.collect();
|
||||
let lines: Vec<_> = lines
|
||||
.into_iter()
|
||||
.filter(|line| selected_page(line.page))
|
||||
.collect();
|
||||
let gid_encoded_pages: HashSet<_> = gid_encoded_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
let FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
} = select_items_with_document_folio_context(
|
||||
items,
|
||||
page_count,
|
||||
options.page_filter.as_ref(),
|
||||
);
|
||||
|
||||
let text_quality = analyze_text_quality(&items);
|
||||
merge_ocr_reasons(&mut ocr_reasons_by_page, text_quality.reasons_by_page);
|
||||
let layout = compute_layout_complexity(&items, &rects, &lines);
|
||||
let layout = compute_layout_complexity(&items, &layout_items, &rects, &lines);
|
||||
|
||||
let md = if options.mode == ProcessMode::Analyze {
|
||||
None
|
||||
@@ -3709,9 +3826,14 @@ fn process_document(
|
||||
options.markdown,
|
||||
&rects,
|
||||
&lines,
|
||||
&page_thresholds,
|
||||
struct_roles.as_ref(),
|
||||
&struct_tables,
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: struct_roles.as_ref(),
|
||||
struct_tables: &struct_tables,
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(removal_mask.as_slice()),
|
||||
},
|
||||
))
|
||||
};
|
||||
|
||||
@@ -3824,7 +3946,7 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: {
|
||||
// Detector reasons (scanned / no_text / vector_text / garbled) merged
|
||||
@@ -5470,9 +5592,50 @@ mod looks_like_partial_table_tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct FolioFilteredItems {
|
||||
items: Vec<types::TextItem>,
|
||||
layout_items: Vec<types::TextItem>,
|
||||
removal_mask: Vec<bool>,
|
||||
removed_pages: HashSet<u32>,
|
||||
}
|
||||
|
||||
/// Resolve folios with complete document context, then select the caller's
|
||||
/// requested pages without losing those decisions.
|
||||
fn select_items_with_document_folio_context(
|
||||
all_items: Vec<types::TextItem>,
|
||||
page_count: u32,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> FolioFilteredItems {
|
||||
let (all_layout_items, all_removed_pages, all_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
let selected_page = |page: u32| page_filter.is_none_or(|filter| filter.contains(&page));
|
||||
|
||||
let (items, removal_mask) = all_items
|
||||
.into_iter()
|
||||
.zip(all_removal_mask)
|
||||
.filter(|(item, _)| selected_page(item.page))
|
||||
.unzip();
|
||||
let layout_items = all_layout_items
|
||||
.into_iter()
|
||||
.filter(|item| selected_page(item.page))
|
||||
.collect();
|
||||
let removed_pages = all_removed_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
|
||||
FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
}
|
||||
}
|
||||
|
||||
/// Analyse extracted items and rects for layout complexity.
|
||||
fn compute_layout_complexity(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
) -> LayoutComplexity {
|
||||
@@ -5494,6 +5657,7 @@ fn compute_layout_complexity(
|
||||
|
||||
// Check for side-by-side layout
|
||||
let owned_items: Vec<types::TextItem> = page_items.iter().map(|i| (*i).clone()).collect();
|
||||
let page_content_width = tables::content_width(&owned_items);
|
||||
let bands = markdown::split_side_by_side(&owned_items);
|
||||
|
||||
let band_ranges: Vec<(f32, f32)> = if bands.is_empty() {
|
||||
@@ -5544,7 +5708,12 @@ fn compute_layout_complexity(
|
||||
break;
|
||||
}
|
||||
// Heuristic fallback for borderless tables
|
||||
let heuristic_tables = tables::detect_tables(&band_items, base_size, false);
|
||||
let heuristic_tables = tables::detect_tables_with_page_width(
|
||||
&band_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
if has_data_table(&heuristic_tables) {
|
||||
found_table = true;
|
||||
break;
|
||||
@@ -5557,7 +5726,7 @@ fn compute_layout_complexity(
|
||||
|
||||
let mut pages_with_columns: Vec<u32> = Vec::new();
|
||||
for page in seen_pages {
|
||||
let cols = extractor::detect_columns(items, page, pages_with_tables.contains(&page));
|
||||
let cols = extractor::detect_columns(column_items, page, pages_with_tables.contains(&page));
|
||||
if cols.len() >= 2 {
|
||||
pages_with_columns.push(page);
|
||||
}
|
||||
@@ -5769,6 +5938,68 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sparse_folios_leave_no_layout_evidence() {
|
||||
let items = vec![
|
||||
test_item("1", 25.0, 20.0, 12.0, 10.0),
|
||||
test_item("2", 520.0, 60.0, 12.0, 10.0),
|
||||
];
|
||||
let (filtered, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(items.clone(), 1);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let filtered = compute_layout_complexity(&items, &filtered, &[], &[]);
|
||||
|
||||
assert!(!filtered.is_complex);
|
||||
assert!(filtered.pages_with_tables.is_empty());
|
||||
assert!(filtered.pages_with_columns.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_selection_keeps_document_wide_folio_layout_decisions() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
for row in 0..8 {
|
||||
let y = 20.0 + row as f32 * 8.0;
|
||||
let mut folio = test_item(&(row * 10 + page).to_string(), 25.0, y, 12.0, 10.0);
|
||||
folio.page = page;
|
||||
let mut footer =
|
||||
test_item(&format!("Footer row {row} summary"), 43.0, y, 470.0, 10.0);
|
||||
footer.page = page;
|
||||
let mut body = test_item(&format!("Body{row}"), 530.0, y, 55.0, 10.0);
|
||||
body.page = page;
|
||||
items.extend([folio, footer, body]);
|
||||
}
|
||||
}
|
||||
|
||||
let page_one_items: Vec<_> = items
|
||||
.iter()
|
||||
.filter(|item| item.page == 1)
|
||||
.cloned()
|
||||
.collect();
|
||||
let (page_local_layout, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(page_one_items.clone(), 4);
|
||||
let page_local = compute_layout_complexity(&page_one_items, &page_local_layout, &[], &[]);
|
||||
assert!(
|
||||
page_local.pages_with_columns.contains(&1),
|
||||
"fixture must reproduce page-local folio column evidence"
|
||||
);
|
||||
|
||||
let selected =
|
||||
select_items_with_document_folio_context(items, 4, Some(&HashSet::from([1])));
|
||||
assert_eq!(
|
||||
selected
|
||||
.removal_mask
|
||||
.iter()
|
||||
.filter(|remove| **remove)
|
||||
.count(),
|
||||
8
|
||||
);
|
||||
let document_wide =
|
||||
compute_layout_complexity(&selected.items, &selected.layout_items, &[], &[]);
|
||||
assert!(!document_wide.pages_with_columns.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_detect_encoding_issues_fffd() {
|
||||
assert!(detect_encoding_issues(
|
||||
|
||||
+165
-28
@@ -975,18 +975,54 @@ pub fn to_markdown_from_items_with_rects(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
) -> String {
|
||||
let document_page_count = items.iter().map(|item| item.page).max().unwrap_or(0);
|
||||
to_markdown_from_items_with_rects_and_page_count(items, options, rects, document_page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to Markdown with an authoritative PDF page count.
|
||||
///
|
||||
/// Use this overload when the owning PDF is available so trailing blank or
|
||||
/// unextracted pages are included in document-level header and folio coverage.
|
||||
/// Item-only callers can continue using [`to_markdown_from_items_with_rects`],
|
||||
/// which falls back to the highest observed item page.
|
||||
pub fn to_markdown_from_items_with_rects_and_page_count(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
document_page_count: u32,
|
||||
) -> String {
|
||||
to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
options,
|
||||
rects,
|
||||
&[],
|
||||
&HashMap::new(),
|
||||
None,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages: None,
|
||||
prefiltered_page_number_mask: None,
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) struct MarkdownDocumentContext<'a> {
|
||||
pub(crate) page_thresholds: &'a HashMap<u32, f32>,
|
||||
pub(crate) struct_roles:
|
||||
Option<&'a HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
pub(crate) struct_tables: &'a [crate::structure_tree::StructTable],
|
||||
pub(crate) page_count: u32,
|
||||
/// Pages where an upstream document-level pass removed folios. This keeps
|
||||
/// table-continuation classification consistent after masked items drop.
|
||||
pub(crate) prefiltered_page_number_pages: Option<&'a HashSet<u32>>,
|
||||
/// Document-level removal decisions aligned with this call's input items.
|
||||
/// Table detection consumes the original items; the mask is applied only
|
||||
/// after table claims have been established.
|
||||
pub(crate) prefiltered_page_number_mask: Option<&'a [bool]>,
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown, using rectangles and line segments for table detection.
|
||||
///
|
||||
/// Line-based detection runs first (strongest structural evidence), then rect-based,
|
||||
@@ -996,27 +1032,43 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
pdf_lines: &[crate::types::PdfLine],
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
struct_roles: Option<&HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
struct_tables: &[crate::structure_tree::StructTable],
|
||||
context: MarkdownDocumentContext<'_>,
|
||||
) -> String {
|
||||
use crate::tables::{
|
||||
detect_tables, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, try_build_rect_guided_table,
|
||||
content_width, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, detect_tables_with_page_width, try_build_rect_guided_table,
|
||||
};
|
||||
use crate::types::ItemType;
|
||||
|
||||
let MarkdownDocumentContext {
|
||||
page_thresholds,
|
||||
struct_roles,
|
||||
struct_tables,
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages,
|
||||
prefiltered_page_number_mask,
|
||||
} = context;
|
||||
|
||||
if items.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Table detection must retain the original collection because short
|
||||
// numeric table cells can be indistinguishable from folios until
|
||||
// structural context is available. A precomputed mask carries the
|
||||
// document-wide decision without removing items before table claims.
|
||||
debug_assert!(prefiltered_page_number_mask.is_none_or(|mask| mask.len() == items.len()));
|
||||
let has_precomputed_page_number_mask = prefiltered_page_number_mask.is_some();
|
||||
let removed_page_number_pages = prefiltered_page_number_pages.cloned().unwrap_or_default();
|
||||
|
||||
// Separate images and links from text items
|
||||
let mut images: Vec<TextItem> = Vec::new();
|
||||
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
let mut links: Vec<TextItem> = Vec::new();
|
||||
let mut text_items: Vec<TextItem> = Vec::new();
|
||||
let mut text_item_page_number_mask: Vec<bool> = Vec::new();
|
||||
|
||||
for item in items {
|
||||
for (input_index, item) in items.into_iter().enumerate() {
|
||||
match &item.item_type {
|
||||
ItemType::Image => {
|
||||
page_image_regions.entry(item.page).or_default().push((
|
||||
@@ -1035,6 +1087,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
ItemType::Text | ItemType::FormField => {
|
||||
text_item_page_number_mask.push(
|
||||
prefiltered_page_number_mask
|
||||
.and_then(|mask| mask.get(input_index))
|
||||
.copied()
|
||||
.unwrap_or(false),
|
||||
);
|
||||
text_items.push(item);
|
||||
}
|
||||
}
|
||||
@@ -1075,7 +1133,6 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
let mut pages: Vec<u32> = page_groups.keys().copied().collect();
|
||||
pages.sort();
|
||||
let page_count = pages.last().copied().unwrap_or(0) + 1;
|
||||
|
||||
// Track band splits per page so we can split non-table items later
|
||||
let mut page_band_splits: HashMap<u32, Vec<(f32, f32)>> = HashMap::new();
|
||||
@@ -1087,6 +1144,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
for page in pages {
|
||||
let group = page_groups.get(&page).unwrap();
|
||||
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
|
||||
let page_content_width = content_width(&page_items);
|
||||
|
||||
// Chart-bar regions: bar charts drawn as filled rects read as cell
|
||||
// rects or aligned text and get gridded into phantom tables. Their
|
||||
@@ -1128,7 +1186,10 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
});
|
||||
let chart_prose_columns = chart_prose_split.is_some();
|
||||
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
// Check for side-by-side table layout using the original items. Sparse
|
||||
// numeric cells need table context before they can be distinguished
|
||||
// safely from folios; cleaned evidence is reserved for column and
|
||||
// final non-table layout decisions.
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// A rect table crossing a proposed split boundary means the "gutter"
|
||||
// is really the gap between ruled and borderless table columns —
|
||||
@@ -1368,7 +1429,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// table can share the prose anchors. Reject only candidates
|
||||
// whose cells prove they are parallel prose fragments.
|
||||
let reject_parallel_prose = chart_prose_columns && !was_split;
|
||||
let tables = detect_tables(subset_items, base_size, false);
|
||||
let tables = detect_tables_with_page_width(
|
||||
subset_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
for table in tables {
|
||||
if reject_parallel_prose && is_parallel_prose_table(&table) {
|
||||
log::debug!(
|
||||
@@ -1549,7 +1615,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// and reject chart-page prose candidates individually below.
|
||||
let skip_body_font =
|
||||
merged_retry_skips_body_font(detected_columns, !chart_regions.is_empty());
|
||||
let heuristic_tables = detect_tables(&chart_free, base_size, skip_body_font);
|
||||
let heuristic_tables = detect_tables_with_page_width(
|
||||
&chart_free,
|
||||
base_size,
|
||||
skip_body_font,
|
||||
page_content_width,
|
||||
);
|
||||
for table in &heuristic_tables {
|
||||
if !chart_regions.is_empty() && is_parallel_prose_table(table) {
|
||||
log::debug!(
|
||||
@@ -1634,16 +1705,20 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
};
|
||||
|
||||
// Filter out table items and process the rest
|
||||
let non_table_items: Vec<TextItem> = text_items
|
||||
let non_table_items: Vec<(usize, TextItem)> = text_items
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.filter(|(idx, _)| !table_items.contains(idx))
|
||||
.map(|(_, item)| item)
|
||||
.collect();
|
||||
|
||||
// Find pages that are table-only (no remaining non-table text)
|
||||
let table_only_pages: HashSet<u32> = {
|
||||
let pages_with_text: HashSet<u32> = non_table_items.iter().map(|i| i.page).collect();
|
||||
let mut pages_with_text: HashSet<u32> =
|
||||
non_table_items.iter().map(|(_, item)| item.page).collect();
|
||||
// Preserve the pre-filter continuation classification: a page that
|
||||
// originally also contained a folio does not become table-only merely
|
||||
// because an upstream document-level pass removed it.
|
||||
pages_with_text.extend(removed_page_number_pages);
|
||||
page_tables
|
||||
.keys()
|
||||
.filter(|p| !pages_with_text.contains(p))
|
||||
@@ -1658,11 +1733,25 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// column detection on pages where table column gaps would be misidentified.
|
||||
let table_page_set: HashSet<u32> = page_tables.keys().copied().collect();
|
||||
|
||||
let non_table_items = if has_precomputed_page_number_mask {
|
||||
non_table_items
|
||||
.into_iter()
|
||||
.filter(|(index, _)| !text_item_page_number_mask[*index])
|
||||
.map(|(_, item)| item)
|
||||
.collect()
|
||||
} else {
|
||||
crate::extractor::filter_markdown_page_numbers_with_removed_pages(
|
||||
non_table_items.into_iter().map(|(_, item)| item).collect(),
|
||||
document_page_count,
|
||||
)
|
||||
.0
|
||||
};
|
||||
|
||||
// Split non-table items by band boundaries before line grouping so that
|
||||
// items from different side-by-side zones (e.g. left/right month columns
|
||||
// in a calendar) don't merge into the same line.
|
||||
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
|
||||
crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
non_table_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1690,13 +1779,14 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
// Process unsplit pages normally
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
let mut all_lines =
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
// Process each split page's bands independently, then interleave
|
||||
// by Y position so paired zones (e.g. left/right months) appear together.
|
||||
let mut split_pages: Vec<u32> = split_page_items.keys().copied().collect();
|
||||
@@ -1714,7 +1804,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !band_items.is_empty() {
|
||||
page_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
band_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1748,7 +1838,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !column_items.is_empty() {
|
||||
zone_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
column_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1789,7 +1879,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
item.y >= low || item_is_in_chart_region(item, chart_regions)
|
||||
});
|
||||
all_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
chart_zone,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1808,7 +1898,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Strip repeated headers/footers before conversion
|
||||
let lines = if options.strip_headers_footers {
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
preprocess::strip_repeated_lines(lines, document_page_count)
|
||||
} else {
|
||||
lines
|
||||
};
|
||||
@@ -1921,6 +2011,53 @@ mod tests {
|
||||
it
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn precomputed_folio_mask_preserves_numeric_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for column in 0..2 {
|
||||
let mut item = make_item_w(
|
||||
110.0 + column as f32 * 100.0,
|
||||
30.0 + row as f32 * 20.0,
|
||||
20.0,
|
||||
1,
|
||||
);
|
||||
item.text = (row * 2 + column + 1).to_string();
|
||||
items.push(item);
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + column as f32 * 100.0,
|
||||
y: 20.0 + row as f32 * 20.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Simulate document-level folio decisions that would remove every
|
||||
// short numeric item if applied before structural table detection.
|
||||
let removal_mask = vec![true; items.len()];
|
||||
let removed_pages = HashSet::from([1]);
|
||||
let markdown = to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
MarkdownOptions::default(),
|
||||
&rects,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: 1,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(&removal_mask),
|
||||
},
|
||||
);
|
||||
|
||||
assert!(markdown.contains("|1|2|"), "{markdown}");
|
||||
assert!(markdown.contains("|7|8|"), "{markdown}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn early_layout_excludes_chart_items_before_column_detection() {
|
||||
let mut items = Vec::new();
|
||||
|
||||
+20
-64
@@ -3,6 +3,7 @@
|
||||
use regex::Regex;
|
||||
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
use crate::text_utils::is_page_number_line;
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
@@ -145,7 +146,7 @@ fn fix_hyphenation(text: &str) -> String {
|
||||
result
|
||||
}
|
||||
|
||||
/// Remove standalone page numbers (lines that are just 1-4 digit numbers)
|
||||
/// Remove isolated page-number expressions from Markdown.
|
||||
fn remove_page_numbers(text: &str) -> String {
|
||||
let mut result = Vec::new();
|
||||
let lines: Vec<&str> = text.lines().collect();
|
||||
@@ -183,69 +184,6 @@ fn remove_page_numbers(text: &str) -> String {
|
||||
result.join("\n")
|
||||
}
|
||||
|
||||
/// Check if a line looks like a page number
|
||||
fn is_page_number_line(trimmed: &str) -> bool {
|
||||
// Empty lines are not page numbers
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Pattern 1: Just a number (1-4 digits)
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Pattern 2: "Page X of Y" or "Page X" or "Page of" (placeholder)
|
||||
let lower = trimmed.to_lowercase();
|
||||
if let Some(rest) = lower.strip_prefix("page") {
|
||||
let rest = rest.trim();
|
||||
// "Page of" (empty page numbers)
|
||||
if rest == "of" || rest.starts_with("of ") {
|
||||
return true;
|
||||
}
|
||||
// "Page X" or "Page X of Y"
|
||||
if rest
|
||||
.chars()
|
||||
.next()
|
||||
.map(|c| c.is_ascii_digit())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Just "Page" followed by whitespace and maybe "of"
|
||||
if rest.is_empty()
|
||||
|| rest
|
||||
.split_whitespace()
|
||||
.all(|w| w == "of" || w.chars().all(|c| c.is_ascii_digit()))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 3: "X of Y" where X and Y are numbers
|
||||
if let Some(of_idx) = trimmed.find(" of ") {
|
||||
let before = trimmed[..of_idx].trim();
|
||||
let after = trimmed[of_idx + 4..].trim();
|
||||
if before.chars().all(|c| c.is_ascii_digit())
|
||||
&& after.chars().all(|c| c.is_ascii_digit())
|
||||
&& !before.is_empty()
|
||||
&& !after.is_empty()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 4: "- X -" centered page number
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if inner.chars().all(|c| c.is_ascii_digit()) && !inner.is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Convert URLs to markdown links
|
||||
fn format_urls(text: &str) -> String {
|
||||
use once_cell::sync::Lazy;
|
||||
@@ -489,12 +427,14 @@ mod tests {
|
||||
fn test_is_page_number_page_x() {
|
||||
assert!(is_page_number_line("Page 5"));
|
||||
assert!(is_page_number_line("page 12"));
|
||||
assert!(is_page_number_line("Page123"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_page_x_of_y() {
|
||||
assert!(is_page_number_line("Page 3 of 10"));
|
||||
assert!(is_page_number_line("page 1 of 5"));
|
||||
assert!(is_page_number_line("Page 3 of 10 Report header"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -526,6 +466,12 @@ mod tests {
|
||||
assert!(!is_page_number_line("Total: 500"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_labeled_running_header() {
|
||||
assert!(is_page_number_line("Page 42 Chapter 5"));
|
||||
assert!(is_page_number_line("Page 42 explains the result"));
|
||||
}
|
||||
|
||||
// --- remove_page_numbers ---
|
||||
|
||||
#[test]
|
||||
@@ -551,6 +497,16 @@ mod tests {
|
||||
assert!(result.contains("42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_labeled_header_with_content() {
|
||||
let input = "Content\n\nPage 42 explains the result\n---\nEnd";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert!(!result.contains("Page 42 explains the result"));
|
||||
assert!(result.contains("Content"));
|
||||
assert!(result.contains("End"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_multiple_patterns() {
|
||||
let input = "\n1\n\nContent\n\n2\n\n---\nMore\n\n3\n";
|
||||
|
||||
@@ -18,7 +18,15 @@ use super::{Table, TableDetectionMode};
|
||||
///
|
||||
/// Returns `(merged_items, index_map)` where `index_map[merged_idx]` contains
|
||||
/// the original item indices that were merged into that item.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Vec<usize>>) {
|
||||
merge_adjacent_items_preserving(items, &std::collections::HashSet::new())
|
||||
}
|
||||
|
||||
fn merge_adjacent_items_preserving(
|
||||
items: &[TextItem],
|
||||
preserved_indices: &std::collections::HashSet<usize>,
|
||||
) -> (Vec<TextItem>, Vec<Vec<usize>>) {
|
||||
if items.is_empty() {
|
||||
return (vec![], vec![]);
|
||||
}
|
||||
@@ -72,6 +80,23 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
break;
|
||||
}
|
||||
|
||||
// Proven replacement cells must retain their own decoration.
|
||||
// Otherwise an adjacent old/new pair inherits only the first
|
||||
// fragment's flags and can lose the live table evidence.
|
||||
let decoration_changes = indices.iter().any(|index| {
|
||||
let merged_item = &items[*index];
|
||||
next_item.is_underline != merged_item.is_underline
|
||||
|| next_item.is_strikeout != merged_item.is_strikeout
|
||||
});
|
||||
if decoration_changes
|
||||
&& (indices
|
||||
.iter()
|
||||
.any(|index| preserved_indices.contains(index))
|
||||
|| preserved_indices.contains(&next_idx))
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
let gap = next_item.x - end_x;
|
||||
// Stop if gap exceeds threshold (inter-column gap)
|
||||
if gap > x_gap_max {
|
||||
@@ -137,17 +162,319 @@ fn expand_consolidated_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<usize>)
|
||||
(expanded, index_map)
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct RedlineEditRegion {
|
||||
x_ranges: Vec<(f32, f32)>,
|
||||
y_min: f32,
|
||||
y_max: f32,
|
||||
spans_page_width: bool,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct UnderlinedTableColumn {
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
}
|
||||
|
||||
pub(crate) fn content_width(items: &[TextItem]) -> f32 {
|
||||
let x_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + item.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
(x_max - x_min).max(1.0)
|
||||
}
|
||||
|
||||
/// Spatial regions where multiple strikeout rows indicate a redline edit block.
|
||||
///
|
||||
/// A lone deletion can occur inside or beside a real table, so it must not
|
||||
/// globally suppress underlined table cells. Closely spaced strikeout rows are
|
||||
/// different: together with nearby underlines they form the overlapping
|
||||
/// old/new text layers used by legislative redlines, and those decorations
|
||||
/// must not become heuristic column evidence.
|
||||
fn redline_edit_regions(items: &[TextItem], page_width: f32) -> Vec<RedlineEditRegion> {
|
||||
const ROW_DEDUP_TOLERANCE: f32 = 8.0;
|
||||
const MAX_CLUSTER_GAP: f32 = 64.0;
|
||||
const Y_PADDING: f32 = 36.0;
|
||||
const X_PADDING: f32 = 12.0;
|
||||
const PAGE_WIDTH_RATIO: f32 = 0.35;
|
||||
const MAX_HORIZONTAL_GAP_RATIO: f32 = 0.20;
|
||||
|
||||
let mut strikeouts: Vec<&TextItem> = items.iter().filter(|item| item.is_strikeout).collect();
|
||||
strikeouts.sort_by(|a, b| a.y.total_cmp(&b.y));
|
||||
|
||||
let mut rows: Vec<(f32, Vec<(f32, f32)>)> = Vec::new();
|
||||
for item in strikeouts {
|
||||
if let Some((_, x_ranges)) = rows
|
||||
.last_mut()
|
||||
.filter(|(y, _)| (item.y - *y).abs() <= ROW_DEDUP_TOLERANCE)
|
||||
{
|
||||
x_ranges.push((item.x, item.x + item.width));
|
||||
} else {
|
||||
rows.push((item.y, vec![(item.x, item.x + item.width)]));
|
||||
}
|
||||
}
|
||||
|
||||
let mut regions = Vec::new();
|
||||
let mut start = 0;
|
||||
while start < rows.len() {
|
||||
let mut end = start + 1;
|
||||
while end < rows.len() && rows[end].0 - rows[end - 1].0 <= MAX_CLUSTER_GAP {
|
||||
end += 1;
|
||||
}
|
||||
if end - start >= 2 {
|
||||
let mut x_ranges: Vec<(f32, f32)> = rows[start..end]
|
||||
.iter()
|
||||
.flat_map(|(_, x_ranges)| x_ranges.iter().copied())
|
||||
.collect();
|
||||
x_ranges.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
let mut merged_ranges: Vec<(f32, f32)> = Vec::new();
|
||||
for (x_min, x_max) in x_ranges {
|
||||
// Modest inline gaps can separate fragments of one flowing
|
||||
// edit; column-scale gaps must remain distinct spatial masks.
|
||||
if let Some((_, merged_max)) = merged_ranges.last_mut().filter(|(_, merged_max)| {
|
||||
x_min - *merged_max <= page_width * MAX_HORIZONTAL_GAP_RATIO
|
||||
}) {
|
||||
*merged_max = merged_max.max(x_max);
|
||||
} else {
|
||||
merged_ranges.push((x_min, x_max));
|
||||
}
|
||||
}
|
||||
let covered_width: f32 = merged_ranges
|
||||
.iter()
|
||||
.map(|(x_min, x_max)| x_max - x_min)
|
||||
.sum();
|
||||
for (x_min, x_max) in &mut merged_ranges {
|
||||
*x_min -= X_PADDING;
|
||||
*x_max += X_PADDING;
|
||||
}
|
||||
// Redlines spread across much of the text width are flowing prose,
|
||||
// so their whole Y-band is ambiguous. Compact edits can be scoped
|
||||
// to their actual horizontal spans without hiding content between
|
||||
// unrelated edits in separate columns.
|
||||
regions.push(RedlineEditRegion {
|
||||
x_ranges: merged_ranges,
|
||||
y_min: rows[start].0 - Y_PADDING,
|
||||
y_max: rows[end - 1].0 + Y_PADDING,
|
||||
spans_page_width: covered_width >= page_width * PAGE_WIDTH_RATIO,
|
||||
});
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
// Padding can make neighboring clusters overlap. Partition that overlap at
|
||||
// its midpoint so each Y position maps to one region without combining
|
||||
// horizontally unrelated edits.
|
||||
for index in 1..regions.len() {
|
||||
let (previous, current) = regions.split_at_mut(index);
|
||||
let previous = &mut previous[index - 1];
|
||||
let current = &mut current[0];
|
||||
if previous.y_max >= current.y_min {
|
||||
let boundary = (previous.y_max + current.y_min) / 2.0;
|
||||
previous.y_max = boundary;
|
||||
current.y_min = boundary;
|
||||
}
|
||||
}
|
||||
regions
|
||||
}
|
||||
|
||||
fn overlaps_redline_x(item: &TextItem, region: &RedlineEditRegion) -> bool {
|
||||
let range_index = region
|
||||
.x_ranges
|
||||
.partition_point(|(_, x_max)| *x_max < item.x);
|
||||
region
|
||||
.x_ranges
|
||||
.get(range_index)
|
||||
.is_some_and(|(x_min, _)| item.x + item.width >= *x_min)
|
||||
}
|
||||
|
||||
fn has_distinct_rows(
|
||||
items: &[&TextItem],
|
||||
underlined_only: bool,
|
||||
required: usize,
|
||||
tolerance: f32,
|
||||
) -> bool {
|
||||
debug_assert!(required <= 3);
|
||||
let mut rows = [0.0; 3];
|
||||
let mut row_count = 0;
|
||||
for item in items {
|
||||
if underlined_only && !item.is_underline {
|
||||
continue;
|
||||
}
|
||||
if rows[..row_count]
|
||||
.iter()
|
||||
.all(|row| (item.y - row).abs() > tolerance)
|
||||
{
|
||||
rows[row_count] = item.y;
|
||||
row_count += 1;
|
||||
if row_count == required {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Aligned live items seeded by underline evidence form revised table columns.
|
||||
/// Compact edits accept one replacement backed by surrounding live rows; wide
|
||||
/// prose-like edits require replacements on at least two distinct rows.
|
||||
fn underlined_table_columns(
|
||||
items: &[TextItem],
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
) -> Vec<Vec<UnderlinedTableColumn>> {
|
||||
const X_ALIGNMENT_TOLERANCE: f32 = 4.0;
|
||||
const ROW_DEDUP_TOLERANCE: f32 = 8.0;
|
||||
|
||||
// Sort once globally by X, then partition candidates into their unique Y
|
||||
// regions. Each regional vector remains X-sorted without another sort.
|
||||
let mut live_items: Vec<&TextItem> = items.iter().filter(|item| !item.is_strikeout).collect();
|
||||
live_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let mut candidates_by_region: Vec<Vec<&TextItem>> = vec![Vec::new(); redline_regions.len()];
|
||||
for item in live_items {
|
||||
if let Some(region_index) = redline_region_at_y(redline_regions, item.y) {
|
||||
if overlaps_redline_x(item, &redline_regions[region_index]) {
|
||||
candidates_by_region[region_index].push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let mut columns_by_region = Vec::with_capacity(redline_regions.len());
|
||||
for (region, candidates) in redline_regions.iter().zip(candidates_by_region) {
|
||||
let mut columns: Vec<UnderlinedTableColumn> = Vec::new();
|
||||
let mut start = 0;
|
||||
while start < candidates.len() {
|
||||
let mut end = start + 1;
|
||||
while end < candidates.len()
|
||||
&& candidates[end].x - candidates[start].x <= X_ALIGNMENT_TOLERANCE
|
||||
{
|
||||
end += 1;
|
||||
}
|
||||
let aligned_items = &candidates[start..end];
|
||||
let enough_live_rows = has_distinct_rows(aligned_items, false, 3, ROW_DEDUP_TOLERANCE);
|
||||
let required_underlined_rows = if region.spans_page_width { 2 } else { 1 };
|
||||
let enough_underlined_rows = has_distinct_rows(
|
||||
aligned_items,
|
||||
true,
|
||||
required_underlined_rows,
|
||||
ROW_DEDUP_TOLERANCE,
|
||||
);
|
||||
if enough_live_rows && enough_underlined_rows {
|
||||
let x_min = candidates[start].x - X_ALIGNMENT_TOLERANCE;
|
||||
let x_max = candidates[end - 1].x + X_ALIGNMENT_TOLERANCE;
|
||||
if let Some(column) = columns.last_mut().filter(|column| column.x_max >= x_min) {
|
||||
column.x_max = column.x_max.max(x_max);
|
||||
} else {
|
||||
columns.push(UnderlinedTableColumn { x_min, x_max });
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
columns_by_region.push(columns);
|
||||
}
|
||||
columns_by_region
|
||||
}
|
||||
|
||||
fn redline_region_at_y(redline_regions: &[RedlineEditRegion], y: f32) -> Option<usize> {
|
||||
let region_index = redline_regions.partition_point(|region| region.y_max < y);
|
||||
redline_regions
|
||||
.get(region_index)
|
||||
.filter(|region| y >= region.y_min)
|
||||
.map(|_| region_index)
|
||||
}
|
||||
|
||||
fn is_heuristic_table_evidence(
|
||||
item: &TextItem,
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
underlined_table_columns: &[Vec<UnderlinedTableColumn>],
|
||||
) -> bool {
|
||||
if item.is_strikeout {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(region_index) = redline_region_at_y(redline_regions, item.y) else {
|
||||
return true;
|
||||
};
|
||||
let region = &redline_regions[region_index];
|
||||
let columns = &underlined_table_columns[region_index];
|
||||
if region.spans_page_width && columns.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlaps_x = overlaps_redline_x(item, region);
|
||||
!overlaps_x || is_revised_table_cell(item, columns)
|
||||
}
|
||||
|
||||
fn is_revised_table_cell(
|
||||
item: &TextItem,
|
||||
underlined_table_columns: &[UnderlinedTableColumn],
|
||||
) -> bool {
|
||||
let column_index = underlined_table_columns.partition_point(|column| column.x_max < item.x);
|
||||
underlined_table_columns
|
||||
.get(column_index)
|
||||
.is_some_and(|column| item.x >= column.x_min)
|
||||
}
|
||||
|
||||
fn revised_table_cell_indices(
|
||||
items: &[TextItem],
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
underlined_table_columns: &[Vec<UnderlinedTableColumn>],
|
||||
) -> std::collections::HashSet<usize> {
|
||||
items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(item_index, item)| {
|
||||
let region_index = redline_region_at_y(redline_regions, item.y)?;
|
||||
let region = &redline_regions[region_index];
|
||||
(item.is_underline
|
||||
&& overlaps_redline_x(item, region)
|
||||
&& is_revised_table_cell(item, &underlined_table_columns[region_index]))
|
||||
.then_some(item_index)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Detect tables in a set of text items from a single page
|
||||
pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bool) -> Vec<Table> {
|
||||
detect_tables_with_page_width(items, base_font_size, skip_body_font, content_width(items))
|
||||
}
|
||||
|
||||
/// Detect tables in a subset while using the full page's text width for
|
||||
/// page-spanning redline classification.
|
||||
pub(crate) fn detect_tables_with_page_width(
|
||||
items: &[TextItem],
|
||||
base_font_size: f32,
|
||||
skip_body_font: bool,
|
||||
page_width: f32,
|
||||
) -> Vec<Table> {
|
||||
if items.len() < 6 {
|
||||
return vec![];
|
||||
}
|
||||
// Compute these before consolidation: adjacent old/new text can merge and
|
||||
// inherit only the first fragment's decoration flags.
|
||||
let redline_regions = redline_edit_regions(items, page_width);
|
||||
let underlined_table_columns = underlined_table_columns(items, &redline_regions);
|
||||
let source_evidence: Vec<bool> = items
|
||||
.iter()
|
||||
.map(|item| is_heuristic_table_evidence(item, &redline_regions, &underlined_table_columns))
|
||||
.collect();
|
||||
let revised_table_cells =
|
||||
revised_table_cell_indices(items, &redline_regions, &underlined_table_columns);
|
||||
|
||||
// Step 1: Merge adjacent single-char items into words (handles per-character PDFs)
|
||||
let (merged_items, merge_map) = merge_adjacent_items(items);
|
||||
let (merged_items, merge_map) = merge_adjacent_items_preserving(items, &revised_table_cells);
|
||||
|
||||
// Step 2: Expand consolidated financial items (e.g. "$ 1,234 $ 5,678" → sub-items)
|
||||
let (expanded_items, expand_map) = expand_consolidated_items(&merged_items);
|
||||
let expanded_evidence: Vec<bool> = expand_map
|
||||
.iter()
|
||||
.map(|&merged_index| {
|
||||
merge_map[merged_index]
|
||||
.iter()
|
||||
.all(|&source_index| source_evidence[source_index])
|
||||
})
|
||||
.collect();
|
||||
let items = &expanded_items[..]; // shadow parameter — all detection uses processed items
|
||||
|
||||
let mut tables = Vec::new();
|
||||
@@ -159,7 +486,11 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
let table_candidates: Vec<(usize, &TextItem)> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| item.font_size <= table_font_threshold && item.font_size >= 6.0)
|
||||
.filter(|(index, item)| {
|
||||
expanded_evidence[*index]
|
||||
&& item.font_size <= table_font_threshold
|
||||
&& item.font_size >= 6.0
|
||||
})
|
||||
.collect();
|
||||
|
||||
if table_candidates.len() >= 6 {
|
||||
@@ -208,6 +539,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
.enumerate()
|
||||
.filter(|(idx, item)| {
|
||||
!claimed_indices.contains(idx)
|
||||
&& expanded_evidence[*idx]
|
||||
&& item.font_size >= body_font_low
|
||||
&& item.font_size <= body_font_high
|
||||
&& item.font_size >= 6.0
|
||||
@@ -1646,6 +1978,362 @@ fn try_add_label_column(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn body_item(text: &str, x: f32, y: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y,
|
||||
width: 90.0,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_preserves_redline_boundaries() {
|
||||
let old = body_item("old value", 220.0, 700.0, true);
|
||||
let mut replacement = body_item("new value", 312.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let preserved_indices = std::collections::HashSet::from([1]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[old, replacement], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0], vec![1]]);
|
||||
assert!(merged[0].is_strikeout);
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_keeps_boundary_after_preserved_fragment() {
|
||||
let mut prefix = body_item("prefix", 20.0, 700.0, false);
|
||||
prefix.is_underline = true;
|
||||
let mut replacement = body_item("replacement", 111.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let deleted = body_item("deleted", 202.0, 700.0, true);
|
||||
let preserved_indices = std::collections::HashSet::from([1]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[prefix, replacement, deleted], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0, 1], vec![2]]);
|
||||
assert!(merged[0].is_underline);
|
||||
assert!(merged[1].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_keeps_preserved_fragment_out_of_mixed_run() {
|
||||
let mut prefix = body_item("prefix", 20.0, 700.0, false);
|
||||
prefix.is_underline = true;
|
||||
let deleted = body_item("deleted", 111.0, 700.0, true);
|
||||
let mut replacement = body_item("replacement", 202.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let preserved_indices = std::collections::HashSet::from([2]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[prefix, deleted, replacement], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0, 1], vec![2]]);
|
||||
assert!(merged[0].is_underline);
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn body_font_redline_deletions_do_not_create_a_table() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("live paragraph text", 50.0, y, false));
|
||||
items.push(body_item("deleted wording", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"aligned strikeout overlays are source edits, not table columns"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn underlined_body_font_table_without_deletions_is_still_detected() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"underline-only tables must keep their heuristic evidence"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn body_font_table_with_one_deletion_is_still_detected() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("row value", 220.0, y, row == 0));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"one revised cell must not suppress an otherwise complete table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unrelated_strikeout_does_not_remove_underlined_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
items.push(body_item("deleted prose", 50.0, 300.0, true));
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"a distant deletion must not suppress underlined table columns"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nearby_redline_rows_do_not_remove_plain_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("row value", 220.0, y, false));
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("deleted prose", 400.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"redline rows must suppress decorations, not nearby live table cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separate_strikeout_columns_do_not_suppress_content_between_them() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 160.0, y, false));
|
||||
items.push(body_item("row value", 280.0, y, false));
|
||||
}
|
||||
for (row, y) in [700.0, 684.0, 668.0, 652.0].into_iter().enumerate() {
|
||||
let x = if row % 2 == 0 { 50.0 } else { 420.0 };
|
||||
let mut deletion = body_item("old", x, y, true);
|
||||
deletion.width = 30.0;
|
||||
items.push(deletion);
|
||||
}
|
||||
|
||||
let regions = redline_edit_regions(&items, content_width(&items));
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert_eq!(regions[0].x_ranges.len(), 2);
|
||||
assert!(!regions[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"separate edit columns must not create a suppression bridge across the page"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_redline_rows_do_not_turn_fragmented_prose_into_a_table() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("line number", 50.0, y, false));
|
||||
items.push(body_item("live prose fragment", 160.0, y, false));
|
||||
let strikeout_x = if row % 2 == 0 { 280.0 } else { 430.0 };
|
||||
items.push(body_item("deleted prose", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"full-width redline prose must not retain table-shaped fragments"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_redline_prose_ignores_underlines_outside_edit_span() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
let mut line_number = body_item("line number", 50.0, y, false);
|
||||
line_number.is_underline = row < 2;
|
||||
items.push(line_number);
|
||||
items.push(body_item("live prose fragment", 160.0, y, false));
|
||||
let strikeout_x = if row % 2 == 0 { 280.0 } else { 430.0 };
|
||||
items.push(body_item("deleted prose", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"underlines outside the edit span must not disable the wide-prose veto"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nearby_redline_rows_do_not_remove_separate_underlined_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("deleted prose", 400.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"redline rows must not suppress a horizontally separate underlined table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_revised_rows_do_not_remove_underlined_table_column() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let replacement_x = 220.0 + (row % 2) as f32 * 2.0;
|
||||
let mut value = body_item("new value", replacement_x, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"aligned replacement cells must preserve a partially revised table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn single_replacement_uses_aligned_live_table_rows() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = row == 0;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"one replacement cell must retain its aligned live table column"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_revised_table_keeps_repeated_replacement_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = row < 2;
|
||||
items.push(value);
|
||||
let strikeout_x = if row % 2 == 0 { 220.0 } else { 400.0 };
|
||||
items.push(body_item("old value", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"wide edits must keep a table column with repeated replacements"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn revised_financial_columns_survive_item_expansion() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..4 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("old value", 180.0, y, true));
|
||||
let mut values = body_item("$ 100 $ 200 $ 300", 180.0, y, false);
|
||||
values.width = 300.0;
|
||||
values.is_underline = true;
|
||||
items.push(values);
|
||||
}
|
||||
|
||||
let tables = detect_tables(&items, 12.0, false);
|
||||
assert!(
|
||||
tables.iter().any(|table| table.columns.len() >= 4),
|
||||
"all expanded financial columns must inherit their source evidence"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_layout_band_uses_full_page_width_for_redline_scope() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 190.0, y, false));
|
||||
items.push(body_item("old value", 300.0, y, true));
|
||||
let mut replacement = body_item("new value", 300.0, y, false);
|
||||
replacement.is_underline = true;
|
||||
items.push(replacement);
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(!redline_edit_regions(&items, 500.0)[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables_with_page_width(&items, 12.0, false, 500.0).is_empty(),
|
||||
"a narrow band must use full-page context to preserve revised table cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adjacent_revised_fragments_preserve_live_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..4 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
let mut replacement = body_item("new value", 312.0, y, false);
|
||||
replacement.is_underline = true;
|
||||
items.push(replacement);
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"adjacent old/new fragments must retain the live revised cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_rejects_toc() {
|
||||
|
||||
+27
-1
@@ -436,7 +436,8 @@ pub(crate) fn recover_header_row(
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
item.font_size > small_font_threshold
|
||||
!item.is_strikeout
|
||||
&& item.font_size > small_font_threshold
|
||||
&& item.y > first_row_y
|
||||
&& item.y <= first_row_y + row_gap_limit
|
||||
})
|
||||
@@ -804,6 +805,31 @@ mod tests {
|
||||
assert_eq!(table.rows.len(), rows_before);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_skips_strikeout_candidates() {
|
||||
let mut old_col1 = make_item("Old Col1", 100.0, 520.0, 12.0);
|
||||
old_col1.is_strikeout = true;
|
||||
let mut old_col2 = make_item("Old Col2", 200.0, 520.0, 12.0);
|
||||
old_col2.is_strikeout = true;
|
||||
let all_items = vec![
|
||||
old_col1,
|
||||
old_col2,
|
||||
make_item("A", 100.0, 500.0, 8.0),
|
||||
make_item("B", 200.0, 500.0, 8.0),
|
||||
];
|
||||
let mut table = Table {
|
||||
columns: vec![100.0, 200.0],
|
||||
rows: vec![500.0, 480.0],
|
||||
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||
item_indices: vec![2, 3],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
recover_header_row(&mut table, &all_items, 9.0);
|
||||
assert_eq!(table.rows.len(), 2);
|
||||
assert_eq!(table.cells[0], vec!["A", "B"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_too_far_above() {
|
||||
let all_items = vec![
|
||||
|
||||
+3
-1
@@ -12,7 +12,9 @@ mod grid;
|
||||
pub mod structured;
|
||||
|
||||
pub use detect_heuristic::detect_tables;
|
||||
pub(crate) use detect_heuristic::is_table_of_contents;
|
||||
pub(crate) use detect_heuristic::{
|
||||
content_width, detect_tables_with_page_width, is_table_of_contents,
|
||||
};
|
||||
pub use detect_lines::detect_tables_from_lines;
|
||||
pub(crate) use detect_lines::detect_vector_grid_tables_from_lines;
|
||||
pub(crate) use detect_rects::cluster_rects;
|
||||
|
||||
@@ -7,6 +7,76 @@
|
||||
use crate::types::TextItem;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
/// Return whether text is an explicit page-number expression.
|
||||
///
|
||||
/// This strict form is suitable before layout, where removing one numeric item
|
||||
/// from substantive text such as `Page 42 explains the result` would lose data.
|
||||
pub(crate) fn is_explicit_page_number_expression(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let is_number = |value: &str| {
|
||||
!value.is_empty() && value.chars().all(|character| character.is_ascii_digit())
|
||||
};
|
||||
|
||||
if trimmed.len() <= 4 && is_number(trimmed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if is_number(inner) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
let lowercase = trimmed.to_ascii_lowercase();
|
||||
if let Some(rest) = lowercase.strip_prefix("page") {
|
||||
let words: Vec<&str> = rest.split_whitespace().collect();
|
||||
if words.len() >= 3 && is_number(words[0]) && words[1] == "of" && is_number(words[2]) {
|
||||
return true;
|
||||
}
|
||||
if words.len() >= 2 && words[0] == "of" && is_number(words[1]) {
|
||||
return true;
|
||||
}
|
||||
return match words.as_slice() {
|
||||
[] | ["of"] => true,
|
||||
[number] => is_number(number),
|
||||
["of", total] => is_number(total),
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
};
|
||||
}
|
||||
|
||||
let words: Vec<&str> = lowercase.split_whitespace().collect();
|
||||
match words.as_slice() {
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return whether a completed Markdown line looks like a page number or a
|
||||
/// labeled running header.
|
||||
///
|
||||
/// At this stage the complete line and surrounding breaks are available, so a
|
||||
/// leading `Page N` remains compatible with the existing header cleanup even
|
||||
/// when the PDF appends a chapter or document title.
|
||||
pub(crate) fn is_page_number_line(text: &str) -> bool {
|
||||
if is_explicit_page_number_expression(text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let lowercase = text.trim().to_ascii_lowercase();
|
||||
lowercase.strip_prefix("page").is_some_and(|rest| {
|
||||
rest.trim_start()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
/// Check if a character is CJK (Chinese, Japanese, Korean).
|
||||
/// CJK languages don't use spaces between words, so word-boundary
|
||||
/// heuristics should not apply when CJK characters are involved.
|
||||
|
||||
+23
-10
@@ -4,11 +4,17 @@
|
||||
|
||||
use log::{debug, warn};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::borrow::Cow;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::glyph_names::glyph_to_char;
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
static BUILTIN_CMAPS: include_dir::Dir<'_> =
|
||||
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
|
||||
|
||||
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ToUnicodeCMap {
|
||||
@@ -1149,9 +1155,7 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
|
||||
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
|
||||
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(name);
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&name)?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -1159,13 +1163,14 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
cmap.code_byte_length = 2;
|
||||
debug!(
|
||||
"Built-in CMap {}: char_map={} ranges={}",
|
||||
path.display(),
|
||||
name,
|
||||
cmap.char_map.len(),
|
||||
cmap.ranges.len()
|
||||
);
|
||||
Some(cmap)
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
|
||||
let p = PathBuf::from(dir);
|
||||
@@ -1182,6 +1187,18 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let path = find_bcmaps_dir()?.join(name);
|
||||
std::fs::read(path).ok().map(Cow::Owned)
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let file = BUILTIN_CMAPS.get_file(name)?;
|
||||
Some(Cow::Borrowed(file.contents()))
|
||||
}
|
||||
|
||||
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
|
||||
let mut stream = BinaryCMapStream::new(data);
|
||||
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
|
||||
@@ -1492,9 +1509,7 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
|
||||
}
|
||||
|
||||
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
parse_binary_cmap_encoding(&data).ok()
|
||||
}
|
||||
|
||||
@@ -1779,9 +1794,7 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
|
||||
if !name.ends_with("UCS2") {
|
||||
return None;
|
||||
}
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
|
||||
+113
-10
@@ -148,14 +148,17 @@ impl TextLine {
|
||||
self.text_with_formatting(false, false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
/// Get text with optional bold/italic/decorative markdown formatting.
|
||||
///
|
||||
/// `format_decorations` enables both geometrically detected source
|
||||
/// decorations: underline (`<u>`) and strikeout (`<s>`).
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_underline: bool,
|
||||
format_decorations: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
if !format_bold && !format_italic && !format_decorations {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -165,6 +168,7 @@ impl TextLine {
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
let mut current_strikeout = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -190,13 +194,16 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
// Check for style changes. Source decorations are exclusive:
|
||||
// `<u>`/`<s>` content stays free of `**`/`*` markers — consumers
|
||||
// (and the eval harnesses this feeds) match tag content literally,
|
||||
// and mixed nesting breaks that. A struck-and-underlined item is
|
||||
// emitted as struck text because deletion is the stronger semantic
|
||||
// distinction in redline documents.
|
||||
let item_strikeout = format_decorations && item.is_strikeout;
|
||||
let item_underline = format_decorations && item.is_underline && !item_strikeout;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline && !item_strikeout;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline && !item_strikeout;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -211,6 +218,10 @@ impl TextLine {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
if current_strikeout && !item_strikeout {
|
||||
result.push_str("</s>");
|
||||
current_strikeout = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -222,6 +233,10 @@ impl TextLine {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_strikeout && !current_strikeout {
|
||||
result.push_str("<s>");
|
||||
current_strikeout = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -244,6 +259,9 @@ impl TextLine {
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
if current_strikeout {
|
||||
result.push_str("</s>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
@@ -309,3 +327,88 @@ impl TextLine {
|
||||
|| space_already_exists)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod formatting_tests {
|
||||
use super::{ItemType, TextItem, TextLine};
|
||||
|
||||
fn item(text: &str, x: f32, width: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y: 100.0,
|
||||
width,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn line(items: Vec<TextItem>) -> TextLine {
|
||||
TextLine {
|
||||
items,
|
||||
y: 100.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_emits_semantic_strikeout() {
|
||||
let line = line(vec![item("deleted", 10.0, 42.0, true)]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_closes_strikeout_before_live_text() {
|
||||
let line = line(vec![
|
||||
item("keep", 10.0, 24.0, false),
|
||||
item("remove", 40.0, 42.0, true),
|
||||
item("keep", 88.0, 24.0, false),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"keep <s>remove</s> keep"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_coalesces_adjacent_struck_items() {
|
||||
let line = line(vec![
|
||||
item("deleted", 10.0, 42.0, true),
|
||||
item("words", 58.0, 30.0, true),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted words</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strikeout_takes_precedence_over_other_styles() {
|
||||
let mut decorated = item("deleted", 10.0, 42.0, true);
|
||||
decorated.is_bold = true;
|
||||
decorated.is_italic = true;
|
||||
decorated.is_underline = true;
|
||||
let line = line(vec![decorated]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
assert_eq!(line.text(), "deleted");
|
||||
}
|
||||
}
|
||||
|
||||
+276
-4
@@ -8,12 +8,13 @@ use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
process_pdf_mem, process_pdf_mem_with_options, process_pdf_with_options, to_markdown,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
@@ -40,10 +41,11 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [{media_box}] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>"
|
||||
),
|
||||
);
|
||||
|
||||
let content = "BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
@@ -79,6 +81,186 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_recurring_contextual_folio_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R 7 0 R 9 0 R] /Count 4 >>",
|
||||
);
|
||||
for page_index in 0..4 {
|
||||
let page_id = 3 + page_index * 2;
|
||||
let content_id = page_id + 1;
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
page_id,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 11 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
),
|
||||
);
|
||||
let page_number = page_index + 1;
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 1 0 0 1 25 30 Tm ({page_number}) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Body page {page_number}) Tj ET"
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
content_id,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
}
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
11,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_malformed_unselected_page() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R] /Count 2 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
let content = "BT /F1 12 Tf 1 0 0 1 25 30 Tm (1) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Selected page text) Tj 0 -16 Td (More selected text) Tj 0 -16 Td (Still selected text) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 6 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Length 3 >>\nstream\nBI \nendstream",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
make_text_pdf(
|
||||
"BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET",
|
||||
"0 0 612 792",
|
||||
)
|
||||
}
|
||||
|
||||
fn make_digit_run_repro_pdf() -> Vec<u8> {
|
||||
let content = r#"BT
|
||||
/F1 12 Tf
|
||||
1 0 0 1 72 780 Tm (A\)) Tj
|
||||
1 0 0 1 96 780 Tm (The) Tj
|
||||
1 0 0 1 126 780 Tm (total) Tj
|
||||
1 0 0 1 166 780 Tm (of) Tj
|
||||
1 0 0 1 186 780 Tm (730) Tj
|
||||
1 0 0 1 220 780 Tm (seats) Tj
|
||||
1 0 0 1 262 780 Tm (was) Tj
|
||||
1 0 0 1 296 780 Tm (approved.) Tj
|
||||
1 0 0 1 72 755 Tm (B\)) Tj
|
||||
1 0 0 1 96 755 Tm (let) Tj
|
||||
1 0 0 1 120 755 Tm (log) Tj
|
||||
1 0 0 1 150 755 Tm (2) Tj
|
||||
1 0 0 1 164 755 Tm (=) Tj
|
||||
1 0 0 1 180 755 Tm (a) Tj
|
||||
1 0 0 1 72 720 Tm (C\) Control: The total of 730 seats was approved. let log 2 = a) Tj
|
||||
ET"#;
|
||||
make_text_pdf(content, "0 0 595 842")
|
||||
}
|
||||
|
||||
fn truncate_eof_marker(mut pdf: Vec<u8>) -> Vec<u8> {
|
||||
assert!(pdf.ends_with(b"%%EOF"));
|
||||
pdf.pop();
|
||||
@@ -330,6 +512,21 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
assert_eq!(lines[0].text(), "First Second Third");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_digit_only_text_runs_are_preserved_in_markdown() {
|
||||
let pdf = make_digit_run_repro_pdf();
|
||||
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract positioned text");
|
||||
assert!(items.iter().any(|item| item.text == "730"));
|
||||
assert!(items.iter().any(|item| item.text == "2"));
|
||||
|
||||
let result = process_pdf_mem(&pdf).expect("convert PDF to markdown");
|
||||
assert_eq!(
|
||||
result.markdown.expect("markdown output").trim(),
|
||||
"A) The total of 730 seats was approved.\nB) let log 2 = a\nC) Control: The total of 730 seats was approved. let log 2 = a"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MarkdownOptions Tests
|
||||
// ============================================================================
|
||||
@@ -547,6 +744,30 @@ fn test_markdown_from_items_page_breaks() {
|
||||
assert!(md.contains("Content on second page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_page_count_overload_includes_trailing_blank_pages_in_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
items.push(make_text_item(value, 25.0, 30.0, 12.0, page));
|
||||
items.push(make_text_item(
|
||||
"Company report footer",
|
||||
41.0,
|
||||
30.0,
|
||||
12.0,
|
||||
page,
|
||||
));
|
||||
}
|
||||
let options = MarkdownOptions {
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = to_markdown_from_items_with_rects_and_page_count(items, options, &[], 20);
|
||||
|
||||
assert!(md.contains("1 Company report footer"));
|
||||
assert!(md.contains("4 Company report footer"));
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Markdown From Lines Tests
|
||||
// ============================================================================
|
||||
@@ -2915,6 +3136,57 @@ fn test_extract_pages_markdown_basic() {
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = extract_pages_markdown_mem(&pdf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 4);
|
||||
for (index, page) in result.pages.iter().enumerate() {
|
||||
assert!(page.markdown.contains("Company report footer"));
|
||||
assert!(
|
||||
!page
|
||||
.markdown
|
||||
.contains(&format!("{} Company report footer", index + 1)),
|
||||
"recurring contextual folio survived on page {}: {}",
|
||||
index + 1,
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_process_pdf_page_filter_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
|
||||
assert!(markdown.contains("Company report footer"));
|
||||
assert!(!markdown.contains("1 Company report footer"), "{markdown}");
|
||||
assert!(markdown.contains("Body page 1"));
|
||||
assert!(!markdown.contains("Body page 2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_selected_page_ignores_context_only_extraction_failure() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
let pages = extract_pages_markdown_mem(&pdf, Some(&[0])).unwrap();
|
||||
assert_eq!(pages.pages.len(), 1);
|
||||
assert!(pages.pages[0].markdown.contains("Selected page text"));
|
||||
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
assert!(markdown.contains("Selected page text"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_requested_page_extraction_failure_remains_fatal() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
assert!(extract_pages_markdown_mem(&pdf, Some(&[1])).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
@@ -56,7 +56,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
Form **4070** Employee’s Report (Rev. July 1996)
|
||||
|
||||
## of Tips to EmployerOMB No. 1545-0065
|
||||
|
||||
@@ -81,4 +81,3 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
Generated
+1304
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,37 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.3"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "README.md"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[dependencies]
|
||||
console_error_panic_hook = "0.1"
|
||||
js-sys = "0.3"
|
||||
pdf-inspector = { path = ".." }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde-wasm-bindgen = "0.6"
|
||||
wasm-bindgen = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
wasm-bindgen-test = "0.3"
|
||||
|
||||
[profile.release]
|
||||
codegen-units = 1
|
||||
lto = true
|
||||
opt-level = "s"
|
||||
strip = true
|
||||
|
||||
[package.metadata.wasm-pack.profile.release]
|
||||
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
|
||||
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
|
||||
# size, and LTO optimizations above.
|
||||
wasm-opt = false
|
||||
@@ -0,0 +1,58 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
Third-party notices
|
||||
===================
|
||||
|
||||
Adobe CMaps
|
||||
-----------
|
||||
|
||||
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
|
||||
|
||||
Copyright 1990-2009 Adobe Systems Incorporated.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
Neither the name of Adobe Systems Incorporated nor the names of its
|
||||
contributors may be used to endorse or promote products derived from this
|
||||
software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -0,0 +1,60 @@
|
||||
# @firecrawl/pdf-inspector-wasm
|
||||
|
||||
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```ts
|
||||
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
|
||||
|
||||
await init();
|
||||
|
||||
const response = await fetch("/annual-report.pdf");
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
Pass options when you need selected pages or compact Markdown:
|
||||
|
||||
```ts
|
||||
const result = processPdf(pdf, {
|
||||
pages: [1, 3, 5],
|
||||
profile: "compact",
|
||||
includePageMarkers: true,
|
||||
});
|
||||
```
|
||||
|
||||
The package also exports:
|
||||
|
||||
- `detectPdf(pdf, options?)` for detection without extraction.
|
||||
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
|
||||
- `extractText(pdf)` for plain text.
|
||||
- `version()` for the WASM package version.
|
||||
|
||||
## Browser behavior
|
||||
|
||||
- Parsing runs locally. PDF bytes are not uploaded anywhere.
|
||||
- The build is single-threaded and does not require cross-origin isolation.
|
||||
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
|
||||
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
|
||||
- Image-only documents still require a separate OCR step.
|
||||
|
||||
## Build from source
|
||||
|
||||
```bash
|
||||
cargo install wasm-pack --version 0.15.0 --locked
|
||||
wasm-pack build wasm --target web --scope firecrawl --release
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
+440
@@ -0,0 +1,440 @@
|
||||
use pdf_inspector::{
|
||||
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
|
||||
ProcessMode,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wasm_bindgen::prelude::*;
|
||||
|
||||
#[wasm_bindgen(typescript_custom_section)]
|
||||
const TYPESCRIPT_TYPES: &str = r#"
|
||||
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
|
||||
export type MarkdownProfile = "fidelity" | "compact";
|
||||
|
||||
export interface ProcessOptions {
|
||||
/** Restrict extraction to these 1-indexed page numbers. */
|
||||
pages?: number[];
|
||||
/** Password for an encrypted PDF. */
|
||||
password?: string;
|
||||
/** Source-faithful output by default, or compact output for fewer tokens. */
|
||||
profile?: MarkdownProfile;
|
||||
/** Insert `<!-- Page N -->` markers between pages. */
|
||||
includePageMarkers?: boolean;
|
||||
/** Include image placeholders in Markdown output. */
|
||||
includeImages?: boolean;
|
||||
}
|
||||
|
||||
export interface PageOcrReasons {
|
||||
/** 1-indexed page number. */
|
||||
page: number;
|
||||
reasons: string[];
|
||||
}
|
||||
|
||||
export interface LayoutComplexity {
|
||||
isComplex: boolean;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithTables: number[];
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithColumns: number[];
|
||||
}
|
||||
|
||||
export interface PdfProcessResult {
|
||||
pdfType: PdfType;
|
||||
markdown?: string;
|
||||
pageCount: number;
|
||||
processingTimeMs: number;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesNeedingOcr: number[];
|
||||
ocrReasonsByPage: PageOcrReasons[];
|
||||
title?: string;
|
||||
confidence: number;
|
||||
layout: LayoutComplexity;
|
||||
hasEncodingIssues: boolean;
|
||||
}
|
||||
|
||||
export interface PdfClassification {
|
||||
pdfType: PdfType;
|
||||
pageCount: number;
|
||||
/** 0-indexed page numbers, matching the native Node.js API. */
|
||||
pagesNeedingOcr: number[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
|
||||
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
|
||||
export function classifyPdf(data: Uint8Array): PdfClassification;
|
||||
export function extractText(data: Uint8Array): string;
|
||||
export function version(): string;
|
||||
"#;
|
||||
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
|
||||
struct WasmProcessOptions {
|
||||
pages: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
profile: Option<WasmMarkdownProfile>,
|
||||
include_page_markers: Option<bool>,
|
||||
include_images: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
enum WasmMarkdownProfile {
|
||||
Fidelity,
|
||||
Compact,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPageOcrReasons {
|
||||
page: u32,
|
||||
reasons: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<PageOcrReasons> for WasmPageOcrReasons {
|
||||
fn from(value: PageOcrReasons) -> Self {
|
||||
Self {
|
||||
page: value.page,
|
||||
reasons: value.reasons,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmLayoutComplexity {
|
||||
is_complex: bool,
|
||||
pages_with_tables: Vec<u32>,
|
||||
pages_with_columns: Vec<u32>,
|
||||
}
|
||||
|
||||
impl From<LayoutComplexity> for WasmLayoutComplexity {
|
||||
fn from(value: LayoutComplexity) -> Self {
|
||||
Self {
|
||||
is_complex: value.is_complex,
|
||||
pages_with_tables: value.pages_with_tables,
|
||||
pages_with_columns: value.pages_with_columns,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfProcessResult {
|
||||
pdf_type: &'static str,
|
||||
markdown: Option<String>,
|
||||
page_count: u32,
|
||||
processing_time_ms: f64,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
|
||||
title: Option<String>,
|
||||
confidence: f64,
|
||||
layout: WasmLayoutComplexity,
|
||||
has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
impl From<PdfProcessResult> for WasmPdfProcessResult {
|
||||
fn from(value: PdfProcessResult) -> Self {
|
||||
Self {
|
||||
pdf_type: pdf_type_name(value.pdf_type),
|
||||
markdown: value.markdown,
|
||||
page_count: value.page_count,
|
||||
processing_time_ms: value.processing_time_ms as f64,
|
||||
pages_needing_ocr: value.pages_needing_ocr,
|
||||
ocr_reasons_by_page: value
|
||||
.ocr_reasons_by_page
|
||||
.into_iter()
|
||||
.map(Into::into)
|
||||
.collect(),
|
||||
title: value.title,
|
||||
confidence: value.confidence as f64,
|
||||
layout: value.layout.into(),
|
||||
has_encoding_issues: value.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfClassification {
|
||||
pdf_type: &'static str,
|
||||
page_count: u32,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
confidence: f64,
|
||||
}
|
||||
|
||||
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
|
||||
match pdf_type {
|
||||
PdfType::TextBased => "TextBased",
|
||||
PdfType::Scanned => "Scanned",
|
||||
PdfType::ImageBased => "ImageBased",
|
||||
PdfType::Mixed => "Mixed",
|
||||
}
|
||||
}
|
||||
|
||||
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
|
||||
js_sys::Error::new(&format!("{context}: {error}")).into()
|
||||
}
|
||||
|
||||
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
|
||||
if value.is_undefined() || value.is_null() {
|
||||
return Ok(WasmProcessOptions::default());
|
||||
}
|
||||
|
||||
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
|
||||
}
|
||||
|
||||
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
|
||||
let options = deserialize_options(value)?;
|
||||
if options
|
||||
.pages
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(js_error(
|
||||
"invalid options",
|
||||
"pages are 1-indexed; page 0 is invalid",
|
||||
));
|
||||
}
|
||||
|
||||
let mut result = PdfOptions::new().mode(mode);
|
||||
if let Some(pages) = options.pages {
|
||||
result = result.pages(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(profile) = options.profile {
|
||||
result.markdown.profile = match profile {
|
||||
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
|
||||
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
|
||||
};
|
||||
}
|
||||
if let Some(include_page_markers) = options.include_page_markers {
|
||||
result.markdown.include_page_numbers = include_page_markers;
|
||||
}
|
||||
if let Some(include_images) = options.include_images {
|
||||
result.markdown.include_images = include_images;
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
|
||||
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
|
||||
}
|
||||
|
||||
fn initialize() {
|
||||
console_error_panic_hook::set_once();
|
||||
}
|
||||
|
||||
/// Process PDF bytes entirely inside WebAssembly.
|
||||
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
|
||||
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::Full)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("process PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Classify PDF bytes without extracting text or producing Markdown.
|
||||
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
|
||||
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::DetectOnly)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("detect PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Return the lightweight classification shape used by the native Node API.
|
||||
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
|
||||
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
|
||||
serialize(&WasmPdfClassification {
|
||||
pdf_type: pdf_type_name(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes without Markdown conversion.
|
||||
#[wasm_bindgen(js_name = extractText, skip_typescript)]
|
||||
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
|
||||
initialize();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
|
||||
.map_err(|error| js_error("extract text", error))?;
|
||||
Ok(
|
||||
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
|
||||
.into_iter()
|
||||
.map(|line| line.text())
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n"),
|
||||
)
|
||||
}
|
||||
|
||||
/// Return the WebAssembly package version.
|
||||
#[wasm_bindgen(skip_typescript)]
|
||||
pub fn version() -> String {
|
||||
env!("CARGO_PKG_VERSION").to_string()
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_arch = "wasm32"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use js_sys::Reflect;
|
||||
use wasm_bindgen_test::*;
|
||||
|
||||
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
|
||||
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
|
||||
|
||||
fn synthetic_korea1_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
|
||||
// There is deliberately no ToUnicode stream: decoding must use the
|
||||
// embedded predefined CMap rather than lopdf's plain-text fallback.
|
||||
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
|
||||
// the bottom edge so they look exactly like a numeric page footer.
|
||||
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn processes_pdf_to_markdown() {
|
||||
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
|
||||
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn rejects_non_pdf_bytes() {
|
||||
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn classifies_and_extracts_plain_text() {
|
||||
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
|
||||
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let text = extract_text(TEXT_PDF).expect("extract text");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!text.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn extracts_cjk_and_preserves_numeric_page_footer() {
|
||||
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
|
||||
|
||||
assert_eq!(text, "가\n42");
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn opens_encrypted_pdf_with_password() {
|
||||
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
|
||||
|
||||
let options = js_sys::Object::new();
|
||||
Reflect::set(
|
||||
&options,
|
||||
&JsValue::from_str("password"),
|
||||
&JsValue::from_str("secret123"),
|
||||
)
|
||||
.expect("set password");
|
||||
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user