Compare commits
+58
-11
@@ -14,44 +14,54 @@ jobs:
|
||||
name: Test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
|
||||
- name: Test developer scripts
|
||||
run: python3 -m unittest discover -s scripts/tests
|
||||
|
||||
fmt:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: rustfmt
|
||||
|
||||
- name: Check formatting
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Check WASM formatting
|
||||
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
|
||||
|
||||
clippy:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: clippy
|
||||
|
||||
@@ -65,15 +75,52 @@ jobs:
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: build
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
key: wasm
|
||||
|
||||
- name: Check WebAssembly bindings
|
||||
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
|
||||
|
||||
- name: Check root package for WebAssembly
|
||||
run: cargo check --target wasm32-unknown-unknown
|
||||
|
||||
- name: Lint WebAssembly bindings
|
||||
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Test WebAssembly package
|
||||
run: wasm-pack test --node --release wasm
|
||||
|
||||
@@ -24,13 +24,13 @@ jobs:
|
||||
name: github-pages
|
||||
url: ${{ steps.deploy.outputs.page_url }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Upload site artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
|
||||
with:
|
||||
path: site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deploy
|
||||
uses: actions/deploy-pages@v4
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5.0.0
|
||||
|
||||
@@ -4,6 +4,9 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['Cargo.toml']
|
||||
# Manual fallback: retry a publish that failed after the version was
|
||||
# already merged (a plain re-push won't register as a version change).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -14,13 +17,17 @@ env:
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches: crates.io trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -28,17 +35,26 @@ jobs:
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch publishes the current version regardless of the
|
||||
# previous commit; the crates.io check below still prevents
|
||||
# double-publishing an already-released version.
|
||||
echo "manual dispatch: publishing v$NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
|
||||
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
|
||||
@@ -69,17 +85,19 @@ jobs:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
uses: rust-lang/crates-io-auth-action@c6f97d42243bad5fab37ca0427f495c86d5b1a18 # v1.0.5
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
name: Publish Python package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['pyproject.toml']
|
||||
# Manual fallback: re-publish the current version without a version bump
|
||||
# (e.g. first run after PyPI trusted publishing is configured).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches too: PyPI trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("pyproject.toml").read_text())["project"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch always rebuilds and publishes. Combined with
|
||||
# skip-existing on the publish step, this repairs partial releases
|
||||
# (PyPI's version endpoint returns 200 even when only some of the
|
||||
# expected wheels were uploaded).
|
||||
echo "manual dispatch: publishing v$NEW_VERSION (skip-existing handles uploaded files)"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# .get(): the parent commit may predate the static version field
|
||||
# (pyproject.toml used dynamic = ["version"]) — treat that as a change
|
||||
# so the very first merge of this workflow publishes.
|
||||
OLD_VERSION=$(git show HEAD~1:pyproject.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["project"].get("version", ""))')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/pypi-version.json --write-out "%{http_code}" \
|
||||
"https://pypi.org/pypi/pdf-inspector/$NEW_VERSION/json")
|
||||
|
||||
case "$HTTP_STATUS" in
|
||||
200)
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
echo "pdf-inspector v$NEW_VERSION is already published to PyPI"
|
||||
;;
|
||||
404)
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
*)
|
||||
cat /tmp/pypi-version.json
|
||||
echo "Unexpected PyPI response: $HTTP_STATUS" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
build:
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
name: Build ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
# macos-13 was retired by GitHub; macos-15-intel is the remaining
|
||||
# Intel runner label (available through 2027).
|
||||
- os: macos-15-intel
|
||||
target: x86_64-apple-darwin
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Build wheel
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: wheels-${{ matrix.target }}
|
||||
path: dist/*.whl
|
||||
if-no-files-found: error
|
||||
|
||||
sdist:
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
name: Build sdist
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Build sdist
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
command: sdist
|
||||
args: --out dist
|
||||
|
||||
- name: Upload sdist
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: sdist
|
||||
path: dist/*.tar.gz
|
||||
if-no-files-found: error
|
||||
|
||||
publish:
|
||||
name: Publish to PyPI
|
||||
needs: [check-version, build, sdist]
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
|
||||
- name: List artifacts
|
||||
run: ls -la dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
with:
|
||||
packages-dir: dist
|
||||
# Tolerate already-uploaded files so a manual re-run can complete
|
||||
# a release that previously failed partway through.
|
||||
skip-existing: true
|
||||
@@ -0,0 +1,112 @@
|
||||
name: Publish WebAssembly package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['wasm/Cargo.toml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
package_exists: ${{ steps.check.outputs.package_exists }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
|
||||
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
|
||||
echo "package_exists=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
echo "The initial package must be published once before trusted publishing can be configured."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "package_exists=true" >> "$GITHUB_OUTPUT"
|
||||
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
publish:
|
||||
name: Build and publish
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Build browser package
|
||||
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
|
||||
- name: Prepare package metadata
|
||||
run: |
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const path = "wasm/pkg/package.json"
|
||||
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
|
||||
pkg.name = "@firecrawl/pdf-inspector-wasm"
|
||||
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
|
||||
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
|
||||
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
|
||||
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
|
||||
pkg.publishConfig = { access: "public" }
|
||||
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
- name: Inspect package contents
|
||||
run: npm pack --dry-run ./wasm/pkg
|
||||
|
||||
- name: Publish package
|
||||
run: npm publish ./wasm/pkg --provenance --access public
|
||||
+181
-19
@@ -4,6 +4,9 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['napi/package.json']
|
||||
# Manual fallback: retry a publish that failed partway (per-package
|
||||
# already-published checks make re-runs idempotent).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -12,12 +15,16 @@ permissions:
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches: npm trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -25,11 +32,21 @@ jobs:
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(node -p "require('./napi/package.json').version")
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch rebuilds and publishes the current version; the
|
||||
# per-package already-published checks in the publish job skip
|
||||
# anything that made it out in a previous partial run.
|
||||
echo "manual dispatch: publishing v$NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
OLD_VERSION=$(git show HEAD~1:napi/package.json | node -p "JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')).version")
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
@@ -44,31 +61,61 @@ jobs:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
# napi-cross builds gnu targets against an old glibc sysroot for
|
||||
# broad distro compatibility; musl targets cross-compile with
|
||||
# zig via cargo-zigbuild (napi's -x flag).
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
build-flags: --use-napi-cross
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: ${{ matrix.target }}
|
||||
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install zig
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: mlugg/setup-zig@d1434d08867e3ee9daa34448df10607b98908d29 # v2.2.1
|
||||
with:
|
||||
version: 0.14.1
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: taiki-e/install-action@67729d5c413db75907f0ad1e39bb04b9c868ff60 # v2.85.7
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
tool: cargo-zigbuild
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
~/.napi-rs
|
||||
napi/target/
|
||||
key: ${{ runner.os }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
key: ${{ matrix.target }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-napi-
|
||||
${{ matrix.target }}-cargo-napi-
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: napi
|
||||
@@ -76,10 +123,10 @@ jobs:
|
||||
|
||||
- name: Build native addon
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release
|
||||
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
|
||||
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
@@ -87,7 +134,7 @@ jobs:
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
@@ -95,34 +142,149 @@ jobs:
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
smoke-test:
|
||||
name: Smoke test ${{ matrix.target }}
|
||||
needs: [check-version, build]
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-gnu
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-musl
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Download native binary
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi
|
||||
|
||||
- name: Download generated JS bindings
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: napi
|
||||
|
||||
# musl binaries must load under a real musl libc, so run inside Alpine.
|
||||
- name: Run smoke test (Alpine)
|
||||
if: contains(matrix.target, 'musl')
|
||||
run: docker run --rm -v "$PWD:/repo" -w /repo/napi node:24-alpine node test.mjs
|
||||
|
||||
- name: Run smoke test
|
||||
if: ${{ !contains(matrix.target, 'musl') }}
|
||||
working-directory: napi
|
||||
run: node test.mjs
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: [check-version, build]
|
||||
needs: [check-version, build, smoke-test]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
- name: Collect binaries and publish
|
||||
- name: Publish platform packages
|
||||
working-directory: napi
|
||||
run: |
|
||||
cp artifacts/bindings-*/*.node .
|
||||
VERSION="${{ needs.check-version.outputs.version }}"
|
||||
|
||||
for node_file in artifacts/bindings-*/pdf-inspector.*.node; do
|
||||
base=$(basename "$node_file")
|
||||
suffix=${base#pdf-inspector.}
|
||||
suffix=${suffix%.node}
|
||||
pkg="@firecrawl/pdf-inspector-$suffix"
|
||||
|
||||
if npm view "$pkg@$VERSION" version >/dev/null 2>&1; then
|
||||
echo "$pkg@$VERSION already published — skipping"
|
||||
continue
|
||||
fi
|
||||
|
||||
dir="npm-dist/$suffix"
|
||||
mkdir -p "$dir"
|
||||
cp "$node_file" "$dir/"
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"linux-x64-musl": { os: ["linux"], cpu: ["x64"], libc: ["musl"] },
|
||||
"linux-arm64-gnu": { os: ["linux"], cpu: ["arm64"], libc: ["glibc"] },
|
||||
"linux-arm64-musl": { os: ["linux"], cpu: ["arm64"], libc: ["musl"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
process.exit(1)
|
||||
}
|
||||
const pkg = {
|
||||
name: `@firecrawl/pdf-inspector-${suffix}`,
|
||||
version,
|
||||
description: `Prebuilt ${suffix} binary for @firecrawl/pdf-inspector`,
|
||||
main: `pdf-inspector.${suffix}.node`,
|
||||
files: [`pdf-inspector.${suffix}.node`],
|
||||
license: "MIT",
|
||||
engines: { node: ">= 10" },
|
||||
repository: { type: "git", url: "https://github.com/firecrawl/pdf-inspector" },
|
||||
publishConfig: { access: "public" },
|
||||
...meta,
|
||||
}
|
||||
require("fs").writeFileSync(`npm-dist/${suffix}/package.json`, JSON.stringify(pkg, null, 2) + "\n")
|
||||
' "$suffix" "$VERSION"
|
||||
|
||||
echo "=== $pkg@$VERSION ==="
|
||||
ls -la "$dir"
|
||||
(cd "$dir" && npm publish --provenance --access public)
|
||||
done
|
||||
|
||||
- name: Publish main package
|
||||
working-directory: napi
|
||||
run: |
|
||||
VERSION="${{ needs.check-version.outputs.version }}"
|
||||
|
||||
if npm view "@firecrawl/pdf-inspector@$VERSION" version >/dev/null 2>&1; then
|
||||
echo "@firecrawl/pdf-inspector@$VERSION already published — skipping"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
cp artifacts/js-bindings/index.js .
|
||||
cp artifacts/js-bindings/index.d.ts .
|
||||
|
||||
echo "=== Package contents ==="
|
||||
ls -la *.node index.js index.d.ts
|
||||
# Stamp optionalDependencies to this exact version so the platform
|
||||
# pins can never drift from the main package version.
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const pkg = JSON.parse(fs.readFileSync("package.json", "utf8"))
|
||||
for (const dep of Object.keys(pkg.optionalDependencies ?? {})) {
|
||||
pkg.optionalDependencies[dep] = pkg.version
|
||||
}
|
||||
fs.writeFileSync("package.json", JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
echo "=== Main package contents ==="
|
||||
npm pack --dry-run
|
||||
|
||||
npm publish --provenance --access public
|
||||
|
||||
+3
-1
@@ -1,10 +1,13 @@
|
||||
# Rust build artifacts
|
||||
/target/
|
||||
/wasm/target/
|
||||
/wasm/pkg/
|
||||
debug/
|
||||
*.pdb
|
||||
|
||||
# Cargo lock (optional for libraries)
|
||||
Cargo.lock
|
||||
!/wasm/Cargo.lock
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
@@ -39,4 +42,3 @@ test_output/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
+28
-9
@@ -1,12 +1,25 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.4"
|
||||
version = "0.1.7"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "docs/rust-api.md"
|
||||
# Explicit allowlist: crates.io caps uploads at 10 MiB and tests/fixtures
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"/src/**",
|
||||
"/external/bcmaps/**",
|
||||
"/docs/rust-api.md",
|
||||
"/LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"/pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
name = "pdf_inspector"
|
||||
@@ -14,20 +27,13 @@ crate-type = ["lib", "cdylib"]
|
||||
|
||||
[dependencies]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
# Parallel processing
|
||||
rayon = "1.10"
|
||||
|
||||
# Logging
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Text processing
|
||||
regex = "1.10"
|
||||
@@ -37,6 +43,19 @@ unicode-normalization = "0.1"
|
||||
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
||||
ttf-parser = "0.25"
|
||||
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.42.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.42.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.3"
|
||||
|
||||
|
||||
@@ -2,9 +2,10 @@
|
||||
|
||||
[](https://crates.io/crates/pdf-inspector)
|
||||
[](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -18,24 +19,28 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only local engines without model-based PDF parsing are shown; OCR was disabled. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | 0.83 | 0.88 | 0.66 | 0.74 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the low end of that range without any OCR, in 4 seconds.
|
||||
Results were refreshed on July 31, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Speed is the median of five alternating or rotating complete corpus runs after an excluded warm-up run, with each parser processing documents sequentially in a single process.
|
||||
|
||||
**Where we do well:** Speed (fastest of all engines), the best table detection of any engine shown, and heading detection now on par with opendataloader. Overall lands within 0.01 of opendataloader at roughly 2.5× the speed.
|
||||
The complete parser configuration, per-document predictions, evaluator output, and generated charts are available in the [reproducible results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
**Where we lag:** Reading order still trails opendataloader slightly, and table structure trails OCR-based engines that can see visual layout.
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. In this comparison, pdf-inspector delivered the higher overall, reading-order, and table scores, along with the fastest complete run. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
|
||||
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
|
||||
|
||||
## Quick start
|
||||
|
||||
@@ -73,6 +78,26 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Browser WebAssembly
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
```javascript
|
||||
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
|
||||
|
||||
await init();
|
||||
const response = await fetch('/document.pdf');
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
> Full API reference: [wasm/README.md](wasm/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
@@ -118,6 +143,9 @@ pdf2md document.pdf --items-json
|
||||
# Raw markdown only (no headers)
|
||||
pdf2md document.pdf --raw
|
||||
|
||||
# Token-efficient output (collapses long dot leaders and similar source padding)
|
||||
pdf2md document.pdf --compact
|
||||
|
||||
# Insert page break markers (<!-- Page N -->)
|
||||
pdf2md document.pdf --pages
|
||||
|
||||
@@ -181,6 +209,7 @@ src/
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
wasm/ — Browser bindings (wasm-bindgen)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
# Benchmarking against OpenDataLoader
|
||||
|
||||
The paired harness runs two `pdf2md` binaries through the same local
|
||||
OpenDataLoader corpus, evaluates both outputs, and reports aggregate and
|
||||
per-document deltas. This avoids comparing results produced from different
|
||||
corpus revisions or evaluator versions.
|
||||
|
||||
Build a candidate and provide a released or worktree build as the baseline:
|
||||
|
||||
```bash
|
||||
cargo build --release
|
||||
python3 scripts/bench_opendataloader.py \
|
||||
--bench-dir ../opendataloader-bench \
|
||||
--baseline ../pdf-inspector-main/target/release/pdf2md \
|
||||
--candidate target/release/pdf2md \
|
||||
--max-document-regression 0.02 \
|
||||
--json-output /tmp/pdf-inspector-benchmark.json
|
||||
```
|
||||
|
||||
Pass `--reference-evaluation path/to/evaluation.json` to report the candidate
|
||||
delta against another evaluation, and add `--require-reference-lead` to make a
|
||||
negative reference delta fail the run. By default, the candidate must not
|
||||
regress the baseline overall score or introduce missing predictions. Use
|
||||
`--min-overall-delta` to require a specific aggregate gain.
|
||||
|
||||
The OpenDataLoader repository is external and keeps its normal
|
||||
`prediction/pdf-inspector` output. Paired evaluation copies each run into a
|
||||
temporary directory before evaluating it, so the baseline and candidate cannot
|
||||
overwrite one another.
|
||||
|
||||
## Published comparison protocol
|
||||
|
||||
The public benchmark table was refreshed on July 31, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.2.6, LiteParse 2.10.1, OpenDataLoader 2.2.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.5. Every engine processed the same 200
|
||||
PDFs sequentially in a single process with OCR disabled. Reported speed is the
|
||||
median of five alternating or rotating complete corpus runs after an excluded
|
||||
warm-up run; quality scores come from the benchmark evaluator over all 200
|
||||
outputs. Raw timings, predictions, evaluations, and charts are available in the
|
||||
[results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Optional backend evidence probe
|
||||
|
||||
The evidence probe compares positioned `pdf2md` items with MuPDF structured
|
||||
text on the same pages. It is intended to find deterministic extraction or
|
||||
layout evidence that could justify a future native implementation; it does not
|
||||
merge MuPDF output into Markdown, invoke OCR, or add a runtime dependency.
|
||||
|
||||
Install MuPDF's `mutool`, build `pdf2md`, then run:
|
||||
|
||||
```bash
|
||||
python3 scripts/probe_backend_evidence.py document.pdf \
|
||||
--pdf2md target/release/pdf2md \
|
||||
--json-output /tmp/backend-evidence.json
|
||||
```
|
||||
|
||||
The report flags pages when MuPDF exposes a material net token gain, repeated
|
||||
alignment anchors absent from local evidence, or additional image blocks. The
|
||||
JSON includes bounded token samples and page-level counts so promising cases
|
||||
can be inspected without treating backend disagreement as automatically
|
||||
correct. Thresholds are configurable with `--min-token-gain`,
|
||||
`--min-alternate-only-ratio`, and `--min-anchor-gain`.
|
||||
@@ -19,3 +19,20 @@ The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
|
||||
## Browser WebAssembly package
|
||||
|
||||
The browser package is published as `@firecrawl/pdf-inspector-wasm`. Its version lives in `wasm/Cargo.toml`, and `.github/workflows/publish-wasm.yml` builds the `web` target with `wasm-pack` before publishing the generated package.
|
||||
|
||||
The npm package must exist before a trusted publisher can be configured. For the first release only:
|
||||
|
||||
1. Build with `wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release`.
|
||||
2. Inspect with `npm pack --dry-run ./wasm/pkg`.
|
||||
3. Publish with `npm publish ./wasm/pkg --access public` from an authorized maintainer session.
|
||||
4. In the package settings on npm, configure the GitHub Actions trusted publisher:
|
||||
- Organization: `firecrawl`
|
||||
- Repository: `pdf-inspector`
|
||||
- Workflow: `publish-wasm.yml`
|
||||
- Allowed action: `npm publish`
|
||||
|
||||
After that one-time bootstrap, bumping the version in `wasm/Cargo.toml` and merging it to `main` publishes through OIDC. Until the package exists, the workflow exits cleanly without attempting an unauthenticated first publish. See npm's [trusted publishing documentation](https://docs.npmjs.com/trusted-publishers/) for the registry-side setup.
|
||||
|
||||
+75
-10
@@ -1,9 +1,39 @@
|
||||
# Python API
|
||||
# pdf-inspector
|
||||
|
||||
Python bindings via [PyO3](https://pyo3.rs). Requires Rust toolchain for building from source.
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — `text_based` / `scanned` / `image_based` / `mixed` in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core, no ML models, no external services; ships type stubs.
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt wheels cover CPython ≥3.8 on Linux (x86_64, aarch64), macOS (Intel, Apple Silicon), and Windows (x64). Other platforms build from source, which requires a Rust toolchain. For local development in a repo checkout:
|
||||
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
@@ -73,16 +103,51 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
|
||||
## Types
|
||||
|
||||
**`PdfResult` fields:** `pdf_type`, `markdown`, `page_count`, `processing_time_ms`, `pages_needing_ocr`, `title`, `confidence`, `is_complex_layout`, `pages_with_tables`, `pages_with_columns`, `has_encoding_issues`
|
||||
Type stubs (`pdf_inspector.pyi`) ship with the package. Result types at a glance:
|
||||
|
||||
**`PdfClassification` fields:** `pdf_type`, `page_count`, `pages_needing_ocr` (0-indexed), `confidence`
|
||||
```python
|
||||
class PdfResult: # process_pdf / detect_pdf
|
||||
pdf_type: str # "text_based" | "scanned" | "image_based" | "mixed"
|
||||
markdown: str | None # extracted Markdown (None for detect_pdf)
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
title: str | None
|
||||
confidence: float # 0.0 - 1.0
|
||||
is_complex_layout: bool
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool # broken font encodings — consider OCR fallback
|
||||
|
||||
**`TextItem` fields:** `text`, `x`, `y`, `width`, `height`, `font`, `font_size`, `page`, `is_bold`, `is_italic`, `item_type`
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
pages_needing_ocr: list[int] # 0-indexed
|
||||
confidence: float
|
||||
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
class TextItem: # extract_text_with_positions
|
||||
text: str
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
font: str
|
||||
font_size: float
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
class PageRegionTexts: # extract_text_in_regions
|
||||
page: int # 0-indexed
|
||||
regions: list[RegionText] # RegionText: text: str, needs_ocr: bool
|
||||
|
||||
**`PageMarkdown` fields:** `page` (0-indexed), `markdown`, `needs_ocr`
|
||||
|
||||
**`PagesExtractionResult` fields:** `pages` (list of PageMarkdown), `pages_with_tables` (1-indexed), `pages_with_columns` (1-indexed), `pages_needing_ocr` (1-indexed), `is_complex`
|
||||
class PagesExtractionResult: # extract_pages_markdown
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr
|
||||
pages_with_tables: list[int] # 1-indexed
|
||||
pages_with_columns: list[int] # 1-indexed
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
is_complex: bool # any page has tables or multi-column layout
|
||||
```
|
||||
|
||||
+40
-2
@@ -1,12 +1,50 @@
|
||||
# Rust API
|
||||
# pdf-inspector
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Pure Rust, no ML models, no external services; the only PDF dependency is [lopdf](https://crates.io/crates/lopdf). Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — TextBased / Scanned / ImageBased / Mixed in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — pure Rust, no ML models, no external services; single PDF dependency ([lopdf](https://crates.io/crates/lopdf)).
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
cargo add pdf-inspector
|
||||
```
|
||||
|
||||
For the latest unreleased changes, use the git dependency instead:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
The crate also ships CLI binaries — `pdf2md` (PDF → Markdown, with `--json`, `--pages`, `--select-pages`, and the opt-in token-saving `--compact` profile) and `detect-pdf` (classification, with `--analyze --json`):
|
||||
|
||||
```bash
|
||||
cargo install pdf-inspector
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
Generated
+25
-3
@@ -499,11 +499,13 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"js-sys",
|
||||
"libc",
|
||||
"r-efi",
|
||||
"rand_core",
|
||||
"wasip2",
|
||||
"wasip3",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -557,6 +559,25 @@ version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
|
||||
|
||||
[[package]]
|
||||
name = "include_dir"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||
dependencies = [
|
||||
"include_dir_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir_macros"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.0"
|
||||
@@ -672,9 +693,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.41.0"
|
||||
version = "0.42.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -830,9 +851,10 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.4"
|
||||
version = "0.1.7"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
"log",
|
||||
"lopdf",
|
||||
"once_cell",
|
||||
|
||||
+33
-5
@@ -4,6 +4,28 @@ Fast PDF classification and region-based text extraction for Node.js/Bun. Native
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract text from PDF structure where possible, fall back to OCR only when needed.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — text-based / scanned / image-based / mixed in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
||||
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~5–6 MB platform binary, TypeScript definitions included.
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **0.470s** |
|
||||
| liteparse | 0.873 | 0.913 | 0.693 | **0.811** | 0.750s |
|
||||
| opendataloader | 0.831 | 0.902 | 0.489 | 0.739 | 2.569s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 17.117s |
|
||||
| markitdown | 0.589 | 0.844 | 0.273 | 0.000 | 16.165s |
|
||||
|
||||
Refreshed July 31, 2026, on Apple M4 Pro; speed is the median of five complete corpus runs after an excluded warm-up. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark), with raw timings and artifacts in the [results branch](https://github.com/firecrawl/opendataloader-bench/tree/abi/pdf-parser-benchmark-results).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
@@ -12,7 +34,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -90,10 +112,16 @@ interface RegionText {
|
||||
|
||||
## Platforms
|
||||
|
||||
| Platform | Architecture | Supported |
|
||||
|----------|-------------|-----------|
|
||||
| Linux | x64 | Yes |
|
||||
| macOS | ARM64 | Yes |
|
||||
Prebuilt binaries ship as platform-specific packages installed automatically via `optionalDependencies`:
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| Linux | x64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-x64-musl` |
|
||||
| Linux | ARM64 (glibc) | `@firecrawl/pdf-inspector-linux-arm64-gnu` |
|
||||
| Linux | ARM64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-arm64-musl` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
@@ -7,6 +7,14 @@
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
"packages": {
|
||||
|
||||
+13
-6
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.10.1",
|
||||
"version": "1.12.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -22,7 +22,6 @@
|
||||
"files": [
|
||||
"index.js",
|
||||
"index.d.ts",
|
||||
"*.node",
|
||||
"bin/",
|
||||
"README.md"
|
||||
],
|
||||
@@ -38,12 +37,12 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-unknown-linux-musl",
|
||||
"aarch64-unknown-linux-gnu",
|
||||
"aarch64-unknown-linux-musl",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
],
|
||||
"package": {
|
||||
"name": "@firecrawl/pdf-inspector-js"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scripts": {
|
||||
"build": "napi build --platform --release",
|
||||
@@ -51,5 +50,13 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
|
||||
}
|
||||
}
|
||||
|
||||
+9
-3
@@ -4,10 +4,11 @@ build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
# Version is sourced from Cargo.toml [package] version by maturin so the Python
|
||||
# artifact always tracks the crate release instead of drifting on its own.
|
||||
dynamic = ["version"]
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.6"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
classifiers = [
|
||||
@@ -19,5 +20,10 @@ classifiers = [
|
||||
"Topic :: Text Processing",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
Repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
Documentation = "https://github.com/firecrawl/pdf-inspector/blob/main/docs/python.md"
|
||||
|
||||
[tool.maturin]
|
||||
features = ["python"]
|
||||
|
||||
@@ -0,0 +1,351 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run a paired pdf-inspector OpenDataLoader benchmark and report deltas."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SCORE_KEYS = (
|
||||
"overall_mean",
|
||||
"nid_mean",
|
||||
"nid_s_mean",
|
||||
"teds_mean",
|
||||
"teds_s_mean",
|
||||
"mhs_mean",
|
||||
"mhs_s_mean",
|
||||
)
|
||||
|
||||
|
||||
def _non_negative_int(value: str) -> int:
|
||||
parsed = int(value)
|
||||
if parsed < 0:
|
||||
raise argparse.ArgumentTypeError("must be non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _non_negative_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed) or parsed < 0.0:
|
||||
raise argparse.ArgumentTypeError("must be finite and non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _finite_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed):
|
||||
raise argparse.ArgumentTypeError("must be finite")
|
||||
return parsed
|
||||
|
||||
|
||||
def _scores(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
score = evaluation.get("metrics", {}).get("score", {})
|
||||
return {key: float(score[key]) for key in SCORE_KEYS if score.get(key) is not None}
|
||||
|
||||
|
||||
def _documents(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
documents: dict[str, float] = {}
|
||||
for document in evaluation.get("documents", []):
|
||||
overall = document.get("scores", {}).get("overall")
|
||||
if overall is not None:
|
||||
documents[str(document["document_id"])] = float(overall)
|
||||
return documents
|
||||
|
||||
|
||||
def compare_evaluations(
|
||||
baseline: dict[str, Any],
|
||||
candidate: dict[str, Any],
|
||||
reference: dict[str, Any] | None = None,
|
||||
*,
|
||||
top: int = 10,
|
||||
) -> dict[str, Any]:
|
||||
"""Build aggregate and per-document deltas from evaluator JSON payloads."""
|
||||
baseline_scores = _scores(baseline)
|
||||
candidate_scores = _scores(candidate)
|
||||
metric_deltas = {
|
||||
key: candidate_scores[key] - baseline_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in baseline_scores and key in candidate_scores
|
||||
}
|
||||
|
||||
baseline_documents = _documents(baseline)
|
||||
candidate_documents = _documents(candidate)
|
||||
shared = sorted(baseline_documents.keys() & candidate_documents.keys())
|
||||
document_deltas = [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"baseline": baseline_documents[document_id],
|
||||
"candidate": candidate_documents[document_id],
|
||||
"delta": candidate_documents[document_id] - baseline_documents[document_id],
|
||||
}
|
||||
for document_id in shared
|
||||
]
|
||||
epsilon = 1e-12
|
||||
improvements = sorted(document_deltas, key=lambda item: item["delta"], reverse=True)
|
||||
regressions = sorted(document_deltas, key=lambda item: item["delta"])
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"baseline": baseline_scores,
|
||||
"candidate": candidate_scores,
|
||||
"deltas": metric_deltas,
|
||||
"missing_predictions": {
|
||||
"baseline": int(baseline.get("metrics", {}).get("missing_predictions", 0)),
|
||||
"candidate": int(candidate.get("metrics", {}).get("missing_predictions", 0)),
|
||||
},
|
||||
"documents": {
|
||||
"shared": len(shared),
|
||||
"improved": sum(item["delta"] > epsilon for item in document_deltas),
|
||||
"regressed": sum(item["delta"] < -epsilon for item in document_deltas),
|
||||
"unchanged": sum(abs(item["delta"]) <= epsilon for item in document_deltas),
|
||||
"largest_improvements": [
|
||||
item for item in improvements if item["delta"] > epsilon
|
||||
][:top],
|
||||
"largest_regressions": [
|
||||
item for item in regressions if item["delta"] < -epsilon
|
||||
][:top],
|
||||
"worst_regression": next(
|
||||
(item for item in regressions if item["delta"] < -epsilon), None
|
||||
),
|
||||
},
|
||||
}
|
||||
if reference is not None:
|
||||
reference_scores = _scores(reference)
|
||||
result["reference"] = reference_scores
|
||||
result["candidate_vs_reference"] = {
|
||||
key: candidate_scores[key] - reference_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in candidate_scores and key in reference_scores
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def evaluate_gates(
|
||||
comparison: dict[str, Any],
|
||||
*,
|
||||
min_overall_delta: float,
|
||||
max_document_regression: float | None,
|
||||
max_missing: int,
|
||||
require_reference_lead: bool,
|
||||
) -> list[str]:
|
||||
"""Return human-readable gate failures; an empty list means pass."""
|
||||
failures: list[str] = []
|
||||
overall_delta = comparison["deltas"].get("overall_mean")
|
||||
if overall_delta is None or overall_delta < min_overall_delta:
|
||||
failures.append(
|
||||
f"overall delta {overall_delta!r} is below {min_overall_delta:+.6f}"
|
||||
)
|
||||
candidate_missing = comparison["missing_predictions"]["candidate"]
|
||||
if candidate_missing > max_missing:
|
||||
failures.append(
|
||||
f"candidate has {candidate_missing} missing predictions (maximum {max_missing})"
|
||||
)
|
||||
if max_document_regression is not None:
|
||||
regression = comparison["documents"].get("worst_regression")
|
||||
if regression is not None and regression["delta"] < -max_document_regression:
|
||||
failures.append(
|
||||
"largest document regression "
|
||||
f"{regression['document_id']}={regression['delta']:+.6f} "
|
||||
f"exceeds {-max_document_regression:+.6f}"
|
||||
)
|
||||
if require_reference_lead:
|
||||
reference_delta = comparison.get("candidate_vs_reference", {}).get("overall_mean")
|
||||
if reference_delta is None:
|
||||
failures.append("reference overall score is unavailable")
|
||||
elif reference_delta < 0.0:
|
||||
failures.append(
|
||||
f"candidate trails reference overall by {reference_delta!r}"
|
||||
)
|
||||
return failures
|
||||
|
||||
|
||||
def _run(command: list[str], *, cwd: Path, env: dict[str, str] | None = None) -> None:
|
||||
print("+", " ".join(command), flush=True)
|
||||
subprocess.run(command, cwd=cwd, env=env, check=True)
|
||||
|
||||
|
||||
def _run_engine(
|
||||
*,
|
||||
bench_dir: Path,
|
||||
python: Path,
|
||||
binary: Path,
|
||||
label: str,
|
||||
scratch_root: Path,
|
||||
) -> dict[str, Any]:
|
||||
env = os.environ.copy()
|
||||
env["PDF_INSPECTOR_BINARY"] = str(binary)
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
if source.exists():
|
||||
if source.is_dir():
|
||||
shutil.rmtree(source)
|
||||
else:
|
||||
source.unlink()
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/pdf_parser.py",
|
||||
"--engine",
|
||||
"pdf-inspector",
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
env=env,
|
||||
)
|
||||
|
||||
if not source.is_dir():
|
||||
raise RuntimeError(f"parser did not produce predictions: {source}")
|
||||
destination = scratch_root / label
|
||||
shutil.copytree(source, destination)
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/evaluator.py",
|
||||
"--prediction-root",
|
||||
str(scratch_root),
|
||||
"--engine",
|
||||
label,
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
)
|
||||
with (destination / "evaluation.json").open(encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def _print_report(comparison: dict[str, Any]) -> None:
|
||||
print("\nMetric baseline candidate delta")
|
||||
print("-------------------- ---------- ---------- ----------")
|
||||
for key in SCORE_KEYS:
|
||||
if key not in comparison["deltas"]:
|
||||
continue
|
||||
print(
|
||||
f"{key:<20} {comparison['baseline'][key]:>10.6f} "
|
||||
f"{comparison['candidate'][key]:>10.6f} "
|
||||
f"{comparison['deltas'][key]:>+10.6f}"
|
||||
)
|
||||
if "reference" in comparison:
|
||||
delta = comparison["candidate_vs_reference"].get("overall_mean")
|
||||
reference = comparison["reference"].get("overall_mean")
|
||||
reference_display = f"{reference:.6f}" if reference is not None else "n/a"
|
||||
delta_display = f"{delta:+.6f}" if delta is not None else "n/a"
|
||||
print(f"\nReference overall: {reference_display}; candidate delta: {delta_display}")
|
||||
|
||||
documents = comparison["documents"]
|
||||
print(
|
||||
"\nDocuments: "
|
||||
f"{documents['improved']} improved, {documents['regressed']} regressed, "
|
||||
f"{documents['unchanged']} unchanged ({documents['shared']} shared)"
|
||||
)
|
||||
for heading, key in (
|
||||
("Largest improvements", "largest_improvements"),
|
||||
("Largest regressions", "largest_regressions"),
|
||||
):
|
||||
print(f"\n{heading}:")
|
||||
rows = documents[key]
|
||||
if not rows:
|
||||
print(" none")
|
||||
for row in rows:
|
||||
print(
|
||||
f" {row['document_id']}: {row['delta']:+.6f} "
|
||||
f"({row['baseline']:.6f} -> {row['candidate']:.6f})"
|
||||
)
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--bench-dir", type=Path, required=True)
|
||||
parser.add_argument("--baseline", type=Path, required=True)
|
||||
parser.add_argument("--candidate", type=Path, required=True)
|
||||
parser.add_argument("--python", type=Path)
|
||||
parser.add_argument("--reference-evaluation", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--top", type=_non_negative_int, default=10)
|
||||
parser.add_argument("--min-overall-delta", type=_finite_float, default=0.0)
|
||||
parser.add_argument("--max-document-regression", type=_non_negative_float)
|
||||
parser.add_argument("--max-missing", type=_non_negative_int, default=0)
|
||||
parser.add_argument("--require-reference-lead", action="store_true")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
bench_dir = args.bench_dir.resolve()
|
||||
baseline = args.baseline.resolve()
|
||||
candidate = args.candidate.resolve()
|
||||
# Keep the virtualenv launcher path intact. Resolving its symlink would
|
||||
# invoke the underlying system interpreter without the benchmark's site
|
||||
# packages.
|
||||
python = (args.python or bench_dir / ".venv" / "bin" / "python").absolute()
|
||||
for path, description in (
|
||||
(bench_dir / "src" / "pdf_parser.py", "OpenDataLoader parser"),
|
||||
(bench_dir / "src" / "evaluator.py", "OpenDataLoader evaluator"),
|
||||
(baseline, "baseline binary"),
|
||||
(candidate, "candidate binary"),
|
||||
(python, "Python interpreter"),
|
||||
):
|
||||
if not path.exists():
|
||||
raise SystemExit(f"{description} not found: {path}")
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="pdf-inspector-opendataloader-") as temporary:
|
||||
scratch_root = Path(temporary)
|
||||
baseline_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=baseline,
|
||||
label="baseline",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
candidate_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=candidate,
|
||||
label="candidate",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
|
||||
reference = None
|
||||
if args.reference_evaluation is not None:
|
||||
with args.reference_evaluation.resolve().open(encoding="utf-8") as handle:
|
||||
reference = json.load(handle)
|
||||
|
||||
comparison = compare_evaluations(
|
||||
baseline_evaluation,
|
||||
candidate_evaluation,
|
||||
reference,
|
||||
top=args.top,
|
||||
)
|
||||
|
||||
_print_report(comparison)
|
||||
if args.json_output is not None:
|
||||
args.json_output.resolve().write_text(
|
||||
json.dumps(comparison, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=args.min_overall_delta,
|
||||
max_document_regression=args.max_document_regression,
|
||||
max_missing=args.max_missing,
|
||||
require_reference_lead=args.require_reference_lead,
|
||||
)
|
||||
if failures:
|
||||
print("\nBenchmark gate failed:", file=sys.stderr)
|
||||
for failure in failures:
|
||||
print(f" - {failure}", file=sys.stderr)
|
||||
return 1
|
||||
print("\nBenchmark gate passed.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,351 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare pdf-inspector evidence with optional MuPDF structured text.
|
||||
|
||||
This is an experiment and diagnostic tool, not an extraction fallback. It runs
|
||||
MuPDF's deterministic ``stext.json`` backend without OCR and highlights pages
|
||||
where that backend exposes materially different text or layout evidence.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from collections import Counter
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any, Iterable
|
||||
|
||||
|
||||
TOKEN_PATTERN = re.compile(r"[^\W_]+(?:[\u2019'][^\W_]+)*", re.UNICODE)
|
||||
|
||||
|
||||
def _tokens(texts: Iterable[str]) -> Counter[str]:
|
||||
tokens: Counter[str] = Counter()
|
||||
for text in texts:
|
||||
for token in TOKEN_PATTERN.findall(text.casefold()):
|
||||
# Lone letters are frequently bullets, chart labels, or fragmented
|
||||
# glyphs. Digits remain useful even when they are one character.
|
||||
if len(token) > 1 or token.isdigit():
|
||||
tokens[token] += 1
|
||||
return tokens
|
||||
|
||||
|
||||
def _repeated_x_anchors(xs: Iterable[float], *, tolerance: float = 4.0) -> int:
|
||||
buckets = Counter(round(float(x) / tolerance) for x in xs)
|
||||
return sum(count >= 3 for count in buckets.values())
|
||||
|
||||
|
||||
def local_pages(payload: dict[str, Any]) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize positioned ``pdf2md --items-json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
for item in payload.get("items", []):
|
||||
page_number = int(item["page"])
|
||||
page = pages.setdefault(
|
||||
page_number,
|
||||
{"texts": [], "xs": [], "text_items": 0, "image_items": 0},
|
||||
)
|
||||
if item.get("item_type") == "image":
|
||||
page["image_items"] += 1
|
||||
continue
|
||||
text = str(item.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
page["xs"].append(float(item.get("x", 0.0)))
|
||||
page["text_items"] += 1
|
||||
return pages
|
||||
|
||||
|
||||
def alternate_pages(
|
||||
payload: dict[str, Any] | list[dict[str, Any]],
|
||||
) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize MuPDF ``stext.json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
raw_pages = payload if isinstance(payload, list) else payload.get("pages", [])
|
||||
for index, raw_page in enumerate(raw_pages, start=1):
|
||||
page_number = int(raw_page.get("number", index))
|
||||
page = {
|
||||
"texts": [],
|
||||
"xs": [],
|
||||
"text_blocks": 0,
|
||||
"text_lines": 0,
|
||||
"image_blocks": 0,
|
||||
}
|
||||
for block in raw_page.get("blocks", []):
|
||||
if block.get("type") == "image":
|
||||
page["image_blocks"] += 1
|
||||
continue
|
||||
if block.get("type") != "text":
|
||||
continue
|
||||
page["text_blocks"] += 1
|
||||
for line in block.get("lines", []):
|
||||
text = str(line.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
bbox = line.get("bbox", {})
|
||||
page["xs"].append(float(bbox.get("x", line.get("x", 0.0))))
|
||||
page["text_lines"] += 1
|
||||
pages[page_number] = page
|
||||
return pages
|
||||
|
||||
|
||||
def compare_page(
|
||||
local: dict[str, Any],
|
||||
alternate: dict[str, Any],
|
||||
*,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
"""Compare semantic and coarse layout evidence for one page."""
|
||||
local_tokens = _tokens(local.get("texts", []))
|
||||
alternate_tokens = _tokens(alternate.get("texts", []))
|
||||
shared = local_tokens & alternate_tokens
|
||||
alternate_only = alternate_tokens - local_tokens
|
||||
local_only = local_tokens - alternate_tokens
|
||||
local_total = sum(local_tokens.values())
|
||||
alternate_total = sum(alternate_tokens.values())
|
||||
shared_total = sum(shared.values())
|
||||
alternate_only_total = sum(alternate_only.values())
|
||||
local_only_total = sum(local_only.values())
|
||||
net_token_gain = alternate_total - local_total
|
||||
alternate_only_ratio = alternate_only_total / max(alternate_total, 1)
|
||||
|
||||
local_anchors = _repeated_x_anchors(local.get("xs", []))
|
||||
alternate_anchors = _repeated_x_anchors(alternate.get("xs", []))
|
||||
anchor_gain = alternate_anchors - local_anchors
|
||||
image_gain = int(alternate.get("image_blocks", 0)) - int(
|
||||
local.get("image_items", 0)
|
||||
)
|
||||
|
||||
reasons: list[str] = []
|
||||
if local_total == 0 and alternate_total >= max(5, min_token_gain // 2):
|
||||
reasons.append("local_text_empty")
|
||||
elif (
|
||||
net_token_gain >= min_token_gain
|
||||
and alternate_only_ratio >= min_alternate_only_ratio
|
||||
):
|
||||
reasons.append("alternate_has_more_text")
|
||||
if anchor_gain >= min_anchor_gain:
|
||||
reasons.append("alternate_has_more_alignment_anchors")
|
||||
if image_gain > 0:
|
||||
reasons.append("alternate_has_more_image_blocks")
|
||||
|
||||
if reasons:
|
||||
classification = "investigate_alternate_evidence"
|
||||
elif local_total - alternate_total >= min_token_gain:
|
||||
classification = "local_has_more_text"
|
||||
elif alternate_only_total + local_only_total:
|
||||
classification = "different_segmentation_or_decoding"
|
||||
else:
|
||||
classification = "equivalent_text_evidence"
|
||||
|
||||
return {
|
||||
"classification": classification,
|
||||
"reasons": reasons,
|
||||
"tokens": {
|
||||
"local": local_total,
|
||||
"alternate": alternate_total,
|
||||
"shared": shared_total,
|
||||
"net_alternate_gain": net_token_gain,
|
||||
"alternate_only": alternate_only_total,
|
||||
"local_only": local_only_total,
|
||||
"alternate_only_ratio": alternate_only_ratio,
|
||||
"alternate_only_sample": sorted(alternate_only)[:12],
|
||||
"local_only_sample": sorted(local_only)[:12],
|
||||
},
|
||||
"layout": {
|
||||
"local_text_items": int(local.get("text_items", 0)),
|
||||
"local_image_items": int(local.get("image_items", 0)),
|
||||
"local_repeated_x_anchors": local_anchors,
|
||||
"alternate_text_blocks": int(alternate.get("text_blocks", 0)),
|
||||
"alternate_text_lines": int(alternate.get("text_lines", 0)),
|
||||
"alternate_image_blocks": int(alternate.get("image_blocks", 0)),
|
||||
"alternate_repeated_x_anchors": alternate_anchors,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def compare_documents(
|
||||
local_payload: dict[str, Any],
|
||||
alternate_payload: dict[str, Any] | list[dict[str, Any]],
|
||||
*,
|
||||
min_token_gain: int = 20,
|
||||
min_alternate_only_ratio: float = 0.15,
|
||||
min_anchor_gain: int = 2,
|
||||
) -> dict[str, Any]:
|
||||
"""Return a page-level evidence report for already extracted payloads."""
|
||||
local = local_pages(local_payload)
|
||||
alternate = alternate_pages(alternate_payload)
|
||||
page_numbers = sorted(local.keys() | alternate.keys())
|
||||
pages = []
|
||||
for page_number in page_numbers:
|
||||
result = compare_page(
|
||||
local.get(page_number, {}),
|
||||
alternate.get(page_number, {}),
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
result["page"] = page_number
|
||||
pages.append(result)
|
||||
|
||||
flagged = [
|
||||
page
|
||||
for page in pages
|
||||
if page["classification"] == "investigate_alternate_evidence"
|
||||
]
|
||||
return {
|
||||
"summary": {
|
||||
"pages": len(pages),
|
||||
"flagged_pages": len(flagged),
|
||||
"flagged_page_numbers": [page["page"] for page in flagged],
|
||||
"local_tokens": sum(page["tokens"]["local"] for page in pages),
|
||||
"alternate_tokens": sum(page["tokens"]["alternate"] for page in pages),
|
||||
"alternate_only_tokens": sum(
|
||||
page["tokens"]["alternate_only"] for page in pages
|
||||
),
|
||||
},
|
||||
"pages": pages,
|
||||
}
|
||||
|
||||
|
||||
def _json_command(command: list[str]) -> Any:
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
check=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as error:
|
||||
detail = error.stderr.strip() or error.stdout.strip() or "no diagnostic output"
|
||||
raise RuntimeError(f"command failed: {' '.join(command)}\n{detail}") from error
|
||||
try:
|
||||
return json.loads(completed.stdout)
|
||||
except json.JSONDecodeError as error:
|
||||
raise RuntimeError(
|
||||
f"command did not return JSON: {' '.join(command)}: {error}"
|
||||
) from error
|
||||
|
||||
|
||||
def probe_pdf(
|
||||
pdf: Path,
|
||||
*,
|
||||
pdf2md: Path,
|
||||
mutool: Path,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
local_payload = _json_command([str(pdf2md), str(pdf), "--items-json"])
|
||||
# `stext.json` is MuPDF's native structured text output. The OCR formats
|
||||
# are intentionally not used so this remains a deterministic no-model
|
||||
# comparison.
|
||||
alternate_payload = _json_command(
|
||||
[str(mutool), "draw", "-q", "-F", "stext.json", "-o", "-", str(pdf)]
|
||||
)
|
||||
report = compare_documents(
|
||||
local_payload,
|
||||
alternate_payload,
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
report["pdf"] = str(pdf)
|
||||
return report
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("pdf", type=Path, nargs="+")
|
||||
parser.add_argument("--pdf2md", type=Path, default=Path("target/release/pdf2md"))
|
||||
parser.add_argument("--mutool", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--min-token-gain", type=int, default=20)
|
||||
parser.add_argument("--min-alternate-only-ratio", type=float, default=0.15)
|
||||
parser.add_argument("--min-anchor-gain", type=int, default=2)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def _print_report(result: dict[str, Any]) -> None:
|
||||
summary = result["summary"]
|
||||
print(f"\n{result['pdf']}")
|
||||
print(
|
||||
f" {summary['flagged_pages']}/{summary['pages']} pages flagged; "
|
||||
f"tokens local={summary['local_tokens']} alternate={summary['alternate_tokens']} "
|
||||
f"alternate-only={summary['alternate_only_tokens']}"
|
||||
)
|
||||
for page in result["pages"]:
|
||||
if page["classification"] != "investigate_alternate_evidence":
|
||||
continue
|
||||
reasons = ", ".join(page["reasons"])
|
||||
tokens = page["tokens"]
|
||||
print(
|
||||
f" page {page['page']}: {reasons}; "
|
||||
f"net tokens={tokens['net_alternate_gain']:+d}, "
|
||||
f"alternate-only={tokens['alternate_only']}"
|
||||
)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
pdf2md = args.pdf2md.absolute()
|
||||
mutool = args.mutool or (Path(found) if (found := shutil.which("mutool")) else None)
|
||||
if not pdf2md.is_file():
|
||||
print(f"error: pdf2md binary not found: {pdf2md}", file=sys.stderr)
|
||||
return 2
|
||||
if mutool is None or not mutool.is_file():
|
||||
print("error: mutool not found; install MuPDF or pass --mutool", file=sys.stderr)
|
||||
return 2
|
||||
if (
|
||||
args.min_token_gain < 0
|
||||
or args.min_anchor_gain < 0
|
||||
or not 0.0 <= args.min_alternate_only_ratio <= 1.0
|
||||
):
|
||||
print("error: thresholds must be non-negative and ratio must be in [0, 1]", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
results = []
|
||||
for pdf in args.pdf:
|
||||
path = pdf.absolute()
|
||||
if not path.is_file():
|
||||
print(f"error: PDF not found: {path}", file=sys.stderr)
|
||||
return 2
|
||||
try:
|
||||
result = probe_pdf(
|
||||
path,
|
||||
pdf2md=pdf2md,
|
||||
mutool=mutool,
|
||||
min_token_gain=args.min_token_gain,
|
||||
min_alternate_only_ratio=args.min_alternate_only_ratio,
|
||||
min_anchor_gain=args.min_anchor_gain,
|
||||
)
|
||||
except RuntimeError as error:
|
||||
print(f"error: {error}", file=sys.stderr)
|
||||
return 1
|
||||
results.append(result)
|
||||
_print_report(result)
|
||||
|
||||
payload = {
|
||||
"schema_version": 1,
|
||||
"experiment": "optional_mupdf_stext_evidence",
|
||||
"ocr": False,
|
||||
"thresholds": {
|
||||
"min_token_gain": args.min_token_gain,
|
||||
"min_alternate_only_ratio": args.min_alternate_only_ratio,
|
||||
"min_anchor_gain": args.min_anchor_gain,
|
||||
},
|
||||
"documents": results,
|
||||
}
|
||||
if args.json_output:
|
||||
args.json_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.json_output.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,203 @@
|
||||
import io
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from contextlib import redirect_stderr, redirect_stdout
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from bench_opendataloader import (
|
||||
_arguments,
|
||||
_print_report,
|
||||
_run_engine,
|
||||
compare_evaluations,
|
||||
evaluate_gates,
|
||||
)
|
||||
|
||||
|
||||
def evaluation(overall, documents, *, missing=0):
|
||||
return {
|
||||
"metrics": {
|
||||
"score": {
|
||||
"overall_mean": overall,
|
||||
"nid_mean": overall + 0.01,
|
||||
},
|
||||
"missing_predictions": missing,
|
||||
},
|
||||
"documents": [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"scores": {"overall": score},
|
||||
}
|
||||
for document_id, score in documents.items()
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class ComparisonTests(unittest.TestCase):
|
||||
def test_reports_metric_and_document_deltas(self):
|
||||
baseline = evaluation(0.80, {"a": 0.8, "b": 0.6, "c": 0.7})
|
||||
candidate = evaluation(0.82, {"a": 0.9, "b": 0.5, "c": 0.7})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, top=1)
|
||||
|
||||
self.assertAlmostEqual(result["deltas"]["overall_mean"], 0.02)
|
||||
self.assertEqual(result["documents"]["improved"], 1)
|
||||
self.assertEqual(result["documents"]["regressed"], 1)
|
||||
self.assertEqual(result["documents"]["unchanged"], 1)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_improvements"][0]["document_id"], "a"
|
||||
)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_regressions"][0]["document_id"], "b"
|
||||
)
|
||||
|
||||
def test_reference_delta_is_reported(self):
|
||||
baseline = evaluation(0.80, {})
|
||||
candidate = evaluation(0.82, {})
|
||||
reference = evaluation(0.81, {})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, reference)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
result["candidate_vs_reference"]["overall_mean"], 0.01
|
||||
)
|
||||
|
||||
def test_gates_cover_aggregate_document_missing_and_reference(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.79, {"a": 0.7}, missing=1),
|
||||
evaluation(0.81, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 4)
|
||||
|
||||
def test_regression_gate_is_independent_of_report_limit(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.80, {"a": 0.7}),
|
||||
top=0,
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=False,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 1)
|
||||
self.assertIn("largest document regression", failures[0])
|
||||
|
||||
def test_report_handles_reference_without_overall_score(self):
|
||||
result = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
{"metrics": {"score": {"nid_mean": 0.81}}},
|
||||
)
|
||||
|
||||
output = io.StringIO()
|
||||
with redirect_stdout(output):
|
||||
_print_report(result)
|
||||
|
||||
self.assertIn("Reference overall: n/a; candidate delta: n/a", output.getvalue())
|
||||
|
||||
def test_reference_gate_reports_missing_score_as_unavailable(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=None,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(failures, ["reference overall score is unavailable"])
|
||||
|
||||
def test_arguments_reject_negative_counts_and_allow_zero_top(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
self.assertEqual(_arguments(required + ["--top", "0"]).top, 0)
|
||||
for option in ("--top", "--max-document-regression", "--max-missing"):
|
||||
with self.subTest(option=option), redirect_stderr(io.StringIO()):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, "-1"])
|
||||
|
||||
def test_arguments_reject_nonfinite_float_thresholds(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
for option in ("--min-overall-delta", "--max-document-regression"):
|
||||
for value in ("nan", "inf", "-inf"):
|
||||
with self.subTest(option=option, value=value), redirect_stderr(
|
||||
io.StringIO()
|
||||
):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, value])
|
||||
|
||||
def test_run_engine_clears_stale_predictions_before_parser(self):
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
bench_dir = root / "bench"
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
source.mkdir(parents=True)
|
||||
(source / "stale.md").write_text("stale", encoding="utf-8")
|
||||
scratch = root / "scratch"
|
||||
scratch.mkdir()
|
||||
|
||||
def fake_run(command, *, cwd, env=None):
|
||||
if any(part.endswith("pdf_parser.py") for part in command):
|
||||
self.assertFalse(source.exists())
|
||||
(source / "markdown").mkdir(parents=True)
|
||||
(source / "markdown" / "new.md").write_text(
|
||||
"new", encoding="utf-8"
|
||||
)
|
||||
else:
|
||||
destination = scratch / "candidate"
|
||||
(destination / "evaluation.json").write_text(
|
||||
json.dumps(evaluation(0.82, {})), encoding="utf-8"
|
||||
)
|
||||
|
||||
with patch("bench_opendataloader._run", side_effect=fake_run):
|
||||
result = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=Path("python"),
|
||||
binary=Path("pdf2md"),
|
||||
label="candidate",
|
||||
scratch_root=scratch,
|
||||
)
|
||||
|
||||
self.assertEqual(result["metrics"]["score"]["overall_mean"], 0.82)
|
||||
self.assertFalse((source / "stale.md").exists())
|
||||
self.assertFalse((scratch / "candidate" / "stale.md").exists())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,103 @@
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from probe_backend_evidence import compare_documents
|
||||
|
||||
|
||||
def local_payload(items):
|
||||
return {"items": items}
|
||||
|
||||
|
||||
def item(page, text, x=10, item_type="text"):
|
||||
return {"page": page, "text": text, "x": x, "item_type": item_type}
|
||||
|
||||
|
||||
def alternate_payload(pages):
|
||||
return {"pages": pages}
|
||||
|
||||
|
||||
def page(lines, *, images=0):
|
||||
blocks = [
|
||||
{
|
||||
"type": "text",
|
||||
"lines": [
|
||||
{"text": text, "bbox": {"x": x, "y": index * 10, "w": 80, "h": 8}}
|
||||
for index, (text, x) in enumerate(lines)
|
||||
],
|
||||
}
|
||||
]
|
||||
blocks.extend({"type": "image"} for _ in range(images))
|
||||
return {"blocks": blocks}
|
||||
|
||||
|
||||
class EvidenceComparisonTests(unittest.TestCase):
|
||||
def test_accepts_real_top_level_page_array(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = [page([("alpha beta gamma", 10)])]
|
||||
|
||||
result = compare_documents(local, alternate)["pages"][0]
|
||||
|
||||
self.assertEqual(result["tokens"]["alternate"], 3)
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 1)
|
||||
|
||||
def test_flags_material_alternate_text_gain(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = alternate_payload(
|
||||
[page([("alpha beta gamma delta epsilon zeta", 10)])]
|
||||
)
|
||||
|
||||
report = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=3,
|
||||
min_alternate_only_ratio=0.2,
|
||||
)
|
||||
|
||||
result = report["pages"][0]
|
||||
self.assertEqual(result["classification"], "investigate_alternate_evidence")
|
||||
self.assertIn("alternate_has_more_text", result["reasons"])
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 4)
|
||||
|
||||
def test_repeated_alignment_and_image_evidence_are_reported(self):
|
||||
local = local_payload([item(1, "one two", 10)])
|
||||
alternate = alternate_payload(
|
||||
[
|
||||
page(
|
||||
[
|
||||
("one two", 10),
|
||||
("row three", 100),
|
||||
("row four", 100),
|
||||
("row five", 100),
|
||||
],
|
||||
images=1,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
result = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=99,
|
||||
min_anchor_gain=1,
|
||||
)["pages"][0]
|
||||
|
||||
self.assertIn("alternate_has_more_alignment_anchors", result["reasons"])
|
||||
self.assertIn("alternate_has_more_image_blocks", result["reasons"])
|
||||
self.assertEqual(result["layout"]["alternate_repeated_x_anchors"], 1)
|
||||
|
||||
def test_token_segmentation_difference_does_not_imply_more_evidence(self):
|
||||
local = local_payload([item(1, "Revenue 2025")])
|
||||
alternate = alternate_payload([page([("Revenue 2024", 10)])])
|
||||
|
||||
result = compare_documents(local, alternate, min_token_gain=2)["pages"][0]
|
||||
|
||||
self.assertEqual(result["classification"], "different_segmentation_or_decoding")
|
||||
self.assertEqual(result["reasons"], [])
|
||||
self.assertEqual(result["tokens"]["alternate_only_sample"], ["2024"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+1100
-356
File diff suppressed because it is too large
Load Diff
+44
-2
@@ -11,6 +11,37 @@ use std::process;
|
||||
use std::time::Instant;
|
||||
|
||||
/// Escape a string for embedding in a JSON string value.
|
||||
fn format_detector_ocr_reasons(reasons: &std::collections::BTreeMap<u32, Vec<String>>) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|(page, page_reasons)| {
|
||||
let reasons_json = page_reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|entry| {
|
||||
let reasons_json = entry
|
||||
.reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn json_escape(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len() + 16);
|
||||
for ch in s.chars() {
|
||||
@@ -32,6 +63,7 @@ fn json_escape(s: &str) -> String {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
@@ -117,11 +149,13 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"detection_time_ms":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"detection_time_ms":{}}}"#,
|
||||
pdf_type_str(&result.pdf_type),
|
||||
result.page_count,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
@@ -144,6 +178,9 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
println!("Page count: {}", result.page_count);
|
||||
if !result.pages_needing_ocr.is_empty() {
|
||||
println!("Pages needing OCR: {:?}", result.pages_needing_ocr);
|
||||
for entry in &result.ocr_reasons_by_page {
|
||||
println!(" page {}: {}", entry.page, entry.reasons.join(", "));
|
||||
}
|
||||
}
|
||||
println!();
|
||||
if result.layout.is_complex {
|
||||
@@ -184,8 +221,9 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_detector_ocr_reasons(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_sampled":{},"pages_with_text":{},"confidence":{:.2},"title":{},"ocr_recommended":{},"pages_needing_ocr":[{}],"detection_time_ms":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_sampled":{},"pages_with_text":{},"confidence":{:.2},"title":{},"ocr_recommended":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"detection_time_ms":{}}}"#,
|
||||
pdf_type_str(&result.pdf_type),
|
||||
result.page_count,
|
||||
result.pages_sampled,
|
||||
@@ -198,6 +236,7 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
.unwrap_or_else(|| "null".to_string()),
|
||||
result.ocr_recommended,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
elapsed.as_millis()
|
||||
);
|
||||
} else {
|
||||
@@ -232,6 +271,9 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
result.pages_needing_ocr, result.page_count
|
||||
);
|
||||
}
|
||||
for (page, reasons) in &result.ocr_reasons_by_page {
|
||||
println!(" page {}: {}", page, reasons.join(", "));
|
||||
}
|
||||
}
|
||||
if let Some(title) = &result.title {
|
||||
println!("Title: {}", title);
|
||||
|
||||
@@ -190,6 +190,7 @@ fn print_layout_info(layout: &LayoutComplexity) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
@@ -206,6 +207,9 @@ fn main() {
|
||||
eprintln!(" --json Output result as JSON");
|
||||
eprintln!(" --items-json Output positioned TextItem JSON");
|
||||
eprintln!(" --raw Output only markdown (no headers)");
|
||||
eprintln!(
|
||||
" --compact Collapse token-heavy source formatting such as dot leaders"
|
||||
);
|
||||
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
|
||||
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
||||
eprintln!(" --password PW Password for an encrypted PDF");
|
||||
@@ -218,6 +222,7 @@ fn main() {
|
||||
let json_output = args.iter().any(|a| a == "--json");
|
||||
let items_json_output = args.iter().any(|a| a == "--items-json");
|
||||
let raw_output = args.iter().any(|a| a == "--raw");
|
||||
let compact_output = args.iter().any(|a| a == "--compact");
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
let analyze = args.iter().any(|a| a == "--analyze");
|
||||
@@ -276,6 +281,9 @@ fn main() {
|
||||
};
|
||||
|
||||
let mut options = PdfOptions::new().mode(process_mode);
|
||||
if compact_output {
|
||||
options.markdown.profile = pdf_inspector::MarkdownProfile::Compact;
|
||||
}
|
||||
options.markdown.include_page_numbers = page_numbers;
|
||||
if let Some(pages) = page_filter {
|
||||
options.page_filter = Some(pages);
|
||||
|
||||
+111
-2
@@ -60,6 +60,10 @@ pub struct PdfTypeResult {
|
||||
/// 1-indexed page numbers that need OCR (image-only or insufficient text).
|
||||
/// Empty for TextBased. All pages for Scanned/ImageBased. Specific pages for Mixed.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Per-page explanation for `pages_needing_ocr`: 1-indexed page → reason
|
||||
/// codes (`scanned`, `no_text`, `vector_text`, `suspected_garbled_text`).
|
||||
/// Only contains pages that need OCR.
|
||||
pub ocr_reasons_by_page: std::collections::BTreeMap<u32, Vec<String>>,
|
||||
}
|
||||
|
||||
/// Configuration for PDF type detection
|
||||
@@ -382,7 +386,12 @@ pub(crate) fn detect_from_document(
|
||||
let analysis = if let Some(cached) = analysis_cache.get(&page_num) {
|
||||
cached.clone()
|
||||
} else if let Some(&page_id) = pages.get(&page_num) {
|
||||
analyze_page_content(doc, page_id)
|
||||
// Cache the fresh analysis so the reason-classification pass
|
||||
// below sees the real signals (vector_text, etc.) instead of
|
||||
// defaulting to "scanned".
|
||||
let a = analyze_page_content(doc, page_id);
|
||||
analysis_cache.insert(page_num, a.clone());
|
||||
a
|
||||
} else {
|
||||
continue;
|
||||
};
|
||||
@@ -429,6 +438,9 @@ pub(crate) fn detect_from_document(
|
||||
let analysis = analyze_page_content(doc, page_id);
|
||||
if analysis.has_identity_h_no_tounicode || analysis.has_only_type3_fonts {
|
||||
pages_needing_ocr.push(page_num);
|
||||
// Cache so the reason pass reports suspected_garbled_text
|
||||
// rather than defaulting to "scanned".
|
||||
analysis_cache.insert(page_num, analysis);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -436,6 +448,19 @@ pub(crate) fn detect_from_document(
|
||||
pages_needing_ocr.sort();
|
||||
pages_needing_ocr.dedup();
|
||||
|
||||
// Explain each OCR-flagged page. Pages we analyzed get a signal-derived
|
||||
// reason; pages flagged only by whole-document classification (unsampled
|
||||
// pages of a Scanned/ImageBased doc) default to `scanned`.
|
||||
let mut ocr_reasons_by_page: std::collections::BTreeMap<u32, Vec<String>> =
|
||||
std::collections::BTreeMap::new();
|
||||
for &page_num in &pages_needing_ocr {
|
||||
let reasons = match analysis_cache.get(&page_num) {
|
||||
Some(analysis) => page_ocr_reasons(analysis),
|
||||
None => vec![crate::OCR_REASON_SCANNED],
|
||||
};
|
||||
ocr_reasons_by_page.insert(page_num, reasons.into_iter().map(String::from).collect());
|
||||
}
|
||||
|
||||
// Try to get title from metadata
|
||||
let title = get_document_title(doc);
|
||||
|
||||
@@ -448,6 +473,7 @@ pub(crate) fn detect_from_document(
|
||||
title,
|
||||
ocr_recommended,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -487,7 +513,7 @@ fn distribute_pages(n: u32, total: u32) -> Vec<u32> {
|
||||
}
|
||||
|
||||
/// Page content analysis result
|
||||
#[derive(Clone)]
|
||||
#[derive(Clone, Default)]
|
||||
struct PageAnalysis {
|
||||
text_operator_count: u32,
|
||||
has_images: bool,
|
||||
@@ -523,6 +549,31 @@ struct PageAnalysis {
|
||||
has_decodable_text_fonts: bool,
|
||||
}
|
||||
|
||||
/// Explain *why* a page needs OCR, from its content analysis. Priority:
|
||||
/// undecodable fonts (`suspected_garbled_text`) and vector-outlined text
|
||||
/// (`vector_text`) come first because they persist even when a text layer is
|
||||
/// present; otherwise a page with no extractable text is `scanned` when an
|
||||
/// image backs it or `no_text` when nothing does.
|
||||
fn page_ocr_reasons(a: &PageAnalysis) -> Vec<&'static str> {
|
||||
let mut reasons = Vec::new();
|
||||
if a.has_identity_h_no_tounicode || a.has_only_type3_fonts {
|
||||
reasons.push(crate::OCR_REASON_SUSPECTED_GARBLED_TEXT);
|
||||
}
|
||||
if a.has_vector_text {
|
||||
reasons.push(crate::OCR_REASON_VECTOR_TEXT);
|
||||
}
|
||||
if reasons.is_empty() {
|
||||
let has_extractable_text = a.text_operator_count > 0 && a.unique_text_chars > 0;
|
||||
if !has_extractable_text && !a.has_images && !a.has_template_image {
|
||||
reasons.push(crate::OCR_REASON_NO_TEXT);
|
||||
} else {
|
||||
// Image-backed with no usable text, or too little text to trust.
|
||||
reasons.push(crate::OCR_REASON_SCANNED);
|
||||
}
|
||||
}
|
||||
reasons
|
||||
}
|
||||
|
||||
/// Extracted font information from a Resource dictionary entry.
|
||||
/// Stores the properties needed for decodability/identity-h checks
|
||||
/// without holding a reference to the document.
|
||||
@@ -1809,6 +1860,64 @@ fn get_document_title(doc: &Document) -> Option<String> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn page_ocr_reasons_classify() {
|
||||
// Scanned: no text, full-page image.
|
||||
let scanned = PageAnalysis {
|
||||
has_template_image: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(page_ocr_reasons(&scanned), vec![crate::OCR_REASON_SCANNED]);
|
||||
|
||||
// Image-only page (no template flag, but has an image).
|
||||
let image_only = PageAnalysis {
|
||||
has_images: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&image_only),
|
||||
vec![crate::OCR_REASON_SCANNED]
|
||||
);
|
||||
|
||||
// No text, no image → no_text.
|
||||
let blank = PageAnalysis::default();
|
||||
assert_eq!(page_ocr_reasons(&blank), vec![crate::OCR_REASON_NO_TEXT]);
|
||||
|
||||
// Vector-outlined text.
|
||||
let vector = PageAnalysis {
|
||||
has_vector_text: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&vector),
|
||||
vec![crate::OCR_REASON_VECTOR_TEXT]
|
||||
);
|
||||
|
||||
// Undecodable fonts → garbled, and it wins over the fall-through.
|
||||
let garbled = PageAnalysis {
|
||||
has_identity_h_no_tounicode: true,
|
||||
has_images: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&garbled),
|
||||
vec![crate::OCR_REASON_SUSPECTED_GARBLED_TEXT]
|
||||
);
|
||||
|
||||
// A page with real extractable text and an image is not flagged here
|
||||
// as scanned/no_text (only reached for pages already needing OCR).
|
||||
let text_with_image = PageAnalysis {
|
||||
text_operator_count: 40,
|
||||
unique_text_chars: 120,
|
||||
has_images: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&text_with_image),
|
||||
vec![crate::OCR_REASON_SCANNED]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scan_content_operators() {
|
||||
let mut uchars = HashSet::new();
|
||||
|
||||
@@ -162,7 +162,7 @@ pub(crate) fn extract_page_text_items(
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
|
||||
// Build font encoding maps from Differences arrays
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
|
||||
|
||||
// Build font width info for accurate text positioning
|
||||
let font_widths = build_font_widths(doc, &fonts);
|
||||
|
||||
+196
-12
@@ -497,9 +497,13 @@ pub(crate) fn get_operand_bytes(obj: &Object) -> Option<&[u8]> {
|
||||
/// Build encoding maps for all fonts on a page.
|
||||
/// Returns `(encodings, has_gid_fonts)` where `has_gid_fonts` is true when
|
||||
/// any font uses raw glyph ID names (gidNNNNN) that can't be decoded.
|
||||
/// Gid names whose codes the font's own ToUnicode CMap maps are decodable
|
||||
/// and do not set the flag (LibreOffice subsets write /gidNNNN Differences
|
||||
/// names alongside a complete ToUnicode CMap).
|
||||
pub(crate) fn build_font_encodings(
|
||||
doc: &Document,
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
cmaps: &FontCMaps,
|
||||
) -> (PageFontEncodings, bool) {
|
||||
let mut encodings = PageFontEncodings::new();
|
||||
let mut has_gid_fonts = false;
|
||||
@@ -508,7 +512,9 @@ pub(crate) fn build_font_encodings(
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
|
||||
if let Some(result) = parse_font_encoding(doc, font_dict) {
|
||||
if result.gid_glyph_count > 0 {
|
||||
if !result.gid_codes.is_empty()
|
||||
&& !tounicode_maps_codes(font_dict, cmaps, &result.gid_codes)
|
||||
{
|
||||
has_gid_fonts = true;
|
||||
}
|
||||
if !result.map.is_empty() {
|
||||
@@ -520,6 +526,34 @@ pub(crate) fn build_font_encodings(
|
||||
(encodings, has_gid_fonts)
|
||||
}
|
||||
|
||||
/// True when the font's ToUnicode CMap maps the gid-named character codes,
|
||||
/// so the Differences entries still decode through the CMap.
|
||||
fn tounicode_maps_codes(font_dict: &lopdf::Dictionary, cmaps: &FontCMaps, codes: &[u8]) -> bool {
|
||||
let Some(obj_ref) = font_dict
|
||||
.get(b"ToUnicode")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
else {
|
||||
return false;
|
||||
};
|
||||
let Some(entry) = cmaps.get_by_obj(obj_ref.0) else {
|
||||
return false;
|
||||
};
|
||||
// At least one gid code usably mapped means the CMap addresses these
|
||||
// codes; remaining unmapped codes are subset leftovers (e.g. the
|
||||
// component glyphs of an emoji ZWJ sequence mapped whole on its first
|
||||
// code). A mapping is usable only when extraction would accept it —
|
||||
// empty or U+FFFD results are rejected there as invalid. Fonts whose
|
||||
// CMap ignores the gid codes entirely stay flagged, and the downstream
|
||||
// garbage/encoding checks still catch partial damage.
|
||||
codes.iter().any(|&code| {
|
||||
entry
|
||||
.primary
|
||||
.lookup(code as u16)
|
||||
.is_some_and(|s| !s.is_empty() && !s.contains('\u{FFFD}'))
|
||||
})
|
||||
}
|
||||
|
||||
/// Parse font encoding from a font dictionary
|
||||
pub(crate) fn parse_font_encoding(
|
||||
doc: &Document,
|
||||
@@ -558,11 +592,10 @@ pub(crate) fn parse_font_encoding(
|
||||
/// Result of parsing an encoding dictionary's Differences array.
|
||||
pub(crate) struct EncodingResult {
|
||||
pub map: FontEncodingMap,
|
||||
/// Number of glyph names matching the `gidNNNNN` pattern (raw glyph IDs).
|
||||
/// These indicate a font with unresolvable encoding — the glyph IDs
|
||||
/// reference the original font's glyph table, but without the original
|
||||
/// font's cmap there is no way to map them to Unicode.
|
||||
pub gid_glyph_count: u32,
|
||||
/// Character codes whose glyph names match the `gidNNNNN` pattern (raw
|
||||
/// glyph IDs). These reference the original font's glyph table and are
|
||||
/// only decodable when the font's ToUnicode CMap maps the code.
|
||||
pub gid_codes: Vec<u8>,
|
||||
}
|
||||
|
||||
/// Parse an encoding dictionary with Differences array
|
||||
@@ -588,7 +621,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
let mut encoding_map = FontEncodingMap::new();
|
||||
let mut current_code: u8 = 0;
|
||||
let mut ligature_count = 0u32;
|
||||
let mut gid_glyph_count = 0u32;
|
||||
let mut gid_codes: Vec<u8> = Vec::new();
|
||||
|
||||
for item in diff_array {
|
||||
match item {
|
||||
@@ -614,7 +647,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
&& glyph_name.len() >= 4
|
||||
&& glyph_name[3..].chars().all(|c| c.is_ascii_digit())
|
||||
{
|
||||
gid_glyph_count += 1;
|
||||
gid_codes.push(current_code);
|
||||
}
|
||||
if let Some(ch) = mapped_char {
|
||||
encoding_map.insert(current_code, ch);
|
||||
@@ -638,16 +671,16 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
);
|
||||
}
|
||||
|
||||
if gid_glyph_count > 0 {
|
||||
if !gid_codes.is_empty() {
|
||||
debug!(
|
||||
" Differences: {} gid-encoded glyphs (unresolvable without original font)",
|
||||
gid_glyph_count
|
||||
" Differences: {} gid-encoded glyphs (decodable only via ToUnicode)",
|
||||
gid_codes.len()
|
||||
);
|
||||
}
|
||||
|
||||
Some(EncodingResult {
|
||||
map: encoding_map,
|
||||
gid_glyph_count,
|
||||
gid_codes,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1187,10 +1220,37 @@ pub(crate) fn extract_text_from_operand(
|
||||
})();
|
||||
result.map(|text| {
|
||||
let text = clean_symbol_pua(text);
|
||||
let text = remap_texcm_math_symbols(text, base_font_name);
|
||||
normalize_cp1252_controls(text, use_cp1252_fallback)
|
||||
})
|
||||
}
|
||||
|
||||
/// Fix a known producer bug in "TeXCMMathsSymbols" subset fonts (IntechOpen
|
||||
/// and sibling academic pipelines): the Computer Modern symbol glyphs are
|
||||
/// misnamed after Latin lookalikes (equal → /onequarter, plus → /thorn, …)
|
||||
/// and the generated ToUnicode faithfully propagates the wrong names. The
|
||||
/// remap applies only to text decoded from that font, keyed on the glyphs'
|
||||
/// observed misnames.
|
||||
fn remap_texcm_math_symbols(text: String, base_font_name: Option<&str>) -> String {
|
||||
let is_texcm = base_font_name.is_some_and(|n| {
|
||||
let n = n.rsplit_once('+').map_or(n, |(_, s)| s);
|
||||
n.eq_ignore_ascii_case("TeXCMMathsSymbols")
|
||||
});
|
||||
if !is_texcm {
|
||||
return text;
|
||||
}
|
||||
text.chars()
|
||||
.map(|c| match c {
|
||||
'¼' => '=',
|
||||
'½' => '-',
|
||||
'þ' => '+',
|
||||
'ð' => '(',
|
||||
'Þ' => ')',
|
||||
_ => c,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn decode_single_byte_fallback(bytes: &[u8], use_cp1252_fallback: bool) -> String {
|
||||
bytes
|
||||
.iter()
|
||||
@@ -1409,6 +1469,21 @@ fn score_text(text: &str) -> i32 {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn texcm_math_symbols_remap() {
|
||||
assert_eq!(
|
||||
super::remap_texcm_math_symbols("S ¼ kB þ 1".into(), Some("EEKVNO+TeXCMMathsSymbols")),
|
||||
"S = kB + 1"
|
||||
);
|
||||
// Other fonts keep their genuine fractions/thorns.
|
||||
assert_eq!(
|
||||
super::remap_texcm_math_symbols("¼ cup þorn".into(), Some("Times-Roman")),
|
||||
"¼ cup þorn"
|
||||
);
|
||||
assert_eq!(super::remap_texcm_math_symbols("¼".into(), None), "¼");
|
||||
}
|
||||
|
||||
use super::*;
|
||||
use lopdf::dictionary;
|
||||
|
||||
@@ -1896,4 +1971,113 @@ mod tests {
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
fn gid_font_doc(bfchar: Option<&str>) -> (Document, lopdf::ObjectId) {
|
||||
use lopdf::Stream;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let cmap = format!(
|
||||
"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
{}
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end",
|
||||
bfchar.unwrap_or_default()
|
||||
);
|
||||
let tounicode_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
cmap.into_bytes(),
|
||||
)));
|
||||
let enc_id = doc.add_object(dictionary! {
|
||||
"Type" => "Encoding",
|
||||
"Differences" => vec![
|
||||
1.into(),
|
||||
Object::Name(b"gid1283".to_vec()),
|
||||
Object::Name(b"gid1464".to_vec()),
|
||||
],
|
||||
});
|
||||
let mut font = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "ABCDEF+OpenSymbol",
|
||||
"Encoding" => Object::Reference(enc_id),
|
||||
};
|
||||
if bfchar.is_some() {
|
||||
font.set("ToUnicode", Object::Reference(tounicode_id));
|
||||
}
|
||||
let font_id = doc.add_object(font);
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn gid_flagged(bfchar: Option<&str>) -> bool {
|
||||
let (doc, page_id) = gid_font_doc(bfchar);
|
||||
let cmaps = FontCMaps::from_doc(&doc);
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap();
|
||||
let (_, has_gid_fonts) = build_font_encodings(&doc, &fonts, &cmaps);
|
||||
has_gid_fonts
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_covering_tounicode_are_not_flagged() {
|
||||
// LibreOffice subsets write /gidNNNN Differences names alongside a
|
||||
// ToUnicode CMap that decodes those codes; the page must not be
|
||||
// flagged as unresolvable (which would suppress the whole document's
|
||||
// markdown when every page carries such a font).
|
||||
assert!(!gid_flagged(Some("<01> <2022>\n<02> <25E6>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_partial_tounicode_are_not_flagged() {
|
||||
// An emoji ZWJ sequence maps whole on its first code; the remaining
|
||||
// component-glyph codes are subset leftovers, not damage.
|
||||
assert!(!gid_flagged(Some(
|
||||
"<01> <D83DDC68200DD83DDC69200DD83DDC67>"
|
||||
)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_without_tounicode_are_flagged() {
|
||||
assert!(
|
||||
gid_flagged(None),
|
||||
"gid glyphs without ToUnicode are unresolvable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_disjoint_tounicode_are_flagged() {
|
||||
// A ToUnicode that never addresses the gid codes leaves them
|
||||
// unresolvable.
|
||||
assert!(gid_flagged(Some("<10> <0041>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn gid_differences_with_replacement_char_tounicode_are_flagged() {
|
||||
// A mapping to U+FFFD is not usable — extraction rejects it as an
|
||||
// invalid CMap result — so it must not clear the gid flag.
|
||||
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
|
||||
}
|
||||
}
|
||||
|
||||
+1006
-19
File diff suppressed because it is too large
Load Diff
+90
-2
@@ -153,16 +153,54 @@ pub(crate) fn extract_form_fields(
|
||||
},
|
||||
Err(_) => return items,
|
||||
};
|
||||
if fields.is_empty() {
|
||||
return items;
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(doc, field_ref, None, "", page_map, &mut items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
field_ref,
|
||||
None,
|
||||
"",
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
items
|
||||
}
|
||||
|
||||
/// Map widget annotation objects back to the page whose `/Annots` array owns
|
||||
/// them. Some valid widgets omit `/P`, so the page tree is the only reliable
|
||||
/// ownership signal available for page-filtered extraction.
|
||||
fn annotation_page_map(
|
||||
doc: &Document,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
) -> HashMap<ObjectId, u32> {
|
||||
let mut annotation_pages = HashMap::new();
|
||||
for (&page_id, &page_num) in page_map {
|
||||
let Some(annotations) = doc
|
||||
.get_dictionary(page_id)
|
||||
.ok()
|
||||
.and_then(|page| page.get(b"Annots").ok())
|
||||
.and_then(|annotations| resolve_array(doc, annotations))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
for annotation in annotations {
|
||||
if let Ok(annotation_id) = annotation.as_reference() {
|
||||
annotation_pages.insert(annotation_id, page_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
annotation_pages
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
@@ -170,6 +208,7 @@ pub(crate) fn walk_form_fields(
|
||||
parent_ft: Option<&[u8]>,
|
||||
parent_name: &str,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
) {
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
@@ -206,7 +245,15 @@ pub(crate) fn walk_form_fields(
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(doc, kid_ref, ft, &full_name, page_map, items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
kid_ref,
|
||||
ft,
|
||||
&full_name,
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -299,6 +346,7 @@ pub(crate) fn walk_form_fields(
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.and_then(|p| page_map.get(&p).copied())
|
||||
.or_else(|| annotation_pages.get(&field_id).copied())
|
||||
.unwrap_or(1);
|
||||
|
||||
let text = if full_name.is_empty() {
|
||||
@@ -324,3 +372,43 @@ pub(crate) fn walk_form_fields(
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::{dictionary, Object};
|
||||
|
||||
#[test]
|
||||
fn widget_without_page_reference_uses_owning_page_annotation() {
|
||||
let mut doc = Document::new();
|
||||
let widget_id = doc.add_object(dictionary! {
|
||||
"Type" => "Annot",
|
||||
"Subtype" => "Widget",
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("customer"),
|
||||
"V" => Object::string_literal("Alice"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
});
|
||||
let page_one_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
});
|
||||
let page_two_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Annots" => vec![Object::Reference(widget_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(widget_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::from([(page_one_id, 1), (page_two_id, 2)]);
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
}
|
||||
|
||||
+1005
-17
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,591 @@
|
||||
//! Region-graph evidence for page reading order.
|
||||
//!
|
||||
//! Whole-page column histograms fail when images or spanning captions occupy
|
||||
//! only part of a page. This module turns image geometry and repeated row
|
||||
//! gutters into a small directed acyclic graph: content above a local column
|
||||
//! band, the left flow, the right flow, and content below it. The graph is
|
||||
//! deliberately evidence-gated; ordinary pages keep the established layout
|
||||
//! path.
|
||||
|
||||
use crate::text_utils::{effective_width, is_cjk_char, is_rtl_text};
|
||||
use crate::types::TextItem;
|
||||
|
||||
const MIN_IMAGE_WIDTH: f32 = 60.0;
|
||||
const MIN_IMAGE_HEIGHT: f32 = 40.0;
|
||||
const MIN_ROW_GUTTER: f32 = 8.0;
|
||||
const SPLIT_CLUSTER_TOLERANCE: f32 = 20.0;
|
||||
const MIN_ALIGNED_ROWS: usize = 4;
|
||||
|
||||
pub(crate) type ImageRegion = (f32, f32, f32, f32);
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub(crate) struct ColumnFlowBand {
|
||||
pub(crate) split_x: f32,
|
||||
pub(crate) y_bottom: f32,
|
||||
pub(crate) y_top: f32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum RegionKind {
|
||||
FullWidth,
|
||||
Column,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RegionNode {
|
||||
pub(crate) kind: RegionKind,
|
||||
pub(crate) items: Vec<TextItem>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Row<'a> {
|
||||
y: f32,
|
||||
items: Vec<&'a TextItem>,
|
||||
}
|
||||
|
||||
fn page_x_bounds(items: &[TextItem], images: &[ImageRegion]) -> Option<(f32, f32)> {
|
||||
let text_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let text_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + effective_width(item))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let image_min = images
|
||||
.iter()
|
||||
.map(|region| region.0.min(region.2))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_max = images
|
||||
.iter()
|
||||
.map(|region| region.0.max(region.2))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let x_min = text_min.min(image_min);
|
||||
let x_max = text_max.max(image_max);
|
||||
(x_min.is_finite() && x_max.is_finite() && x_max > x_min).then_some((x_min, x_max))
|
||||
}
|
||||
|
||||
fn group_rows(items: &[TextItem]) -> Vec<Row<'_>> {
|
||||
const Y_TOLERANCE: f32 = 3.0;
|
||||
let mut sorted: Vec<&TextItem> = items.iter().collect();
|
||||
sorted.sort_by(|left, right| right.y.total_cmp(&left.y));
|
||||
let mut rows: Vec<Row<'_>> = Vec::new();
|
||||
for item in sorted {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| (row.y - item.y).abs() <= Y_TOLERANCE)
|
||||
{
|
||||
row.items.push(item);
|
||||
row.y = row.items.iter().map(|member| member.y).sum::<f32>() / row.items.len() as f32;
|
||||
} else {
|
||||
rows.push(Row {
|
||||
y: item.y,
|
||||
items: vec![item],
|
||||
});
|
||||
}
|
||||
}
|
||||
for row in &mut rows {
|
||||
row.items.sort_by(|left, right| left.x.total_cmp(&right.x));
|
||||
}
|
||||
rows
|
||||
}
|
||||
|
||||
fn side_is_prose(items: &[&TextItem]) -> bool {
|
||||
let text = items
|
||||
.iter()
|
||||
.map(|item| item.text.trim())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let alphabetic_count = text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphabetic())
|
||||
.count();
|
||||
let cjk_count = text
|
||||
.chars()
|
||||
.filter(|character| is_cjk_char(*character))
|
||||
.count();
|
||||
(text.split_whitespace().count() >= 3 || cjk_count >= 10) && alphabetic_count >= 10
|
||||
}
|
||||
|
||||
fn aligned_row_split(row: &Row<'_>, x_min: f32, x_max: f32) -> Option<f32> {
|
||||
if row.items.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let page_width = x_max - x_min;
|
||||
let center_low = x_min + page_width * 0.25;
|
||||
let center_high = x_min + page_width * 0.75;
|
||||
row.items
|
||||
.windows(2)
|
||||
.filter_map(|pair| {
|
||||
let left_end = pair[0].x + effective_width(pair[0]);
|
||||
let right_start = pair[1].x;
|
||||
let gap = right_start - left_end;
|
||||
let split_x = (left_end + right_start) / 2.0;
|
||||
if gap < MIN_ROW_GUTTER || split_x < center_low || split_x > center_high {
|
||||
return None;
|
||||
}
|
||||
let left: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 < split_x)
|
||||
.collect();
|
||||
let right: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 >= split_x)
|
||||
.collect();
|
||||
(side_is_prose(&left) && side_is_prose(&right)).then_some((split_x, gap))
|
||||
})
|
||||
.max_by(|left, right| left.1.total_cmp(&right.1))
|
||||
.map(|candidate| candidate.0)
|
||||
}
|
||||
|
||||
fn local_flow_below_full_width_image(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
let full_width_images: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let width = (x1 - x0).abs();
|
||||
let height = (y1 - y0).abs();
|
||||
width >= page_width * 0.65 && height >= 60.0
|
||||
})
|
||||
.collect();
|
||||
// A local column flow below an image is only unambiguous for a single,
|
||||
// nearly square hero/figure. Wide report banners and full-page artwork
|
||||
// frequently sit above unrelated page furniture whose aligned labels can
|
||||
// mimic prose columns.
|
||||
if full_width_images.len() != 1 {
|
||||
return None;
|
||||
}
|
||||
let (image_x0, _, image_x1, _) = full_width_images[0];
|
||||
let anchor_width = (image_x1 - image_x0).abs();
|
||||
let anchor_height = (full_width_images[0].3 - full_width_images[0].1).abs();
|
||||
if anchor_width < page_width * 0.85
|
||||
|| anchor_height < anchor_width * 0.85
|
||||
|| anchor_height > anchor_width * 1.2
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let image_bottom = full_width_images
|
||||
.iter()
|
||||
.map(|&(_, y0, _, y1)| y0.min(y1))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if !image_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let below: Vec<TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| item.y < image_bottom && item.y >= image_bottom - 220.0)
|
||||
.cloned()
|
||||
.collect();
|
||||
let candidates: Vec<(f32, f32)> = group_rows(&below)
|
||||
.into_iter()
|
||||
.filter_map(|row| aligned_row_split(&row, x_min, x_max).map(|split| (split, row.y)))
|
||||
.collect();
|
||||
if candidates.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut clusters: Vec<Vec<(f32, f32)>> = Vec::new();
|
||||
for candidate in candidates {
|
||||
if let Some(cluster) = clusters.iter_mut().find(|cluster| {
|
||||
let mean = cluster.iter().map(|entry| entry.0).sum::<f32>() / cluster.len() as f32;
|
||||
(mean - candidate.0).abs() <= SPLIT_CLUSTER_TOLERANCE
|
||||
}) {
|
||||
cluster.push(candidate);
|
||||
} else {
|
||||
clusters.push(vec![candidate]);
|
||||
}
|
||||
}
|
||||
let dominant = clusters.into_iter().max_by_key(Vec::len)?;
|
||||
if dominant.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
let split_x = dominant.iter().map(|entry| entry.0).sum::<f32>() / dominant.len() as f32;
|
||||
let y_top = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
let image_gap = image_bottom - y_top;
|
||||
if !(60.0..=120.0).contains(&image_gap) {
|
||||
return None;
|
||||
}
|
||||
let y_bottom = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if y_top - y_bottom > 130.0 {
|
||||
return None;
|
||||
}
|
||||
log::debug!(
|
||||
"page {}: full-width image flow images={} aligned_rows={} split={:.1} page=[{:.1}..{:.1}] image_bottom={:.1} y=[{:.1}..{:.1}] full_width={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
images.len(),
|
||||
dominant.len(),
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
image_bottom,
|
||||
y_bottom,
|
||||
y_top,
|
||||
full_width_images
|
||||
);
|
||||
Some(ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
})
|
||||
}
|
||||
|
||||
fn paired_column_images(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
split_x: f32,
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
if split_x < x_min + page_width * 0.4 || split_x > x_min + page_width * 0.6 {
|
||||
return None;
|
||||
}
|
||||
let qualifying: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let image_left = x0.min(x1);
|
||||
let image_right = x0.max(x1);
|
||||
let confined_to_one_column = image_right <= split_x || image_left >= split_x;
|
||||
confined_to_one_column
|
||||
&& (x1 - x0).abs() >= MIN_IMAGE_WIDTH
|
||||
&& (y1 - y0).abs() >= MIN_IMAGE_HEIGHT
|
||||
})
|
||||
.collect();
|
||||
let wide_images: Vec<ImageRegion> = qualifying
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(x0, _, x1, _)| (x1 - x0).abs() >= page_width * 0.35)
|
||||
.collect();
|
||||
if qualifying.len() < 3 || wide_images.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
let has_left = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 < split_x);
|
||||
let has_right = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 >= split_x);
|
||||
if !has_left || !has_right {
|
||||
return None;
|
||||
}
|
||||
// A meaningful image-backed column flow spans multiple vertical panels.
|
||||
// Three same-row header/logo images can otherwise satisfy the image count
|
||||
// and send an ordinary asymmetric page through sequential column order.
|
||||
let image_y_min = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.min(region.3))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_y_max = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let has_vertical_stack = wide_images.iter().enumerate().any(|(index, left)| {
|
||||
wide_images.iter().skip(index + 1).any(|right| {
|
||||
let same_side =
|
||||
((left.0 + left.2) / 2.0 < split_x) == ((right.0 + right.2) / 2.0 < split_x);
|
||||
let left_center = (left.1 + left.3) / 2.0;
|
||||
let right_center = (right.1 + right.3) / 2.0;
|
||||
let left_height = (left.3 - left.1).abs();
|
||||
let right_height = (right.3 - right.1).abs();
|
||||
let vertical_gap = if left.1.max(left.3) < right.1.min(right.3) {
|
||||
right.1.min(right.3) - left.1.max(left.3)
|
||||
} else if right.1.max(right.3) < left.1.min(left.3) {
|
||||
left.1.min(left.3) - right.1.max(right.3)
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
same_side
|
||||
&& (left_center - right_center).abs() >= left_height.min(right_height) * 0.5
|
||||
&& vertical_gap <= left_height.max(right_height) * 0.5
|
||||
})
|
||||
});
|
||||
if image_y_max - image_y_min < page_width * 0.45 || !has_vertical_stack {
|
||||
return None;
|
||||
}
|
||||
let y_top = qualifying
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
// Only column-confined text proves the lower extent of the flow. A
|
||||
// spanning heading or caption below the columns must become the trailing
|
||||
// full-width node rather than stretching the column band to the page foot.
|
||||
let y_bottom = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
let item_right = item.x + effective_width(item);
|
||||
item.y <= y_top && (item_right <= split_x || item.x >= split_x)
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if !y_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
let distinct_rows = |right: bool| {
|
||||
let mut ys: Vec<f32> = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
item.y <= y_top && (item.x + effective_width(item) / 2.0 >= split_x) == right
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.collect();
|
||||
ys.sort_by(|left, right| left.total_cmp(right));
|
||||
ys.dedup_by(|left, right| (*left - *right).abs() <= 3.0);
|
||||
ys.len()
|
||||
};
|
||||
let left_rows = distinct_rows(false);
|
||||
let right_rows = distinct_rows(true);
|
||||
let line_balance = left_rows.min(right_rows) as f32 / left_rows.max(right_rows).max(1) as f32;
|
||||
(left_rows >= 5 && right_rows >= 5 && line_balance < 0.55).then(|| {
|
||||
log::debug!(
|
||||
"page {}: paired-image flow qualifying_images={} rows={}/{} split={:.1} page=[{:.1}..{:.1}] y=[{:.1}..{:.1}] images={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
qualifying.len(),
|
||||
left_rows,
|
||||
right_rows,
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
y_bottom,
|
||||
y_top,
|
||||
qualifying
|
||||
);
|
||||
ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn infer_image_anchored_flow(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
detected_split: Option<f32>,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
if items.is_empty() || images.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let (x_min, x_max) = page_x_bounds(items, images)?;
|
||||
detected_split
|
||||
.and_then(|split_x| paired_column_images(items, images, split_x, x_min, x_max))
|
||||
.or_else(|| local_flow_below_full_width_image(items, images, x_min, x_max))
|
||||
}
|
||||
|
||||
/// Partition a page into the topological order `above -> left -> right -> below`.
|
||||
/// These edges encode the reading-order DAG; empty nodes are omitted.
|
||||
pub(crate) fn build_region_graph(items: Vec<TextItem>, band: ColumnFlowBand) -> Vec<RegionNode> {
|
||||
let mut above = Vec::new();
|
||||
let mut left = Vec::new();
|
||||
let mut right = Vec::new();
|
||||
let mut below = Vec::new();
|
||||
for item in items {
|
||||
if item.y > band.y_top {
|
||||
above.push(item);
|
||||
} else if item.y < band.y_bottom {
|
||||
below.push(item);
|
||||
} else if item.x + effective_width(&item) / 2.0 < band.split_x {
|
||||
left.push(item);
|
||||
} else {
|
||||
right.push(item);
|
||||
}
|
||||
}
|
||||
let rtl = is_rtl_text(left.iter().chain(right.iter()).map(|item| &item.text));
|
||||
let mut ordered = vec![(RegionKind::FullWidth, above)];
|
||||
if rtl {
|
||||
ordered.push((RegionKind::Column, right));
|
||||
ordered.push((RegionKind::Column, left));
|
||||
} else {
|
||||
ordered.push((RegionKind::Column, left));
|
||||
ordered.push((RegionKind::Column, right));
|
||||
}
|
||||
ordered.push((RegionKind::FullWidth, below));
|
||||
ordered
|
||||
.into_iter()
|
||||
.filter_map(|(kind, items)| (!items.is_empty()).then_some(RegionNode { kind, items }))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.into(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 11.0,
|
||||
font: "F1".into(),
|
||||
font_size: 11.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_local_two_column_flow() {
|
||||
let mut items = vec![
|
||||
item("A full width caption", 55.0, 230.0, 430.0),
|
||||
item("A trailing full width heading", 55.0, 80.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
let band = infer_image_anchored_flow(&items, &images, None).unwrap();
|
||||
assert!((band.split_x - 272.5).abs() < 2.0);
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph.len(), 4);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "A trailing full width heading");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_cjk_column_flow() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("左栏这是没有空格的正文内容", 55.0, y, 210.0));
|
||||
items.push(item("右栏这是没有空格的正文内容", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_images_anchor_unbalanced_column_flows() {
|
||||
let mut items = vec![
|
||||
item("running header", 55.0, 700.0, 430.0),
|
||||
item("trailing full width caption", 55.0, 300.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
500.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 5..12 {
|
||||
items.push(item(
|
||||
"right continuation prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 530.0, 255.0, 680.0),
|
||||
(55.0, 380.0, 255.0, 530.0),
|
||||
(280.0, 560.0, 490.0, 680.0),
|
||||
];
|
||||
let band = infer_image_anchored_flow(&items, &images, Some(270.0)).unwrap();
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "trailing full width caption");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rtl_region_graph_reads_right_column_first() {
|
||||
let items = vec![
|
||||
item("A long English report header", 55.0, 250.0, 430.0),
|
||||
item("نص العمود الأيسر", 55.0, 150.0, 180.0),
|
||||
item("نص العمود الأيمن", 300.0, 150.0, 180.0),
|
||||
];
|
||||
let graph = build_region_graph(
|
||||
items,
|
||||
ColumnFlowBand {
|
||||
split_x: 270.0,
|
||||
y_bottom: 100.0,
|
||||
y_top: 200.0,
|
||||
},
|
||||
);
|
||||
|
||||
assert_eq!(graph.len(), 3);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert!(graph[1].items[0].x > graph[2].items[0].x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_header_logos_do_not_anchor_page_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 0..30 {
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 720.0, 205.0, 770.0),
|
||||
(60.0, 718.0, 210.0, 768.0),
|
||||
(280.0, 720.0, 450.0, 770.0),
|
||||
];
|
||||
assert!(infer_image_anchored_flow(&items, &images, Some(270.0)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_banner_does_not_anchor_local_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
let y = 270.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 310.0, 490.0, 550.0)];
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_none());
|
||||
}
|
||||
}
|
||||
+288
-7
@@ -115,7 +115,13 @@ fn rules_from_graphics(rects: &[PdfRect], lines: &[UnderlineLine], page: u32) ->
|
||||
rules
|
||||
}
|
||||
|
||||
fn discard_repeated_ruling_rules(rules: Vec<Rule>) -> Vec<Rule> {
|
||||
fn discard_repeated_ruling_rules(
|
||||
rules: Vec<Rule>,
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) -> Vec<Rule> {
|
||||
if rules.len() < MIN_REPEATED_RULE_LEVELS {
|
||||
return rules;
|
||||
}
|
||||
@@ -123,12 +129,142 @@ fn discard_repeated_ruling_rules(rules: Vec<Rule>) -> Vec<Rule> {
|
||||
rules
|
||||
.iter()
|
||||
.filter(|rule| {
|
||||
!is_repeated_ruling_rule(rule, &rules) && !is_segmented_row_ruling_rule(rule, &rules)
|
||||
// A rule snugly owned by one text line is an underline even when
|
||||
// span-similar rules repeat down the page — documents that
|
||||
// underline many full-width lines (dense CJK business docs) look
|
||||
// exactly like table rulings to the repetition check, which used
|
||||
// to discard every one of them. Table rulings fail snugness:
|
||||
// row separators extend past their cells' text (or have no text
|
||||
// on the baseline above), and multi-column matches are still
|
||||
// culled by the tabular filter afterwards.
|
||||
// Same-row segmented rules (column-header separators) are
|
||||
// always rulings — each segment snugly owns its column label,
|
||||
// so snugness must not override that check.
|
||||
!is_segmented_row_ruling_rule(rule, &rules)
|
||||
&& ((has_snug_text_owner(rule, items)
|
||||
&& !has_flanking_verticals(rule, rects, lines, page))
|
||||
|| !is_repeated_ruling_rule(rule, &rules))
|
||||
})
|
||||
.cloned()
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// True when a single text item both matches the rule vertically (baseline
|
||||
/// window) and horizontally contains it: the rule may not extend past the
|
||||
/// item's span by more than ~0.75em on either side. Underlines are drawn to
|
||||
/// the width of the text they decorate; table/form rulings span cells or
|
||||
/// full table width and overshoot any single item.
|
||||
/// A rule flanked by vertical strokes at its ends is a table/box border
|
||||
/// row edge, not an underline — underlined text lines have no vertical
|
||||
/// rules rising from their ends. Checked against raw stroked lines: a
|
||||
/// near-vertical segment whose x sits at either end of the rule and whose
|
||||
/// y-range covers the rule's row.
|
||||
fn has_flanking_verticals(
|
||||
rule: &Rule,
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) -> bool {
|
||||
// A drawn rect that CONTAINS the rule vetoes rescue only with GRID
|
||||
// EVIDENCE: another drawn rect abutting it vertically (cell rows tile).
|
||||
// Height alone can't separate a table cell from a decorative callout
|
||||
// panel — genuine underlines live inside isolated filled panels, and
|
||||
// multiline table cells can be arbitrarily tall.
|
||||
let norm = |r: &PdfRect| {
|
||||
let (x_lo, x_hi) = if r.width >= 0.0 {
|
||||
(r.x, r.x + r.width)
|
||||
} else {
|
||||
(r.x + r.width, r.x)
|
||||
};
|
||||
let (y_lo, y_hi) = if r.height >= 0.0 {
|
||||
(r.y, r.y + r.height)
|
||||
} else {
|
||||
(r.y + r.height, r.y)
|
||||
};
|
||||
(x_lo, x_hi, y_lo, y_hi)
|
||||
};
|
||||
let page_rects: Vec<(f32, f32, f32, f32)> = rects
|
||||
.iter()
|
||||
.filter(|r| r.page == page && r.height.abs() > 6.0)
|
||||
.map(norm)
|
||||
.collect();
|
||||
let rect_flank = page_rects.iter().any(|&(x_lo, x_hi, y_lo, y_hi)| {
|
||||
let contains = x_lo <= rule.x1 + 2.0
|
||||
&& x_hi >= rule.x2 - 2.0
|
||||
&& y_lo <= rule.y + 2.0
|
||||
&& y_hi >= rule.y - 2.0;
|
||||
if !contains {
|
||||
return false;
|
||||
}
|
||||
// Grid evidence: a vertically abutting neighbor box with x-overlap.
|
||||
page_rects.iter().any(|&(nx_lo, nx_hi, ny_lo, ny_hi)| {
|
||||
let x_overlap = nx_hi.min(x_hi) - nx_lo.max(x_lo);
|
||||
if x_overlap <= 10.0 {
|
||||
return false;
|
||||
}
|
||||
(ny_lo - y_hi).abs() <= 3.0 || (y_lo - ny_hi).abs() <= 3.0
|
||||
})
|
||||
});
|
||||
if rect_flank {
|
||||
return true;
|
||||
}
|
||||
lines.iter().any(|l| {
|
||||
if l.page != page || (l.x1 - l.x2).abs() > 2.0 {
|
||||
return false;
|
||||
}
|
||||
let x = (l.x1 + l.x2) / 2.0;
|
||||
let near_end = (x - rule.x1).abs() <= 6.0 || (x - rule.x2).abs() <= 6.0;
|
||||
if !near_end {
|
||||
return false;
|
||||
}
|
||||
let (y_lo, y_hi) = if l.y1 <= l.y2 {
|
||||
(l.y1, l.y2)
|
||||
} else {
|
||||
(l.y2, l.y1)
|
||||
};
|
||||
y_lo <= rule.y + 2.0 && y_hi >= rule.y - 2.0
|
||||
})
|
||||
}
|
||||
|
||||
fn has_snug_text_owner(rule: &Rule, items: &[TextItem]) -> bool {
|
||||
// Underlines are drawn to the width of the text they decorate, but the
|
||||
// text may be split into several runs (CJK lines mix scripts and font
|
||||
// switches) — so ownership is judged against the UNION of the runs on
|
||||
// the rule's baseline row. Table/form rulings overshoot their row's
|
||||
// text (row separators span cell padding and empty columns), so they
|
||||
// fail either containment or coverage.
|
||||
let matched: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
|
||||
.collect();
|
||||
if matched.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let x1 = matched.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let x2 = matched
|
||||
.iter()
|
||||
.map(|i| i.x + i.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let max_fs = matched.iter().map(|i| i.font_size).fold(0.0, f32::max);
|
||||
let pad = (max_fs * 0.75).max(4.0);
|
||||
if rule.x1 < x1 - pad || rule.x2 > x2 + pad {
|
||||
return false;
|
||||
}
|
||||
let covered: f32 = matched.iter().map(|i| i.width).sum();
|
||||
if covered < rule.width() * 0.6 {
|
||||
return false;
|
||||
}
|
||||
// A table row also unions to the rule's span — but its cells sit apart.
|
||||
// An underlined text line is contiguous runs with word-sized gaps; any
|
||||
// column-sized hole between matched runs means this is a row ruling.
|
||||
let mut sorted = matched;
|
||||
sorted.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
sorted.windows(2).all(|pair| {
|
||||
let gap = pair[1].x - (pair[0].x + pair[0].width);
|
||||
gap <= (max_fs * 2.0).max(12.0)
|
||||
})
|
||||
}
|
||||
|
||||
fn is_repeated_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut y_levels: Vec<f32> = rules
|
||||
.iter()
|
||||
@@ -215,9 +351,11 @@ fn is_underline_candidate(item: &TextItem) -> bool {
|
||||
|
||||
fn rule_matches_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
// Vertical window: underlines sit at or slightly below the baseline.
|
||||
// Fonts draw them at roughly 5-15% of the em below; allow up to 35%
|
||||
// (min 3pt) below and 1pt above for rounding.
|
||||
let below = (item.font_size * 0.35).max(3.0);
|
||||
// Latin fonts draw them at roughly 5-15% of the em below; CJK layouts
|
||||
// put them under the full em box, measured up to ~0.67em below the
|
||||
// baseline (text_dense__underline). Allow 0.72em (min 3pt) below and
|
||||
// 1pt above for rounding.
|
||||
let below = (item.font_size * 0.72).max(3.0);
|
||||
let y_min = item.y - below;
|
||||
let y_max = item.y + 1.0;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
@@ -260,12 +398,48 @@ pub(crate) fn mark_underlined_items(
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) {
|
||||
let rules = discard_repeated_ruling_rules(rules_from_graphics(rects, lines, page));
|
||||
let rules = discard_repeated_ruling_rules(
|
||||
rules_from_graphics(rects, lines, page),
|
||||
items,
|
||||
rects,
|
||||
lines,
|
||||
page,
|
||||
);
|
||||
if rules.is_empty() {
|
||||
return;
|
||||
}
|
||||
let tabular_rules = tabular_row_separator_rule_indices(&rules, items);
|
||||
|
||||
// Math fraction bars are short horizontal lines with the numerator just
|
||||
// above AND the denominator just below — underline geometry from above,
|
||||
// but no underline has text hanging directly beneath it at fraction
|
||||
// distance. Only narrow rules qualify: real underlines under short
|
||||
// labels have their next text line a full line-pitch away.
|
||||
let fraction_rules: HashSet<usize> = rules
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, rule)| {
|
||||
rule.width() <= 60.0
|
||||
&& items.iter().any(|item| {
|
||||
if !is_underline_candidate(item) {
|
||||
return false;
|
||||
}
|
||||
// A denominator HUGS the bar (fraction typesetting
|
||||
// leaves ~0.1-0.2em) and is bar-sized. Both bounds
|
||||
// matter: a short last-line of a paragraph at normal
|
||||
// leading sits further below, and a full next text
|
||||
// line is far wider than the rule.
|
||||
let dy = rule.y - (item.y + item.height);
|
||||
let overlap = rule.x2.min(item.x + item.width) - rule.x1.max(item.x);
|
||||
dy > 0.0
|
||||
&& dy <= item.font_size * 0.3
|
||||
&& overlap > rule.width() * 0.5
|
||||
&& item.width <= rule.width() * 1.5
|
||||
})
|
||||
})
|
||||
.map(|(i, _)| i)
|
||||
.collect();
|
||||
|
||||
for item in items.iter_mut() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
@@ -275,7 +449,10 @@ pub(crate) fn mark_underlined_items(
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
if rule_matches_item(rule, item) {
|
||||
// The fraction guard only gates UNDERLINE marking — a rule that
|
||||
// reads as a fraction bar from below can still legitimately
|
||||
// strike through a line above it.
|
||||
if !fraction_rules.contains(&rule_idx) && rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
@@ -323,6 +500,16 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn cell_rect(x: f32, y: f32, width: f32, height: f32) -> PdfRect {
|
||||
PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
fn thin_rect(x: f32, y: f32, width: f32) -> PdfRect {
|
||||
PdfRect {
|
||||
x,
|
||||
@@ -543,6 +730,100 @@ mod tests {
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_snug_underlines_survive_ruling_filter() {
|
||||
// Dense docs underline many full-width lines: span-similar rules at
|
||||
// 3+ y-levels used to be discarded wholesale as table rulings.
|
||||
// Each rule here snugly matches one text line, so all must mark.
|
||||
let mut items = vec![
|
||||
item("first underlined line of text", 50.0, 700.0, 300.0, 11.0),
|
||||
item("second underlined line here", 50.0, 650.0, 300.0, 11.0),
|
||||
item("third underlined line as well", 50.0, 600.0, 300.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 350.0, 697.0),
|
||||
hline(50.0, 350.0, 647.0),
|
||||
hline(50.0, 350.0, 597.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rescue_spans_split_runs_on_one_line() {
|
||||
// A single underlined line is often split into several runs (script
|
||||
// or font switches). The union of touching runs owns the rule.
|
||||
let mut items = vec![
|
||||
item("run one", 50.0, 700.0, 100.0, 11.0),
|
||||
item("run two", 150.5, 700.0, 100.0, 11.0),
|
||||
item("run three", 251.0, 700.0, 99.0, 11.0),
|
||||
item("other a", 50.0, 650.0, 300.0, 11.0),
|
||||
item("other b", 50.0, 600.0, 300.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 350.0, 697.0),
|
||||
hline(50.0, 350.0, 647.0),
|
||||
hline(50.0, 350.0, 597.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items[0].is_underline && items[1].is_underline && items[2].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rescue_denied_for_row_with_cell_gaps() {
|
||||
// A full-width rule whose baseline row is several items separated by
|
||||
// column-sized gaps is a table row separator, not an underline —
|
||||
// even when span-similar rules repeat down the page.
|
||||
let mut items = vec![
|
||||
item("cell a", 50.0, 700.0, 60.0, 11.0),
|
||||
item("cell b", 190.0, 700.0, 60.0, 11.0),
|
||||
item("cell c", 330.0, 700.0, 70.0, 11.0),
|
||||
item("cell d", 50.0, 650.0, 60.0, 11.0),
|
||||
item("cell e", 190.0, 650.0, 60.0, 11.0),
|
||||
item("cell f", 330.0, 650.0, 70.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 400.0, 697.0),
|
||||
hline(50.0, 400.0, 647.0),
|
||||
hline(50.0, 400.0, 597.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rescue_denied_inside_cell_box() {
|
||||
// A rule snugly under one text line but enclosed by a drawn cell
|
||||
// box that TILES with vertical neighbors (grid evidence) is a row
|
||||
// ruling of a rect-grid table. Isolated boxes (callout panels) do
|
||||
// not veto — see repeated_snug_underlines_survive_ruling_filter.
|
||||
let mut items = vec![
|
||||
item("one wide cell row", 50.0, 700.0, 300.0, 11.0),
|
||||
item("second wide cell", 50.0, 650.0, 300.0, 11.0),
|
||||
item("third wide cell", 50.0, 600.0, 300.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 350.0, 697.0),
|
||||
hline(50.0, 350.0, 647.0),
|
||||
hline(50.0, 350.0, 597.0),
|
||||
];
|
||||
let boxes = vec![
|
||||
cell_rect(45.0, 690.0, 320.0, 50.0),
|
||||
cell_rect(45.0, 640.0, 320.0, 50.0),
|
||||
cell_rect(45.0, 590.0, 320.0, 50.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &boxes, &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_row_spaced_rule_segments_do_not_mark_column_labels() {
|
||||
let mut items = vec![
|
||||
|
||||
@@ -162,7 +162,7 @@ fn extract_form_xobject_text_inner(
|
||||
|
||||
// Get fonts from the Form's Resources
|
||||
let form_fonts = get_form_fonts(doc, &stream.dict);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
|
||||
|
||||
// Build font width info for the form
|
||||
let font_widths = build_font_widths(doc, &form_fonts);
|
||||
|
||||
+354
-58
@@ -53,7 +53,8 @@ pub use extractor::{
|
||||
extract_text_with_positions_pages,
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -67,10 +68,56 @@ use text_quality::{
|
||||
};
|
||||
use tounicode::FontCMaps;
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
struct ProcessingTimer(std::time::Instant);
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
struct ProcessingTimer;
|
||||
|
||||
impl ProcessingTimer {
|
||||
fn start() -> Self {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
Self(std::time::Instant::now())
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
Self
|
||||
}
|
||||
}
|
||||
|
||||
fn elapsed_ms(&self) -> u64 {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
self.0.elapsed().as_millis() as u64
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
// The wasm32-unknown-unknown standard library has no clock.
|
||||
// Browser bindings measure with JavaScript's host clock.
|
||||
0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reason emitted when the extracted text layer appears garbled due to
|
||||
/// broken font decoding or mojibake.
|
||||
pub const OCR_REASON_SUSPECTED_GARBLED_TEXT: &str = "suspected_garbled_text";
|
||||
|
||||
/// OCR reason: the page is a scanned image (a full-page raster / image-only
|
||||
/// page) with no usable text layer.
|
||||
pub const OCR_REASON_SCANNED: &str = "scanned";
|
||||
|
||||
/// OCR reason: the page has no extractable text and no image to OCR — blank,
|
||||
/// or content the parser cannot reach.
|
||||
pub const OCR_REASON_NO_TEXT: &str = "no_text";
|
||||
|
||||
/// OCR reason: the page's text is drawn as vector outlines (path operators)
|
||||
/// rather than real text operators, so it cannot be extracted as characters.
|
||||
pub const OCR_REASON_VECTOR_TEXT: &str = "vector_text";
|
||||
|
||||
// =========================================================================
|
||||
// Result type
|
||||
// =========================================================================
|
||||
@@ -237,7 +284,7 @@ pub fn process_pdf_with_options<P: AsRef<Path>>(
|
||||
path: P,
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = std::time::Instant::now();
|
||||
let start = ProcessingTimer::start();
|
||||
validate_pdf_file(&path)?;
|
||||
|
||||
// Load the document once — shared by detection AND extraction.
|
||||
@@ -264,7 +311,7 @@ pub fn process_pdf_mem_with_options(
|
||||
buffer: &[u8],
|
||||
options: PdfOptions,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
let start = std::time::Instant::now();
|
||||
let start = ProcessingTimer::start();
|
||||
validate_pdf_bytes(buffer)?;
|
||||
|
||||
let (doc, page_count) =
|
||||
@@ -415,16 +462,39 @@ pub fn extract_pages_markdown_mem(
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats.
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
// unselected page cannot make a valid requested page fail, but errors on a
|
||||
// requested page retain the normal extraction semantics.
|
||||
let required_pages: Option<HashSet<u32>> = pages.map(|pages| {
|
||||
pages
|
||||
.iter()
|
||||
.filter_map(|page| page.checked_add(1))
|
||||
.collect()
|
||||
});
|
||||
let ((all_items, all_rects, all_lines), page_thresholds, gid_pages) =
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||
if let Some(required_pages) = required_pages.as_ref() {
|
||||
extractor::extract_positioned_text_for_document_analysis(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
required_pages,
|
||||
)?
|
||||
} else {
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?
|
||||
};
|
||||
let text_quality = analyze_text_quality(&all_items);
|
||||
|
||||
// Compute layout complexity from full document (near-zero cost).
|
||||
let complexity = compute_layout_complexity(&all_items, &all_rects, &all_lines);
|
||||
// Resolve page numbers with full-document context before partitioning.
|
||||
// Per-page Markdown receives the original items plus these decisions so
|
||||
// table detection can retain legitimate numeric cells.
|
||||
let (filtered_items, removed_page_number_pages, page_number_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
|
||||
// Tables need the original numeric cells; columns use folio-cleaned
|
||||
// evidence so removed page numbers cannot create false layout metadata.
|
||||
let complexity = compute_layout_complexity(&all_items, &filtered_items, &all_rects, &all_lines);
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -455,12 +525,13 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
let page_1idx = page_0idx + 1;
|
||||
|
||||
// Filter items/rects for this page only
|
||||
let page_items: Vec<TextItem> = all_items
|
||||
// Partition items, removal decisions, and rects for this page only.
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page_1idx)
|
||||
.cloned()
|
||||
.collect();
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
let page_rects: Vec<PdfRect> = all_rects
|
||||
.iter()
|
||||
@@ -487,9 +558,14 @@ pub fn extract_pages_markdown_mem(
|
||||
options,
|
||||
&page_rects,
|
||||
&[],
|
||||
&page_thresholds,
|
||||
None,
|
||||
&[],
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_page_number_pages),
|
||||
prefiltered_page_number_mask: Some(&page_number_removal_mask),
|
||||
},
|
||||
)
|
||||
};
|
||||
|
||||
@@ -661,18 +737,51 @@ pub fn extract_text_in_regions_mem(
|
||||
|
||||
let mut page_results = Vec::with_capacity(regions.len());
|
||||
|
||||
for rect in regions {
|
||||
let [rx1, ry1, rx2, ry2] = *rect;
|
||||
// Exclusive item->region assignment: overlapping layout regions used
|
||||
// to extract shared items into EVERY region they touched (the
|
||||
// 1.5pt inclusion margin makes borders generous), duplicating whole
|
||||
// lines in the final markdown on 21% of bench docs — and downstream
|
||||
// duplicate-handling sometimes dropped the variant holding a
|
||||
// sentence tail, turning duplication into content LOSS. Each item
|
||||
// now belongs to the single region with the largest overlap area;
|
||||
// items are partitioned, never suppressed, so no content can vanish.
|
||||
let all_bounds: Vec<RegionBounds> = regions
|
||||
.iter()
|
||||
.map(|rect| {
|
||||
let [rx1, ry1, rx2, ry2] = *rect;
|
||||
region_bounds(rx1, ry1, rx2, ry2, page_h, coords)
|
||||
})
|
||||
.collect();
|
||||
// Single pass over items: assign each to the best-overlap region and
|
||||
// bucket the clone directly (review: avoid a second O(items x
|
||||
// regions) traversal). `had_candidates` marks regions that touched
|
||||
// at least one item even if every one was assigned elsewhere.
|
||||
let mut region_items: Vec<Vec<TextItem>> = vec![Vec::new(); regions.len()];
|
||||
let mut had_candidates: Vec<bool> = vec![false; regions.len()];
|
||||
if let Some(items) = items {
|
||||
for item in items {
|
||||
let mut best: Option<usize> = None;
|
||||
let mut best_area = 0.0_f32;
|
||||
for (ri, b) in all_bounds.iter().enumerate() {
|
||||
if !region_overlaps_item(item, *b) {
|
||||
continue;
|
||||
}
|
||||
had_candidates[ri] = true;
|
||||
let area = region_item_overlap_area(item, *b);
|
||||
if area > best_area {
|
||||
best_area = area;
|
||||
best = Some(ri);
|
||||
}
|
||||
}
|
||||
if let Some(ri) = best {
|
||||
region_items[ri].push(item.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let bounds = region_bounds(rx1, ry1, rx2, ry2, page_h, coords);
|
||||
let matched: Vec<TextItem> = match items {
|
||||
Some(items) => items
|
||||
.iter()
|
||||
.filter(|item| region_overlaps_item(item, bounds))
|
||||
.cloned()
|
||||
.collect(),
|
||||
None => Vec::new(),
|
||||
};
|
||||
for (region_idx, _rect) in regions.iter().enumerate() {
|
||||
let matched: Vec<TextItem> = std::mem::take(&mut region_items[region_idx]);
|
||||
let assigned_count = matched.len();
|
||||
let has_text_quality_issue = region_items_have_decoding_issue(&matched);
|
||||
let text = collect_text_from_matched_items(matched, adaptive_threshold);
|
||||
let has_cid_issue = is_cid_garbage(&text);
|
||||
@@ -686,8 +795,21 @@ pub fn extract_text_in_regions_mem(
|
||||
// Check per-region text quality instead of blanket page-level
|
||||
// GID rejection. A GID font in a logo elsewhere on the page
|
||||
// shouldn't force GPU OCR for clean text regions.
|
||||
let needs_ocr =
|
||||
ocr_reason.is_some() || text.trim().is_empty() || is_garbage_text(&text);
|
||||
// A region whose ONLY overlapping items were assigned to a
|
||||
// better-overlapping neighbor must not fall back to OCR: the
|
||||
// pixels it would re-read belong to that neighbor, and OCR
|
||||
// would reintroduce the duplication exclusivity removed.
|
||||
// Before exclusive assignment these regions were non-empty
|
||||
// native (no OCR), so this preserves the old OCR load too.
|
||||
// Requires ZERO items assigned HERE: a region whose own
|
||||
// assigned items materialize to empty text (whitespace-only,
|
||||
// collector-filtered) keeps its OCR fallback.
|
||||
let lost_to_neighbor = text.trim().is_empty()
|
||||
&& ocr_reason.is_none()
|
||||
&& assigned_count == 0
|
||||
&& had_candidates[region_idx];
|
||||
let needs_ocr = !lost_to_neighbor
|
||||
&& (ocr_reason.is_some() || text.trim().is_empty() || is_garbage_text(&text));
|
||||
|
||||
page_results.push(RegionText {
|
||||
text,
|
||||
@@ -1115,7 +1237,7 @@ pub fn detect_vector_grid_in_region_mem(
|
||||
}
|
||||
|
||||
let line_tables =
|
||||
tables::detect_tables_from_lines(&items_in_region, &lines_in_region, page_1idx);
|
||||
tables::detect_vector_grid_tables_from_lines(&items_in_region, &lines_in_region, page_1idx);
|
||||
for table in line_tables {
|
||||
if let Some(result) = vector_grid_result_from_table(
|
||||
&table,
|
||||
@@ -1468,16 +1590,10 @@ mod vector_grid_tests {
|
||||
/// strip while body rows are drawn with `m`/`l` operators, so the rect
|
||||
/// cluster has only 2 Y-edges and `try_build_grid` rejects.
|
||||
///
|
||||
/// IGNORED: lifting this shape required the exact-duplicate early-dedup
|
||||
/// (PR #76 first iteration), which had broad collateral damage on
|
||||
/// SEC 10-K TOCs and similar docs that draw rule-rects above + below
|
||||
/// section dividers (production diff: 0001104659-25-093871 lost its
|
||||
/// TOC structure, perf-graph data table, and qualifications matrix).
|
||||
/// Re-enable once a more surgical lift exists in `try_build_grid` or
|
||||
/// `snap_edges` that handles cell-border + inner-fill + text-bg rect
|
||||
/// triplets without page-wide dedup.
|
||||
/// The page repeats a full-page background many times. Those fills must be
|
||||
/// removed from clustering before chart/table evidence is evaluated, or
|
||||
/// they swamp the real cell rectangles and make this table look chart-like.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn greencomp_competence_two_cols() {
|
||||
let tables = detect_rect_tables_in_fixture("tests/fixtures/greencomp_competence.pdf");
|
||||
assert!(
|
||||
@@ -3203,8 +3319,26 @@ fn region_bounds(
|
||||
}
|
||||
}
|
||||
|
||||
/// Inclusion margin shared by the region/item overlap predicates and the
|
||||
/// exclusive-assignment area score — these MUST stay in sync: an item that
|
||||
/// passes the boolean guard must always have positive overlap area.
|
||||
const REGION_MARGIN: f32 = 1.5;
|
||||
|
||||
/// Overlap area between an item and region bounds (same margin as the
|
||||
/// boolean test) — the exclusive-assignment score.
|
||||
fn region_item_overlap_area(item: &TextItem, bounds: RegionBounds) -> f32 {
|
||||
let item_x_max = item.x + text_utils::effective_width(item);
|
||||
let item_y_max = item.y + item.height;
|
||||
let x_overlap = (item_x_max.min(bounds.x_max + REGION_MARGIN)
|
||||
- item.x.max(bounds.x_min - REGION_MARGIN))
|
||||
.max(0.0);
|
||||
let y_overlap = (item_y_max.min(bounds.y_max + REGION_MARGIN)
|
||||
- item.y.max(bounds.y_min - REGION_MARGIN))
|
||||
.max(0.0);
|
||||
x_overlap * y_overlap
|
||||
}
|
||||
|
||||
fn region_overlaps_item(item: &TextItem, bounds: RegionBounds) -> bool {
|
||||
const REGION_MARGIN: f32 = 1.5;
|
||||
let item_x_min = item.x;
|
||||
let item_x_max = item.x + text_utils::effective_width(item);
|
||||
let item_y_min = item.y;
|
||||
@@ -3220,7 +3354,6 @@ fn region_overlaps_item(item: &TextItem, bounds: RegionBounds) -> bool {
|
||||
}
|
||||
|
||||
fn region_overlaps_rect(rect: &PdfRect, bounds: RegionBounds) -> bool {
|
||||
const REGION_MARGIN: f32 = 1.5;
|
||||
let (x_min, y_min, x_max, y_max) = normalized_rect_edges(rect);
|
||||
ranges_overlap(
|
||||
x_min,
|
||||
@@ -3236,7 +3369,6 @@ fn region_overlaps_rect(rect: &PdfRect, bounds: RegionBounds) -> bool {
|
||||
}
|
||||
|
||||
fn region_overlaps_line(line: &PdfLine, bounds: RegionBounds) -> bool {
|
||||
const REGION_MARGIN: f32 = 1.5;
|
||||
let x_min = line.x1.min(line.x2);
|
||||
let x_max = line.x1.max(line.x2);
|
||||
let y_min = line.y1.min(line.y2);
|
||||
@@ -3457,7 +3589,7 @@ fn process_document(
|
||||
doc: Document,
|
||||
page_count: u32,
|
||||
options: PdfOptions,
|
||||
start: std::time::Instant,
|
||||
start: ProcessingTimer,
|
||||
) -> Result<PdfProcessResult, PdfError> {
|
||||
// Step 1 — Detection (cheap: scans content streams for text operators)
|
||||
let detection = detector::detect_from_document(&doc, page_count, &options.detection)?;
|
||||
@@ -3465,6 +3597,7 @@ fn process_document(
|
||||
let pages_needing_ocr = detection.pages_needing_ocr;
|
||||
let title = detection.title;
|
||||
let confidence = detection.confidence;
|
||||
let detection_ocr_reasons = detection.ocr_reasons_by_page;
|
||||
|
||||
// DetectOnly → return immediately
|
||||
if options.mode == ProcessMode::DetectOnly {
|
||||
@@ -3472,9 +3605,9 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: Vec::new(),
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
confidence,
|
||||
layout: LayoutComplexity::default(),
|
||||
@@ -3488,9 +3621,9 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown: None,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: Vec::new(),
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(detection_ocr_reasons),
|
||||
title,
|
||||
confidence,
|
||||
layout: LayoutComplexity::default(),
|
||||
@@ -3501,7 +3634,10 @@ fn process_document(
|
||||
// Step 2 — Extraction (reuses the already-loaded document)
|
||||
let extracted = {
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extractor::extract_positioned_text_from_doc(
|
||||
// Most page-filtered requests extract only the selected pages. Gather
|
||||
// other pages only when a selected contextual folio needs cross-page
|
||||
// evidence; failures on those context-only pages are non-fatal.
|
||||
let result = extractor::extract_positioned_text_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3512,9 +3648,19 @@ fn process_document(
|
||||
// This unlocks OCR text layers behind scanned images.
|
||||
if pdf_type == PdfType::Mixed {
|
||||
if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) {
|
||||
let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect();
|
||||
let sample: String = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&item.page))
|
||||
})
|
||||
.take(200)
|
||||
.map(|item| item.text.as_str())
|
||||
.collect();
|
||||
if is_garbage_text(&sample) || sample.trim().is_empty() {
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3524,7 +3670,7 @@ fn process_document(
|
||||
}
|
||||
} else {
|
||||
// Normal extraction failed — try invisible as fallback
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3588,6 +3734,13 @@ fn process_document(
|
||||
let mut garbage_pages: std::collections::HashSet<u32> =
|
||||
std::collections::HashSet::new();
|
||||
for &pg in &ocr_set {
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|filter| !filter.contains(&pg))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let page_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.page == pg)
|
||||
@@ -3627,9 +3780,38 @@ fn process_document(
|
||||
}
|
||||
};
|
||||
|
||||
let selected_page = |page: u32| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&page))
|
||||
};
|
||||
let rects: Vec<_> = rects
|
||||
.into_iter()
|
||||
.filter(|rect| selected_page(rect.page))
|
||||
.collect();
|
||||
let lines: Vec<_> = lines
|
||||
.into_iter()
|
||||
.filter(|line| selected_page(line.page))
|
||||
.collect();
|
||||
let gid_encoded_pages: HashSet<_> = gid_encoded_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
let FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
} = select_items_with_document_folio_context(
|
||||
items,
|
||||
page_count,
|
||||
options.page_filter.as_ref(),
|
||||
);
|
||||
|
||||
let text_quality = analyze_text_quality(&items);
|
||||
merge_ocr_reasons(&mut ocr_reasons_by_page, text_quality.reasons_by_page);
|
||||
let layout = compute_layout_complexity(&items, &rects, &lines);
|
||||
let layout = compute_layout_complexity(&items, &layout_items, &rects, &lines);
|
||||
|
||||
let md = if options.mode == ProcessMode::Analyze {
|
||||
None
|
||||
@@ -3639,9 +3821,14 @@ fn process_document(
|
||||
options.markdown,
|
||||
&rects,
|
||||
&lines,
|
||||
&page_thresholds,
|
||||
struct_roles.as_ref(),
|
||||
&struct_tables,
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: struct_roles.as_ref(),
|
||||
struct_tables: &struct_tables,
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(removal_mask.as_slice()),
|
||||
},
|
||||
))
|
||||
};
|
||||
|
||||
@@ -3754,9 +3941,15 @@ fn process_document(
|
||||
pdf_type,
|
||||
markdown,
|
||||
page_count,
|
||||
processing_time_ms: start.elapsed().as_millis() as u64,
|
||||
processing_time_ms: start.elapsed_ms(),
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page: page_ocr_reasons_vec(text_quality_reasons_by_page),
|
||||
ocr_reasons_by_page: {
|
||||
// Detector reasons (scanned / no_text / vector_text / garbled) merged
|
||||
// with the markdown-stage garbled detection, deduped per page.
|
||||
let mut merged = detection_ocr_reasons;
|
||||
merge_ocr_reasons(&mut merged, text_quality_reasons_by_page);
|
||||
page_ocr_reasons_vec(merged)
|
||||
},
|
||||
title,
|
||||
confidence,
|
||||
layout,
|
||||
@@ -5394,9 +5587,50 @@ mod looks_like_partial_table_tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct FolioFilteredItems {
|
||||
items: Vec<types::TextItem>,
|
||||
layout_items: Vec<types::TextItem>,
|
||||
removal_mask: Vec<bool>,
|
||||
removed_pages: HashSet<u32>,
|
||||
}
|
||||
|
||||
/// Resolve folios with complete document context, then select the caller's
|
||||
/// requested pages without losing those decisions.
|
||||
fn select_items_with_document_folio_context(
|
||||
all_items: Vec<types::TextItem>,
|
||||
page_count: u32,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> FolioFilteredItems {
|
||||
let (all_layout_items, all_removed_pages, all_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
let selected_page = |page: u32| page_filter.is_none_or(|filter| filter.contains(&page));
|
||||
|
||||
let (items, removal_mask) = all_items
|
||||
.into_iter()
|
||||
.zip(all_removal_mask)
|
||||
.filter(|(item, _)| selected_page(item.page))
|
||||
.unzip();
|
||||
let layout_items = all_layout_items
|
||||
.into_iter()
|
||||
.filter(|item| selected_page(item.page))
|
||||
.collect();
|
||||
let removed_pages = all_removed_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
|
||||
FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
}
|
||||
}
|
||||
|
||||
/// Analyse extracted items and rects for layout complexity.
|
||||
fn compute_layout_complexity(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
) -> LayoutComplexity {
|
||||
@@ -5481,7 +5715,7 @@ fn compute_layout_complexity(
|
||||
|
||||
let mut pages_with_columns: Vec<u32> = Vec::new();
|
||||
for page in seen_pages {
|
||||
let cols = extractor::detect_columns(items, page, pages_with_tables.contains(&page));
|
||||
let cols = extractor::detect_columns(column_items, page, pages_with_tables.contains(&page));
|
||||
if cols.len() >= 2 {
|
||||
pages_with_columns.push(page);
|
||||
}
|
||||
@@ -5693,6 +5927,68 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sparse_folios_leave_no_layout_evidence() {
|
||||
let items = vec![
|
||||
test_item("1", 25.0, 20.0, 12.0, 10.0),
|
||||
test_item("2", 520.0, 60.0, 12.0, 10.0),
|
||||
];
|
||||
let (filtered, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(items.clone(), 1);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let filtered = compute_layout_complexity(&items, &filtered, &[], &[]);
|
||||
|
||||
assert!(!filtered.is_complex);
|
||||
assert!(filtered.pages_with_tables.is_empty());
|
||||
assert!(filtered.pages_with_columns.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_selection_keeps_document_wide_folio_layout_decisions() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
for row in 0..8 {
|
||||
let y = 20.0 + row as f32 * 8.0;
|
||||
let mut folio = test_item(&(row * 10 + page).to_string(), 25.0, y, 12.0, 10.0);
|
||||
folio.page = page;
|
||||
let mut footer =
|
||||
test_item(&format!("Footer row {row} summary"), 43.0, y, 470.0, 10.0);
|
||||
footer.page = page;
|
||||
let mut body = test_item(&format!("Body{row}"), 530.0, y, 55.0, 10.0);
|
||||
body.page = page;
|
||||
items.extend([folio, footer, body]);
|
||||
}
|
||||
}
|
||||
|
||||
let page_one_items: Vec<_> = items
|
||||
.iter()
|
||||
.filter(|item| item.page == 1)
|
||||
.cloned()
|
||||
.collect();
|
||||
let (page_local_layout, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(page_one_items.clone(), 4);
|
||||
let page_local = compute_layout_complexity(&page_one_items, &page_local_layout, &[], &[]);
|
||||
assert!(
|
||||
page_local.pages_with_columns.contains(&1),
|
||||
"fixture must reproduce page-local folio column evidence"
|
||||
);
|
||||
|
||||
let selected =
|
||||
select_items_with_document_folio_context(items, 4, Some(&HashSet::from([1])));
|
||||
assert_eq!(
|
||||
selected
|
||||
.removal_mask
|
||||
.iter()
|
||||
.filter(|remove| **remove)
|
||||
.count(),
|
||||
8
|
||||
);
|
||||
let document_wide =
|
||||
compute_layout_complexity(&selected.items, &selected.layout_items, &[], &[]);
|
||||
assert!(!document_wide.pages_with_columns.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_detect_encoding_issues_fffd() {
|
||||
assert!(detect_encoding_issues(
|
||||
|
||||
@@ -374,6 +374,15 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
for line in lines {
|
||||
if let Some(first) = line.items.first() {
|
||||
if first.font_size / base_size >= 1.2 {
|
||||
// Digit-only lines (page numbers, issue numbers) must not
|
||||
// define heading tiers: a large bold folio claims tier 0 and
|
||||
// blocks the bold-size fallback for the document's real
|
||||
// same-size headings.
|
||||
let text = line.text();
|
||||
let t = text.trim();
|
||||
if !t.is_empty() && t.chars().all(|c| !c.is_alphabetic()) {
|
||||
continue;
|
||||
}
|
||||
heading_sizes.push(first.font_size);
|
||||
}
|
||||
}
|
||||
@@ -391,11 +400,45 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
}
|
||||
}
|
||||
|
||||
// Books often set section headings barely above body size (e.g. 11pt
|
||||
// bold over 10pt text). When nothing clears the 1.2x ratio gate, fall
|
||||
// back to bold lines modestly larger than body so those documents still
|
||||
// get an H1 instead of every bold heading defaulting to H2.
|
||||
if tiers.is_empty() {
|
||||
let mut bold_sizes: Vec<f32> = lines
|
||||
.iter()
|
||||
.filter(|line| {
|
||||
let text = line.text();
|
||||
let t = text.trim();
|
||||
!t.is_empty() && t.chars().any(|c| c.is_alphabetic())
|
||||
})
|
||||
.filter_map(|line| line.items.first())
|
||||
.filter(|it| it.is_bold && it.font_size / base_size >= 1.05)
|
||||
.map(|it| it.font_size)
|
||||
.collect();
|
||||
bold_sizes.sort_by(|a, b| b.total_cmp(a));
|
||||
for size in bold_sizes {
|
||||
if !tiers.iter().any(|&t| (t - size).abs() < 0.5) {
|
||||
tiers.push(size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Cap at 4 tiers
|
||||
tiers.truncate(4);
|
||||
tiers
|
||||
}
|
||||
|
||||
/// Boldness of a line judged by character mass, so a heading with an
|
||||
/// unbold section-number prefix ("4. " + bold title) still counts as bold.
|
||||
pub(crate) fn line_is_mostly_bold(line: &TextLine) -> bool {
|
||||
let (bold, total) = line.items.iter().fold((0usize, 0usize), |(b, t), it| {
|
||||
let n = it.text.trim().chars().count();
|
||||
(b + if it.is_bold { n } else { 0 }, t + n)
|
||||
});
|
||||
total > 0 && bold * 2 >= total
|
||||
}
|
||||
|
||||
/// Detect header level from font size using document-specific heading tiers.
|
||||
/// When tiers are available, maps tier 0→H1, tier 1→H2, etc.
|
||||
/// Falls back to ratio-based thresholds when no tiers exist.
|
||||
@@ -403,9 +446,21 @@ pub(crate) fn detect_header_level(
|
||||
font_size: f32,
|
||||
base_size: f32,
|
||||
heading_tiers: &[f32],
|
||||
is_bold: bool,
|
||||
) -> Option<usize> {
|
||||
let ratio = font_size / base_size;
|
||||
|
||||
// Tier matches are trusted below the 1.2x gate (down to 1.05x) only for
|
||||
// bold lines: sub-gate tiers come from the bold fallback, and honoring
|
||||
// them for non-bold text at the same size would promote captions.
|
||||
if (1.05..1.2).contains(&ratio) && is_bold && !heading_tiers.is_empty() {
|
||||
for (i, &tier_size) in heading_tiers.iter().enumerate() {
|
||||
if (font_size - tier_size).abs() < 0.5 {
|
||||
return Some(i + 1); // tier 0 → H1, tier 1 → H2, etc.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if ratio < 1.2 {
|
||||
return None; // Regular text
|
||||
}
|
||||
@@ -442,6 +497,75 @@ pub(crate) fn detect_header_level(
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn line_of(text: &str, font_size: f32, bold: bool, y: f32) -> crate::types::TextLine {
|
||||
let item = crate::types::TextItem {
|
||||
text: text.into(),
|
||||
x: 72.0,
|
||||
y,
|
||||
width: text.len() as f32 * font_size * 0.5,
|
||||
height: font_size,
|
||||
font: "Test".into(),
|
||||
font_size,
|
||||
page: 1,
|
||||
is_bold: bold,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
};
|
||||
crate::types::TextLine {
|
||||
items: vec![item],
|
||||
y,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn digit_only_lines_do_not_define_tiers() {
|
||||
// A 14pt bold page number must not claim tier 0 — that both demotes
|
||||
// every real heading a level and blocks the bold-size fallback.
|
||||
let lines = vec![
|
||||
line_of("76", 14.0, true, 760.0),
|
||||
line_of("Replace", 11.0, true, 700.0),
|
||||
line_of("body text at eleven points", 11.0, false, 680.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 11.0);
|
||||
assert!(tiers.is_empty(), "page number claimed a tier: {tiers:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_fallback_tiers_when_nothing_clears_ratio_gate() {
|
||||
// 10pt body, 11pt bold section headings (book-style): no size clears
|
||||
// 1.2x, so bold sizes modestly above body form the tiers.
|
||||
let lines = vec![
|
||||
line_of("4. Entropy", 11.0, true, 700.0),
|
||||
line_of("body text about entropy", 10.0, false, 680.0),
|
||||
line_of("5. The dynamics", 11.0, true, 500.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 10.0);
|
||||
assert_eq!(tiers, vec![11.0]);
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, true), Some(1));
|
||||
// Non-bold text at the fallback size must not become a heading.
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, false), None);
|
||||
// Non-tier body text stays regular.
|
||||
assert_eq!(detect_header_level(10.0, 10.0, &tiers, true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_fallback_skipped_when_real_tiers_exist() {
|
||||
let lines = vec![
|
||||
line_of("Chapter One", 18.0, false, 700.0),
|
||||
line_of("bold label", 11.0, true, 600.0),
|
||||
line_of("body", 10.0, false, 580.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 10.0);
|
||||
assert_eq!(tiers, vec![18.0]);
|
||||
// The 11pt bold label does not match any tier and stays non-heading.
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn toc_entry_with_single_dot_group() {
|
||||
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
|
||||
|
||||
+519
-97
@@ -1,5 +1,6 @@
|
||||
//! Core line-to-markdown conversion loop with table/image interleaving.
|
||||
|
||||
use std::cmp::Ordering;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use crate::structure_tree::StructRole;
|
||||
@@ -13,9 +14,149 @@ use super::analysis::{
|
||||
use super::classify::{
|
||||
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
||||
};
|
||||
use super::heading::classify_heading_sequences;
|
||||
use super::postprocess::clean_markdown;
|
||||
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
||||
use super::MarkdownOptions;
|
||||
use super::{item_is_in_chart_region, MarkdownOptions, CHART_SEPARATOR_PAD};
|
||||
|
||||
/// Logical stream geometry for a page where one full-width chart separates
|
||||
/// two prose columns. Positioned non-text blocks use this same ordering so a
|
||||
/// right-column table or image cannot jump ahead of left-column prose.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub(super) struct ChartProseOrder {
|
||||
split_x: f32,
|
||||
chart_region: (f32, f32, f32, f32),
|
||||
}
|
||||
|
||||
impl ChartProseOrder {
|
||||
pub(super) fn new(split_x: f32, chart_region: (f32, f32, f32, f32)) -> Self {
|
||||
Self {
|
||||
split_x,
|
||||
chart_region,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Markdown block with its physical position and optional logical chart-page
|
||||
/// stream. Tables and images share this representation because both are
|
||||
/// removed before text-line grouping and reinserted during conversion.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(super) struct PositionedMarkdown {
|
||||
y: f32,
|
||||
x: f32,
|
||||
markdown: String,
|
||||
chart_order: Option<ChartProseOrder>,
|
||||
}
|
||||
|
||||
impl PositionedMarkdown {
|
||||
pub(super) fn new(
|
||||
y: f32,
|
||||
x: f32,
|
||||
markdown: String,
|
||||
chart_order: Option<ChartProseOrder>,
|
||||
) -> Self {
|
||||
Self {
|
||||
y,
|
||||
x,
|
||||
markdown,
|
||||
chart_order,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn chart_stream_position(
|
||||
y: f32,
|
||||
x: f32,
|
||||
claimed_by_chart: bool,
|
||||
order: ChartProseOrder,
|
||||
) -> (u8, u8) {
|
||||
let (_, y0, _, y1) = order.chart_region;
|
||||
let low = y0.min(y1) - CHART_SEPARATOR_PAD;
|
||||
let high = y0.max(y1) + CHART_SEPARATOR_PAD;
|
||||
let in_chart_zone = claimed_by_chart || (y >= low && y <= high);
|
||||
let zone = if in_chart_zone {
|
||||
1
|
||||
} else if y > high {
|
||||
0
|
||||
} else {
|
||||
2
|
||||
};
|
||||
let column = if in_chart_zone || x < order.split_x {
|
||||
0
|
||||
} else {
|
||||
1
|
||||
};
|
||||
(zone, column)
|
||||
}
|
||||
|
||||
fn positioned_block_precedes_line(block: &PositionedMarkdown, line: &TextLine) -> bool {
|
||||
let Some(order) = block.chart_order else {
|
||||
return block.y > line.y;
|
||||
};
|
||||
let line_x = line.items.first().map(|item| item.x).unwrap_or(0.0);
|
||||
let line_claimed_by_chart = line
|
||||
.items
|
||||
.iter()
|
||||
.any(|item| item_is_in_chart_region(item, &[order.chart_region]));
|
||||
let block_position = chart_stream_position(block.y, block.x, false, order);
|
||||
let line_position = chart_stream_position(line.y, line_x, line_claimed_by_chart, order);
|
||||
block_position < line_position || (block_position == line_position && block.y > line.y)
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
||||
enum PositionedBlockKind {
|
||||
Table,
|
||||
Image,
|
||||
}
|
||||
|
||||
type PositionedBlockRef<'a> = (PositionedBlockKind, usize, &'a PositionedMarkdown);
|
||||
|
||||
fn compare_positioned_blocks(
|
||||
(a_kind, a_idx, a): &PositionedBlockRef<'_>,
|
||||
(b_kind, b_idx, b): &PositionedBlockRef<'_>,
|
||||
) -> Ordering {
|
||||
if let (Some(a_order), Some(b_order)) = (a.chart_order, b.chart_order) {
|
||||
let a_position = chart_stream_position(a.y, a.x, false, a_order);
|
||||
let b_position = chart_stream_position(b.y, b.x, false, b_order);
|
||||
return a_position
|
||||
.cmp(&b_position)
|
||||
.then_with(|| b.y.total_cmp(&a.y))
|
||||
.then_with(|| a.x.total_cmp(&b.x))
|
||||
.then_with(|| a_kind.cmp(b_kind))
|
||||
.then_with(|| a_idx.cmp(b_idx));
|
||||
}
|
||||
|
||||
// Preserve the legacy ordering for ordinary pages: tables in detection
|
||||
// order, followed by images in input order. Chart pages give every block
|
||||
// a chart order and use the logical stream comparison above.
|
||||
a_kind.cmp(b_kind).then_with(|| a_idx.cmp(b_idx))
|
||||
}
|
||||
|
||||
fn positioned_blocks_for_page<'a>(
|
||||
page: u32,
|
||||
page_tables: &'a HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
page_images: &'a HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
) -> Vec<PositionedBlockRef<'a>> {
|
||||
let mut blocks = Vec::new();
|
||||
if let Some(tables) = page_tables.get(&page) {
|
||||
blocks.extend(
|
||||
tables
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(idx, table)| (PositionedBlockKind::Table, idx, table)),
|
||||
);
|
||||
}
|
||||
if let Some(images) = page_images.get(&page) {
|
||||
blocks.extend(
|
||||
images
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(idx, image)| (PositionedBlockKind::Image, idx, image)),
|
||||
);
|
||||
}
|
||||
blocks.sort_by(compare_positioned_blocks);
|
||||
blocks
|
||||
}
|
||||
|
||||
/// Pre-scan struct heading tags to find levels that are overused — i.e., tagged on
|
||||
/// so many lines that they clearly represent body text, not real headings.
|
||||
@@ -160,6 +301,112 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
|
||||
/// wrapped visual line as "standalone" once the first line is misclassified,
|
||||
/// producing a stack of `##` headings. Multi-line body-size bold runs with a
|
||||
/// paragraph-sized word count should stay paragraph text.
|
||||
/// Merge 2-3 consecutive all-bold body-size lines into one line when the
|
||||
/// group is isolated (paragraph break before and after) and short enough to
|
||||
/// be a heading. Longer/wordier bold runs are wrapped bold paragraphs and
|
||||
/// are left for `find_wrapped_bold_paragraph_lines` to suppress.
|
||||
/// "9.5. ", "12.3.1. " — section-numbered heading prefix followed by a word.
|
||||
fn starts_with_section_number(t: &str) -> bool {
|
||||
let t = t.trim_start();
|
||||
let mut rest = t;
|
||||
let mut groups = 0;
|
||||
loop {
|
||||
let digits = rest.chars().take_while(|c| c.is_ascii_digit()).count();
|
||||
if digits == 0 || digits > 3 {
|
||||
break;
|
||||
}
|
||||
groups += 1;
|
||||
rest = &rest[digits..];
|
||||
if let Some(r) = rest.strip_prefix('.') {
|
||||
rest = r;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Two components minimum ("9.5. "): a single "1. " is an ordered list
|
||||
// item, and this prefix bypasses the isolation checks entirely.
|
||||
groups >= 2
|
||||
&& rest.starts_with(char::is_whitespace)
|
||||
&& rest.trim_start().starts_with(|c: char| c.is_alphabetic())
|
||||
}
|
||||
|
||||
fn merge_wrapped_bold_heading_groups(
|
||||
lines: Vec<TextLine>,
|
||||
base_size: f32,
|
||||
para_threshold: f32,
|
||||
) -> Vec<TextLine> {
|
||||
let mut out: Vec<TextLine> = Vec::with_capacity(lines.len());
|
||||
let mut i = 0usize;
|
||||
while i < lines.len() {
|
||||
if !is_body_size_all_bold_line(&lines[i], base_size) {
|
||||
out.push(lines[i].clone());
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
let start = i;
|
||||
let mut end = i;
|
||||
let mut word_count = lines[i].text().split_whitespace().count();
|
||||
while end + 1 < lines.len()
|
||||
&& is_body_size_all_bold_line(&lines[end + 1], base_size)
|
||||
&& is_wrapped_same_style_line(&lines[end], &lines[end + 1], para_threshold)
|
||||
{
|
||||
end += 1;
|
||||
word_count += lines[end].text().split_whitespace().count();
|
||||
}
|
||||
let line_count = end - start + 1;
|
||||
// Column-local isolation: on interleaved multi-column pages the
|
||||
// vector neighbors may be the other column's lines, so judge the
|
||||
// break by x-overlapping lines only.
|
||||
let gx0 = lines[start..=end]
|
||||
.iter()
|
||||
.flat_map(|l| l.items.iter().map(|i| i.x))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let gx1 = lines[start..=end]
|
||||
.iter()
|
||||
.flat_map(|l| l.items.iter().map(|i| i.x + i.width))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let overlaps_x = |l: &TextLine| {
|
||||
let lx0 = l.items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let lx1 = l
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| i.x + i.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
lx0 < gx1 && lx1 > gx0
|
||||
};
|
||||
let page = lines[start].page;
|
||||
let break_before = !lines.iter().any(|l| {
|
||||
l.page == page
|
||||
&& l.y > lines[start].y
|
||||
&& l.y - lines[start].y <= para_threshold
|
||||
&& overlaps_x(l)
|
||||
});
|
||||
let break_after = !lines.iter().any(|l| {
|
||||
l.page == page
|
||||
&& l.y < lines[end].y
|
||||
&& lines[end].y - l.y <= para_threshold
|
||||
&& overlaps_x(l)
|
||||
});
|
||||
let numbered = starts_with_section_number(&lines[start].text());
|
||||
if (2..=3).contains(&line_count)
|
||||
&& word_count <= 15
|
||||
&& ((break_before && break_after) || numbered)
|
||||
{
|
||||
let mut merged = lines[start].clone();
|
||||
for l in &lines[start + 1..=end] {
|
||||
merged.items.extend(l.items.iter().cloned());
|
||||
}
|
||||
out.push(merged);
|
||||
} else {
|
||||
for l in &lines[start..=end] {
|
||||
out.push(l.clone());
|
||||
}
|
||||
}
|
||||
i = end + 1;
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn find_wrapped_bold_paragraph_lines(
|
||||
lines: &[TextLine],
|
||||
base_size: f32,
|
||||
@@ -277,7 +524,7 @@ fn struct_role_heading_level(role: &StructRole) -> Option<usize> {
|
||||
/// We strip their header+separator rows and append their data rows to the first page's
|
||||
/// table, then remove them from later pages.
|
||||
pub(super) fn merge_continuation_tables(
|
||||
page_tables: &mut std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_tables: &mut std::collections::HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
table_only_pages: &HashSet<u32>,
|
||||
) {
|
||||
let mut sorted_pages: Vec<u32> = page_tables.keys().copied().collect();
|
||||
@@ -305,7 +552,7 @@ pub(super) fn merge_continuation_tables(
|
||||
continue;
|
||||
}
|
||||
|
||||
let first_col_count = count_table_columns(&first_tables[0].1);
|
||||
let first_col_count = count_table_columns(&first_tables[0].markdown);
|
||||
if first_col_count == 0 {
|
||||
i += 1;
|
||||
continue;
|
||||
@@ -336,7 +583,7 @@ pub(super) fn merge_continuation_tables(
|
||||
_ => break,
|
||||
};
|
||||
|
||||
let next_col_count = count_table_columns(&next_tables[0].1);
|
||||
let next_col_count = count_table_columns(&next_tables[0].markdown);
|
||||
if next_col_count != first_col_count {
|
||||
break;
|
||||
}
|
||||
@@ -350,7 +597,7 @@ pub(super) fn merge_continuation_tables(
|
||||
let mut extra_rows = String::new();
|
||||
for &cont_page in &continuation_pages {
|
||||
if let Some(tables) = page_tables.get(&cont_page) {
|
||||
let table_md = &tables[0].1;
|
||||
let table_md = &tables[0].markdown;
|
||||
// Skip header row (line 1) and separator row (line 2), keep the rest
|
||||
for (line_idx, line) in table_md.lines().enumerate() {
|
||||
if line_idx >= 2 {
|
||||
@@ -363,7 +610,7 @@ pub(super) fn merge_continuation_tables(
|
||||
|
||||
// Append continuation rows to the first page's table
|
||||
if let Some(tables) = page_tables.get_mut(&first_page) {
|
||||
tables[0].1.push_str(&extra_rows);
|
||||
tables[0].markdown.push_str(&extra_rows);
|
||||
}
|
||||
|
||||
// Remove continuation pages from the map
|
||||
@@ -395,37 +642,35 @@ fn count_table_columns(table_md: &str) -> usize {
|
||||
/// Flush any remaining tables and images for a given page
|
||||
fn flush_page_tables_and_images(
|
||||
page: u32,
|
||||
page_tables: &std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_images: &std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_blocks: &HashMap<u32, Vec<PositionedBlockRef<'_>>>,
|
||||
inserted_tables: &mut HashSet<(u32, usize)>,
|
||||
inserted_images: &mut HashSet<(u32, usize)>,
|
||||
output: &mut String,
|
||||
in_paragraph: &mut bool,
|
||||
) {
|
||||
if let Some(tables) = page_tables.get(&page) {
|
||||
for (idx, (_, table_md)) in tables.iter().enumerate() {
|
||||
if !inserted_tables.contains(&(page, idx)) {
|
||||
if *in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
*in_paragraph = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(table_md);
|
||||
output.push('\n');
|
||||
let Some(blocks) = page_blocks.get(&page) else {
|
||||
return;
|
||||
};
|
||||
for &(kind, idx, block) in blocks {
|
||||
let already_inserted = match kind {
|
||||
PositionedBlockKind::Table => inserted_tables.contains(&(page, idx)),
|
||||
PositionedBlockKind::Image => inserted_images.contains(&(page, idx)),
|
||||
};
|
||||
if already_inserted {
|
||||
continue;
|
||||
}
|
||||
if *in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
*in_paragraph = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(&block.markdown);
|
||||
output.push('\n');
|
||||
match kind {
|
||||
PositionedBlockKind::Table => {
|
||||
inserted_tables.insert((page, idx));
|
||||
}
|
||||
}
|
||||
}
|
||||
if let Some(images) = page_images.get(&page) {
|
||||
for (idx, (_, image_md)) in images.iter().enumerate() {
|
||||
if !inserted_images.contains(&(page, idx)) {
|
||||
if *in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
*in_paragraph = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(image_md);
|
||||
output.push('\n');
|
||||
PositionedBlockKind::Image => {
|
||||
inserted_images.insert((page, idx));
|
||||
}
|
||||
}
|
||||
@@ -436,8 +681,9 @@ fn flush_page_tables_and_images(
|
||||
pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
lines: Vec<TextLine>,
|
||||
options: MarkdownOptions,
|
||||
page_tables: std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_images: std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_tables: std::collections::HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
page_images: std::collections::HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
page_chart_regions: &std::collections::HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
band_split_pages: &HashSet<u32>,
|
||||
struct_roles: Option<
|
||||
&std::collections::HashMap<u32, std::collections::HashMap<i64, StructRole>>,
|
||||
@@ -468,6 +714,17 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// threshold and cause every line to be treated as a paragraph break.
|
||||
let para_threshold = compute_paragraph_threshold(&lines, base_size);
|
||||
|
||||
// Merge wrapped bold headings: a 2-3 line group of consecutive all-bold
|
||||
// body-size lines that is isolated as a group (paragraph break before
|
||||
// and after) is one heading that wrapped. Left split, the internal line
|
||||
// gap breaks each line's isolation and neither classifies as a heading —
|
||||
// the whole group then merges into the following body paragraph.
|
||||
let lines = if std::env::var("PI_NO_MERGE").is_ok() {
|
||||
lines
|
||||
} else {
|
||||
merge_wrapped_bold_heading_groups(lines, base_size, para_threshold)
|
||||
};
|
||||
|
||||
// Pre-scan: identify isolated lines (paragraph break before AND after).
|
||||
// These are heading candidates even without bold/large font — common in
|
||||
// academic papers where section titles like "Acknowledgements" sit alone
|
||||
@@ -477,6 +734,33 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
let wrapped_bold_paragraph_lines =
|
||||
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
|
||||
|
||||
let mut sequence_excluded_lines = wrapped_bold_paragraph_lines.clone();
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
if page_chart_regions.get(&line.page).is_some_and(|regions| {
|
||||
line.items
|
||||
.iter()
|
||||
.any(|item| item_is_in_chart_region(item, regions))
|
||||
}) {
|
||||
sequence_excluded_lines.insert(line_idx);
|
||||
}
|
||||
}
|
||||
if let Some(roles) = struct_roles {
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
if resolve_line_struct_role(line, roles)
|
||||
.is_some_and(|role| role.is_non_heading_content())
|
||||
{
|
||||
sequence_excluded_lines.insert(line_idx);
|
||||
}
|
||||
}
|
||||
}
|
||||
let sequence_heading_levels = classify_heading_sequences(
|
||||
&lines,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
&isolated_lines,
|
||||
&sequence_excluded_lines,
|
||||
);
|
||||
|
||||
// Detect struct heading levels that are overused (body text mistagged as headings)
|
||||
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
|
||||
|
||||
@@ -502,6 +786,18 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
.collect();
|
||||
all_content_pages.sort();
|
||||
all_content_pages.dedup();
|
||||
// Build the unified table/image order once per page. This is only a
|
||||
// meaningful sort on chart/prose pages; ordinary pages retain their
|
||||
// legacy table-then-image order without repeating work for every line.
|
||||
let page_blocks: HashMap<u32, Vec<PositionedBlockRef<'_>>> = all_content_pages
|
||||
.iter()
|
||||
.map(|&page| {
|
||||
(
|
||||
page,
|
||||
positioned_blocks_for_page(page, &page_tables, &page_images),
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
// Page break
|
||||
@@ -514,8 +810,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
current_page,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -539,8 +834,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
p,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -563,38 +857,32 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we should insert a table before this line
|
||||
if let Some(tables) = page_tables.get(¤t_page) {
|
||||
for (idx, (table_y, table_md)) in tables.iter().enumerate() {
|
||||
// Insert table when we pass its Y position
|
||||
if *table_y > line.y && !inserted_tables.contains(&(current_page, idx)) {
|
||||
// Insert tables and images through one ordered stream. Chart/prose
|
||||
// pages sort by zone, column, and physical Y; ordinary pages retain
|
||||
// the legacy table-then-image input order.
|
||||
if let Some(blocks) = page_blocks.get(¤t_page) {
|
||||
for &(kind, idx, block) in blocks {
|
||||
let already_inserted = match kind {
|
||||
PositionedBlockKind::Table => inserted_tables.contains(&(current_page, idx)),
|
||||
PositionedBlockKind::Image => inserted_images.contains(&(current_page, idx)),
|
||||
};
|
||||
if positioned_block_precedes_line(block, line) && !already_inserted {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(table_md);
|
||||
output.push_str(&block.markdown);
|
||||
output.push('\n');
|
||||
inserted_tables.insert((current_page, idx));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we should insert an image before this line
|
||||
if let Some(images) = page_images.get(¤t_page) {
|
||||
for (idx, (image_y, image_md)) in images.iter().enumerate() {
|
||||
// Insert image when we pass its Y position
|
||||
if *image_y > line.y && !inserted_images.contains(&(current_page, idx)) {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
match kind {
|
||||
PositionedBlockKind::Table => {
|
||||
inserted_tables.insert((current_page, idx));
|
||||
}
|
||||
PositionedBlockKind::Image => {
|
||||
inserted_images.insert((current_page, idx));
|
||||
}
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(image_md);
|
||||
output.push('\n');
|
||||
inserted_images.insert((current_page, idx));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -721,7 +1009,13 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
&& toc_suppress_page != Some(line.page)
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
detect_header_level(
|
||||
line_font_size,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
.or_else(|| {
|
||||
// Rarity-based heading detection (inspired by opendataloader).
|
||||
// Heading probability scoring with lookahead context.
|
||||
// Score = rarity * 0.5 + bold * 0.3 + standalone * 0.2
|
||||
@@ -754,16 +1048,23 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// paragraph continuity and minor font-size variation
|
||||
// inflates rarity scores.
|
||||
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
|
||||
// Single-word headings ("IMPLEMENTATION", "CONTENTS") are common;
|
||||
// accept them only with the strongest signal combination.
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words && has_strong_signal {
|
||||
// Single-word headings ("IMPLEMENTATION", "CONTENTS",
|
||||
// "Replace") are common. All-bold single words qualify when
|
||||
// standalone (paragraph break before / page top) — headings
|
||||
// hug their section's first paragraph, so requiring a break
|
||||
// after as well missed most of them. Mixed bold lead-ins
|
||||
// ("Note: ...") are excluded by all_bold.
|
||||
let enough_words = word_count >= 2 || (all_bold && plain_trimmed.len() >= 4);
|
||||
let numbered_bold = all_bold && starts_with_section_number(plain_trimmed);
|
||||
if numbered_bold
|
||||
|| (score >= 0.5 && standalone && enough_words && has_strong_signal)
|
||||
{
|
||||
Some(bold_heading_level(&heading_tiers))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
.or_else(|| sequence_heading_levels.get(&line_idx).copied())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
@@ -915,8 +1216,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// (handles table-only pages after the last text line, and trailing image-only pages)
|
||||
flush_page_tables_and_images(
|
||||
current_page,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -928,8 +1228,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
p,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -973,6 +1272,13 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||
let wrapped_bold_paragraph_lines =
|
||||
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
|
||||
let sequence_heading_levels = classify_heading_sequences(
|
||||
&lines,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
&isolated_lines,
|
||||
&wrapped_bold_paragraph_lines,
|
||||
);
|
||||
|
||||
let mut output = String::new();
|
||||
let mut current_page = 0u32;
|
||||
@@ -1067,33 +1373,39 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
&& !(options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)))
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
if let Some(header_level) =
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
if line_font_size < base_size * 0.95 {
|
||||
return None;
|
||||
}
|
||||
let word_count = plain_trimmed.split_whitespace().count();
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
if wrapped_bold_paragraph_lines.contains(&line_idx) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
let isolated = isolated_lines.contains(&line_idx);
|
||||
let score = rarity * 0.5
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words {
|
||||
return Some(bold_heading_level(&heading_tiers));
|
||||
}
|
||||
None
|
||||
})
|
||||
if let Some(header_level) = detect_header_level(
|
||||
line_font_size,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
.or_else(|| {
|
||||
if line_font_size < base_size * 0.95 {
|
||||
return None;
|
||||
}
|
||||
let word_count = plain_trimmed.split_whitespace().count();
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
if wrapped_bold_paragraph_lines.contains(&line_idx) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
let isolated = isolated_lines.contains(&line_idx);
|
||||
let score = rarity * 0.5
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words {
|
||||
return Some(bold_heading_level(&heading_tiers));
|
||||
}
|
||||
None
|
||||
})
|
||||
.or_else(|| sequence_heading_levels.get(&line_idx).copied())
|
||||
{
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
@@ -1204,6 +1516,20 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn section_number_prefix_detection() {
|
||||
assert!(starts_with_section_number(
|
||||
"9.5. Adapting to the New Normal"
|
||||
));
|
||||
assert!(starts_with_section_number("12.3.1. Deep subsection"));
|
||||
assert!(starts_with_section_number("2.1 Systems thinking"));
|
||||
assert!(!starts_with_section_number("1. First item in a list"));
|
||||
assert!(!starts_with_section_number("24% in October 2020."));
|
||||
assert!(!starts_with_section_number("2020 was a hard year"));
|
||||
assert!(!starts_with_section_number("Introduction"));
|
||||
}
|
||||
|
||||
use super::*;
|
||||
use crate::structure_tree::StructRole;
|
||||
use crate::types::TextItem;
|
||||
@@ -1245,6 +1571,91 @@ mod tests {
|
||||
make_line(vec![item])
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_page_blocks_follow_zone_and_column_stream() {
|
||||
let line = |text: &str, x: f32, y: f32| {
|
||||
let mut item = make_item(text, 1, None);
|
||||
item.x = x;
|
||||
item.y = y;
|
||||
make_line(vec![item])
|
||||
};
|
||||
// Logical newspaper order for the prose zone: all left-column lines,
|
||||
// then all right-column lines, even though their physical Y values
|
||||
// jump back upward at the column switch.
|
||||
let lines = vec![
|
||||
line("Left column upper prose.", 90.0, 700.0),
|
||||
line("Left column lower prose.", 90.0, 500.0),
|
||||
line("Right column upper prose.", 340.0, 700.0),
|
||||
line("Right column lower prose.", 340.0, 500.0),
|
||||
];
|
||||
let order = ChartProseOrder::new(280.0, (100.0, 300.0, 500.0, 400.0));
|
||||
let mut tables = HashMap::new();
|
||||
tables.insert(
|
||||
1,
|
||||
vec![
|
||||
// Detection order is deliberately right before left.
|
||||
PositionedMarkdown::new(
|
||||
600.0,
|
||||
340.0,
|
||||
"| Right metric | Value |\n|---|---|\n| A | 1 |\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
PositionedMarkdown::new(
|
||||
550.0,
|
||||
90.0,
|
||||
"| Left metric | Value |\n|---|---|\n| B | 2 |\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
],
|
||||
);
|
||||
let mut images = HashMap::new();
|
||||
images.insert(
|
||||
1,
|
||||
vec\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
PositionedMarkdown::new(
|
||||
575.0,
|
||||
340.0,
|
||||
"\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
],
|
||||
);
|
||||
|
||||
let md = to_markdown_from_lines_with_tables_and_images(
|
||||
lines,
|
||||
MarkdownOptions::default(),
|
||||
tables,
|
||||
images,
|
||||
&HashMap::new(),
|
||||
&HashSet::from([1]),
|
||||
None,
|
||||
);
|
||||
let positions = [
|
||||
"Left column upper prose.",
|
||||
"",
|
||||
"| Left metric | Value |",
|
||||
"Left column lower prose.",
|
||||
"Right column upper prose.",
|
||||
"| Right metric | Value |",
|
||||
"",
|
||||
"Right column lower prose.",
|
||||
]
|
||||
.map(|needle| {
|
||||
md.find(needle)
|
||||
.unwrap_or_else(|| panic!("missing {needle:?} in {md}"))
|
||||
});
|
||||
assert!(
|
||||
positions.windows(2).all(|pair| pair[0] < pair[1]),
|
||||
"blocks must follow the logical chart-page stream: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn isolated_lines_kept_on_sparse_pages() {
|
||||
// A ToC page with a lone title and one entry far below: the density
|
||||
@@ -1297,6 +1708,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1325,6 +1737,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1366,6 +1779,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1402,6 +1816,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1437,6 +1852,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1466,6 +1882,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
@@ -1507,6 +1924,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1555,6 +1973,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
@@ -1656,6 +2075,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
@@ -1703,6 +2123,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1842,6 +2263,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1361
-97
File diff suppressed because it is too large
Load Diff
+43
-67
@@ -2,12 +2,16 @@
|
||||
|
||||
use regex::Regex;
|
||||
|
||||
use super::MarkdownOptions;
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
use crate::text_utils::is_page_number_line;
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
// Collapse dot leaders (e.g. TOC entries: "Introduction...............................1")
|
||||
text = collapse_dot_leaders(&text);
|
||||
if options.profile == MarkdownProfile::Compact {
|
||||
// Dot-leader collapse saves tokens but changes source text, so it is
|
||||
// reserved for the explicit compact profile.
|
||||
text = collapse_dot_leaders(&text);
|
||||
}
|
||||
|
||||
// Fix hyphenation first (before other processing)
|
||||
if options.fix_hyphenation {
|
||||
@@ -142,7 +146,7 @@ fn fix_hyphenation(text: &str) -> String {
|
||||
result
|
||||
}
|
||||
|
||||
/// Remove standalone page numbers (lines that are just 1-4 digit numbers)
|
||||
/// Remove isolated page-number expressions from Markdown.
|
||||
fn remove_page_numbers(text: &str) -> String {
|
||||
let mut result = Vec::new();
|
||||
let lines: Vec<&str> = text.lines().collect();
|
||||
@@ -180,69 +184,6 @@ fn remove_page_numbers(text: &str) -> String {
|
||||
result.join("\n")
|
||||
}
|
||||
|
||||
/// Check if a line looks like a page number
|
||||
fn is_page_number_line(trimmed: &str) -> bool {
|
||||
// Empty lines are not page numbers
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Pattern 1: Just a number (1-4 digits)
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Pattern 2: "Page X of Y" or "Page X" or "Page of" (placeholder)
|
||||
let lower = trimmed.to_lowercase();
|
||||
if let Some(rest) = lower.strip_prefix("page") {
|
||||
let rest = rest.trim();
|
||||
// "Page of" (empty page numbers)
|
||||
if rest == "of" || rest.starts_with("of ") {
|
||||
return true;
|
||||
}
|
||||
// "Page X" or "Page X of Y"
|
||||
if rest
|
||||
.chars()
|
||||
.next()
|
||||
.map(|c| c.is_ascii_digit())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Just "Page" followed by whitespace and maybe "of"
|
||||
if rest.is_empty()
|
||||
|| rest
|
||||
.split_whitespace()
|
||||
.all(|w| w == "of" || w.chars().all(|c| c.is_ascii_digit()))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 3: "X of Y" where X and Y are numbers
|
||||
if let Some(of_idx) = trimmed.find(" of ") {
|
||||
let before = trimmed[..of_idx].trim();
|
||||
let after = trimmed[of_idx + 4..].trim();
|
||||
if before.chars().all(|c| c.is_ascii_digit())
|
||||
&& after.chars().all(|c| c.is_ascii_digit())
|
||||
&& !before.is_empty()
|
||||
&& !after.is_empty()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 4: "- X -" centered page number
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if inner.chars().all(|c| c.is_ascii_digit()) && !inner.is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Convert URLs to markdown links
|
||||
fn format_urls(text: &str) -> String {
|
||||
use once_cell::sync::Lazy;
|
||||
@@ -355,6 +296,23 @@ fn format_urls(text: &str) -> String {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn fidelity_profile_preserves_dot_leaders() {
|
||||
let input = "Introduction............................1".to_string();
|
||||
let result = clean_markdown(input.clone(), &MarkdownOptions::default());
|
||||
assert_eq!(result, format!("{input}\n"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compact_profile_collapses_dot_leaders() {
|
||||
let input = "Introduction............................1".to_string();
|
||||
let options = MarkdownOptions {
|
||||
profile: MarkdownProfile::Compact,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
assert_eq!(clean_markdown(input, &options), "Introduction ... 1\n");
|
||||
}
|
||||
|
||||
// --- collapse_dot_leaders ---
|
||||
|
||||
#[test]
|
||||
@@ -469,12 +427,14 @@ mod tests {
|
||||
fn test_is_page_number_page_x() {
|
||||
assert!(is_page_number_line("Page 5"));
|
||||
assert!(is_page_number_line("page 12"));
|
||||
assert!(is_page_number_line("Page123"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_page_x_of_y() {
|
||||
assert!(is_page_number_line("Page 3 of 10"));
|
||||
assert!(is_page_number_line("page 1 of 5"));
|
||||
assert!(is_page_number_line("Page 3 of 10 Report header"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -506,6 +466,12 @@ mod tests {
|
||||
assert!(!is_page_number_line("Total: 500"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_labeled_running_header() {
|
||||
assert!(is_page_number_line("Page 42 Chapter 5"));
|
||||
assert!(is_page_number_line("Page 42 explains the result"));
|
||||
}
|
||||
|
||||
// --- remove_page_numbers ---
|
||||
|
||||
#[test]
|
||||
@@ -531,6 +497,16 @@ mod tests {
|
||||
assert!(result.contains("42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_labeled_header_with_content() {
|
||||
let input = "Content\n\nPage 42 explains the result\n---\nEnd";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert!(!result.contains("Page 42 explains the result"));
|
||||
assert!(result.contains("Content"));
|
||||
assert!(result.contains("End"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_multiple_patterns() {
|
||||
let input = "\n1\n\nContent\n\n2\n\n---\nMore\n\n3\n";
|
||||
|
||||
@@ -42,7 +42,12 @@ fn effective_heading_level(
|
||||
|
||||
// Fall back to font-size heuristic
|
||||
let font = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(font, base_size, heading_tiers)
|
||||
detect_header_level(
|
||||
font,
|
||||
base_size,
|
||||
heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
}
|
||||
|
||||
/// Merge consecutive heading lines at the same level into a single line.
|
||||
|
||||
@@ -914,7 +914,126 @@ fn looks_like_number(s: &str) -> bool {
|
||||
///
|
||||
/// Used by format.rs to render TOCs as flat lists instead of markdown tables.
|
||||
pub fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells)
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells) || is_page_number_toc(cells)
|
||||
}
|
||||
|
||||
/// Parse a page-number-like token: a short arabic integer (≤4 digits) or a
|
||||
/// canonical roman numeral (front-matter pages: i, ii, …, xxxviii). Roman
|
||||
/// parsing is shared with the formatter via `super::canonical_roman_value` so
|
||||
/// the two stay in sync.
|
||||
fn page_number_value(token: &str) -> Option<u32> {
|
||||
let t = token.trim();
|
||||
if t.is_empty() {
|
||||
return None;
|
||||
}
|
||||
if t.chars().all(|c| c.is_ascii_digit()) && t.len() <= 4 {
|
||||
return t.parse().ok();
|
||||
}
|
||||
super::canonical_roman_value(t)
|
||||
}
|
||||
|
||||
/// Page-number-column TOC: title-based contents with no dot leaders and no
|
||||
/// section numbers (e.g. "About the Publisher vii", "Experiment #1 … 3").
|
||||
/// The signature is a text-title first column and a last column that is almost
|
||||
/// entirely page numbers whose values are *mostly non-decreasing* — the
|
||||
/// monotonic run is what separates a real TOC from an incidental 2-column
|
||||
/// numeric data table.
|
||||
pub(super) fn is_page_number_toc(cells: &[Vec<String>]) -> bool {
|
||||
let num_cols = cells.first().map(|r| r.len()).unwrap_or(0);
|
||||
// A page-number TOC is a narrow list (title + page, optionally a leader
|
||||
// column). Wider grids are data tables, not contents.
|
||||
if !(2..=3).contains(&num_cols) || cells.len() < 5 {
|
||||
return false;
|
||||
}
|
||||
let last = num_cols - 1;
|
||||
|
||||
// No header row: a TOC's first row is already an entry, so its last cell is
|
||||
// a page number. A data table's first row is a column header (non-numeric,
|
||||
// or an empty units cell like "Category | ") — the tell that separates
|
||||
// "Mineral | CEC" tables from real contents. Check the actual first row,
|
||||
// not the first non-empty one, so a blank header cell still rejects.
|
||||
let first_last = cells[0].get(last).map(|s| s.trim()).unwrap_or("");
|
||||
if page_number_value(first_last).is_none() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Last column: page numbers on ≥70% of filled rows; collect their values.
|
||||
let mut filled = 0u32;
|
||||
let mut page_vals: Vec<u32> = Vec::new();
|
||||
for row in cells {
|
||||
let cell = row.get(last).map(|s| s.trim()).unwrap_or("");
|
||||
if cell.is_empty() {
|
||||
continue;
|
||||
}
|
||||
filled += 1;
|
||||
if let Some(v) = page_number_value(cell) {
|
||||
page_vals.push(v);
|
||||
}
|
||||
}
|
||||
if filled < 4 || (page_vals.len() as f32) < 0.7 * filled as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// First column: mostly text titles (has alphabetic content). This rejects
|
||||
// numeric-vs-numeric grids.
|
||||
let text_first = cells
|
||||
.iter()
|
||||
.filter(|row| {
|
||||
row.first()
|
||||
.is_some_and(|c| c.chars().any(|ch| ch.is_alphabetic()))
|
||||
})
|
||||
.count();
|
||||
if (text_first as f32) < 0.6 * cells.len() as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Page numbers mostly ascend (allow front-matter→body resets and noise).
|
||||
if page_vals.len() < 2 {
|
||||
return false;
|
||||
}
|
||||
let non_decreasing = page_vals.windows(2).filter(|w| w[1] >= w[0]).count();
|
||||
if (non_decreasing as f32) < 0.7 * (page_vals.len() - 1) as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Stronger TOC signal. Real page numbers SPAN the document — entries skip
|
||||
// (3, 6, 13, 24, …) so their range exceeds the entry count. A rank / ID /
|
||||
// ordinal column is instead a *perfectly dense* consecutive run (1,2,3,… or
|
||||
// 100,101,102,…). Accept anything with page gaps; for a dense run — which a
|
||||
// one-page-per-entry TOC can also produce — fall back to a title signal:
|
||||
// real contents entries are multi-word headings, rank labels are short.
|
||||
let min = *page_vals.iter().min().unwrap();
|
||||
let max = *page_vals.iter().max().unwrap();
|
||||
let span = max.saturating_sub(min);
|
||||
if span > page_vals.len() as u32 {
|
||||
return true;
|
||||
}
|
||||
let dense_consecutive = (span as usize) + 1 == page_vals.len() && {
|
||||
let mut sorted = page_vals.clone();
|
||||
sorted.sort_unstable();
|
||||
sorted.dedup();
|
||||
sorted.len() == page_vals.len()
|
||||
};
|
||||
if !dense_consecutive {
|
||||
// Narrow range but with a gap or repeat — still contents-like.
|
||||
return true;
|
||||
}
|
||||
// Dense counter: only a TOC if the titles read like headings, not the
|
||||
// short single-word labels typical of rank/leaderboard/ID tables.
|
||||
let (total_words, titled_rows) = cells
|
||||
.iter()
|
||||
.filter_map(|row| row.first())
|
||||
.filter(|c| c.chars().any(|ch| ch.is_alphabetic()))
|
||||
.fold((0usize, 0usize), |(w, n), c| {
|
||||
(
|
||||
w + c
|
||||
.split_whitespace()
|
||||
.filter(|t| t.chars().any(|ch| ch.is_alphabetic()))
|
||||
.count(),
|
||||
n + 1,
|
||||
)
|
||||
});
|
||||
titled_rows > 0 && (total_words as f32) / titled_rows as f32 >= 1.8
|
||||
}
|
||||
|
||||
/// Dot-leader TOC: any "Chapter 1 ........ 42" style with explicit leader
|
||||
@@ -1886,4 +2005,166 @@ mod tests {
|
||||
assert!(!starts_with_section_number(""));
|
||||
assert!(!starts_with_section_number("Hello world"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_value_rejects_roman_lookalike_words() {
|
||||
// Ordinary words made only of {i,v,x,l,c} are not page numbers.
|
||||
assert!(page_number_value("civil").is_none());
|
||||
assert!(page_number_value("mix").is_none());
|
||||
assert!(page_number_value("ill").is_none());
|
||||
assert!(page_number_value("lil").is_none());
|
||||
// Canonical roman numerals still parse.
|
||||
assert_eq!(page_number_value("vii"), Some(7));
|
||||
assert_eq!(page_number_value("ix"), Some(9));
|
||||
assert_eq!(page_number_value("xii"), Some(12));
|
||||
assert_eq!(page_number_value("42"), Some(42));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_matches_consecutive_pages_with_titles() {
|
||||
// A short chapter-per-page contents: pages are a dense 1..n run, but
|
||||
// the multi-word titles mark it as a real TOC (recovered by the title
|
||||
// signal rather than rejected for lacking page gaps).
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Introduction to the Study".into(), "1".into()],
|
||||
vec!["Materials and Methods".into(), "2".into()],
|
||||
vec!["Results and Discussion".into(), "3".into()],
|
||||
vec!["Summary of Findings".into(), "4".into()],
|
||||
vec!["References and Notes".into(), "5".into()],
|
||||
];
|
||||
assert!(is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_dense_ordinal_column() {
|
||||
// Headerless title | rank table: values are a consecutive 1..n
|
||||
// sequence (monotonic, no header, text first column) but their range
|
||||
// ~= the row count, so it is data, not a table of contents.
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Alice".into(), "1".into()],
|
||||
vec!["Bob".into(), "2".into()],
|
||||
vec!["Carol".into(), "3".into()],
|
||||
vec!["Dave".into(), "4".into()],
|
||||
vec!["Erin".into(), "5".into()],
|
||||
vec!["Frank".into(), "6".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_blank_header_cell() {
|
||||
// First row is a header whose last cell is blank ("Category | ");
|
||||
// must not be flattened even though later rows look TOC-like.
|
||||
let cells = vec![
|
||||
vec!["Category".into(), "".into()],
|
||||
vec!["Alpha".into(), "3".into()],
|
||||
vec!["Beta".into(), "9".into()],
|
||||
vec!["Gamma".into(), "14".into()],
|
||||
vec!["Delta".into(), "20".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_matches_title_based_contents() {
|
||||
// Title-left, page-number-right, no dot leaders, no section numbers.
|
||||
let cells = vec![
|
||||
vec!["About the Publisher".into(), "vii".into()],
|
||||
vec!["About This Project".into(), "ix".into()],
|
||||
vec!["Acknowledgments".into(), "xi".into()],
|
||||
vec!["Experiment #1: Hydrostatic Pressure".into(), "3".into()],
|
||||
vec!["Experiment #2: Bernoulli's Theorem".into(), "13".into()],
|
||||
vec![
|
||||
"Experiment #3: Energy Loss in Pipe Fittings".into(),
|
||||
"24".into(),
|
||||
],
|
||||
];
|
||||
assert!(is_page_number_toc(&cells));
|
||||
assert!(is_table_of_contents(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_numeric_data_table() {
|
||||
// Real 2-col data table: numeric first column, non-monotonic values.
|
||||
let cells = vec![
|
||||
vec!["101".into(), "45".into()],
|
||||
vec!["102".into(), "12".into()],
|
||||
vec!["103".into(), "88".into()],
|
||||
vec!["104".into(), "7".into()],
|
||||
vec!["105".into(), "63".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_non_monotonic_pages() {
|
||||
// Text labels but the "page" column jumps around — a small data table,
|
||||
// not a contents listing. 5 rows so the row-count guard passes and the
|
||||
// monotonicity check is what does the rejecting.
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Apples".into(), "42".into()],
|
||||
vec!["Oranges".into(), "7".into()],
|
||||
vec!["Pears".into(), "91".into()],
|
||||
vec!["Plums".into(), "3".into()],
|
||||
vec!["Grapes".into(), "60".into()],
|
||||
];
|
||||
// Sanity: this input clears the row-count and header guards, so a
|
||||
// failure here is genuinely the monotonicity check.
|
||||
assert!(cells.len() >= 5 && page_number_value(cells[0][1].trim()).is_some());
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_header_row_data_table() {
|
||||
// Real 2-col data table with a header row ("Mineral | CEC") and
|
||||
// ascending values that mimic page numbers — the header tells us it
|
||||
// is data, not contents.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Mineral or colloid type".into(),
|
||||
"CEC of pure colloid".into(),
|
||||
],
|
||||
vec!["kaolinite".into(), "10".into()],
|
||||
vec!["illite".into(), "30".into()],
|
||||
vec!["montmorillonite".into(), "100".into()],
|
||||
vec!["vermiculite".into(), "150".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_wide_data_grid() {
|
||||
// A 4-column regional data table must not be read as a TOC even with a
|
||||
// text first column and integer last column.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"REGIONS".into(),
|
||||
"2007".into(),
|
||||
"2010".into(),
|
||||
"2016".into(),
|
||||
],
|
||||
vec![
|
||||
"National Capital Region".into(),
|
||||
"9".into(),
|
||||
"8".into(),
|
||||
"5".into(),
|
||||
],
|
||||
vec!["Cordillera".into(), "1".into(), "2".into(), "1".into()],
|
||||
vec!["Ilocos Region".into(), "1".into(), "5".into(), "4".into()],
|
||||
vec!["Cagayan Valley".into(), "1".into(), "3".into(), "5".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_needs_page_number_last_column() {
|
||||
// Last column is prose, not page numbers.
|
||||
let cells = vec![
|
||||
vec!["Section A".into(), "see appendix".into()],
|
||||
vec!["Section B".into(), "see notes".into()],
|
||||
vec!["Section C".into(), "later".into()],
|
||||
vec!["Section D".into(), "TBD".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
}
|
||||
|
||||
+2111
-21
File diff suppressed because it is too large
Load Diff
+785
-9
@@ -8,6 +8,9 @@ use crate::types::{PdfRect, TextItem};
|
||||
|
||||
use super::Table;
|
||||
|
||||
const DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS: usize = 8;
|
||||
const COMPETING_TABLE_MIN_ROWS: usize = 8;
|
||||
|
||||
/// Disjoint-set (union-find) with component sizes for clustering indices.
|
||||
struct UnionFind {
|
||||
parent: Vec<usize>,
|
||||
@@ -226,6 +229,77 @@ pub struct RectHintRegion {
|
||||
/// Also returns hint regions: bounding boxes of cell-sized rects from clusters
|
||||
/// that failed full grid validation. These can be used to scope heuristic
|
||||
/// detection and prevent unrelated items from being merged into tables.
|
||||
/// Bounding boxes of chart-bar clusters on the page. Text inside these
|
||||
/// regions (axis labels, data values, legends) belongs to a figure and must
|
||||
/// not be gridded into a table by any detection strategy.
|
||||
pub fn detect_chart_regions(
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
page: u32,
|
||||
) -> Vec<(f32, f32, f32, f32)> {
|
||||
// Match detect_tables_from_rects: image placeholders are not text and
|
||||
// would defeat the bar-content check.
|
||||
let items_owned: Vec<TextItem> = items
|
||||
.iter()
|
||||
.filter(|i| crate::extractor::is_text_layout_item(i))
|
||||
.cloned()
|
||||
.collect();
|
||||
let items = items_owned.as_slice();
|
||||
let page_rects: Vec<(f32, f32, f32, f32)> = rects
|
||||
.iter()
|
||||
.filter(|r| r.page == page)
|
||||
.map(|r| {
|
||||
let (x, w) = if r.width < 0.0 {
|
||||
(r.x + r.width, -r.width)
|
||||
} else {
|
||||
(r.x, r.width)
|
||||
};
|
||||
let (y, h) = if r.height < 0.0 {
|
||||
(r.y + r.height, -r.height)
|
||||
} else {
|
||||
(r.y, r.height)
|
||||
};
|
||||
(x, y, w, h)
|
||||
})
|
||||
// Origin-anchored page backgrounds/clipping paths are never chart
|
||||
// geometry, and letting one bridge into a bar cluster would inflate
|
||||
// the region to the whole page.
|
||||
.filter(|&(x, y, w, h)| w >= 5.0 && h >= 5.0 && !(x < 5.0 && y < 5.0))
|
||||
.collect();
|
||||
if page_rects.len() < 6 {
|
||||
return Vec::new();
|
||||
}
|
||||
let mut regions = Vec::new();
|
||||
for cluster in &cluster_rects(&page_rects, 3.0, 6) {
|
||||
let group: Vec<(f32, f32, f32, f32)> = cluster.iter().map(|&i| page_rects[i]).collect();
|
||||
if is_chart_bar_cluster(items, &group, page) {
|
||||
let bbox = group.iter().fold(
|
||||
(
|
||||
f32::INFINITY,
|
||||
f32::INFINITY,
|
||||
f32::NEG_INFINITY,
|
||||
f32::NEG_INFINITY,
|
||||
),
|
||||
|(x0, y0, x1, y1), &(x, y, w, h)| {
|
||||
(x0.min(x), y0.min(y), x1.max(x + w), y1.max(y + h))
|
||||
},
|
||||
);
|
||||
regions.push(bbox);
|
||||
}
|
||||
}
|
||||
regions
|
||||
}
|
||||
|
||||
fn detect_direct_rect_table(
|
||||
items: &[TextItem],
|
||||
rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> Option<Table> {
|
||||
detect_table_from_rect_group(items, rects, page)
|
||||
.or_else(|| detect_row_stripe_table(items, rects, page))
|
||||
.or_else(|| detect_stacked_box_table(items, rects, page))
|
||||
}
|
||||
|
||||
pub fn detect_tables_from_rects(
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
@@ -379,12 +453,77 @@ pub fn detect_tables_from_rects(
|
||||
.collect();
|
||||
|
||||
debug!("page {}: {} clusters with >= 6 rects", page, clusters.len());
|
||||
for cluster_indices in &clusters {
|
||||
let mut merge_excluded_cluster_ids: Vec<usize> = Vec::new();
|
||||
for (cluster_id, cluster_indices) in clusters.iter().enumerate() {
|
||||
let group_rects: Vec<(f32, f32, f32, f32)> =
|
||||
cluster_indices.iter().map(|&i| page_rects[i]).collect();
|
||||
if let Some(table) = detect_table_from_rect_group(items, &group_rects, page) {
|
||||
tables.push(table);
|
||||
} else if let Some(table) = detect_row_stripe_table(items, &group_rects, page) {
|
||||
// Chart bars are neither table cells nor a hint region — gridding
|
||||
// a chart's axis labels scrambles the page. Skip the cluster
|
||||
// entirely so it can't reach any detector, the merged fallback,
|
||||
// or the hint fallback.
|
||||
if is_chart_bar_cluster(items, &group_rects, page) {
|
||||
// Repeated page fills can dominate the geometry and make a
|
||||
// real shaded-cell table look like a chart. Remove those fills,
|
||||
// re-cluster the remaining geometry, and evaluate valid table
|
||||
// candidates as a competing hypothesis before the chart
|
||||
// rejection wins.
|
||||
let normalized = without_dominant_page_backgrounds(&group_rects);
|
||||
let normalized_table = (normalized.len() < group_rects.len())
|
||||
.then(|| {
|
||||
cluster_rects(&normalized, 3.0, 6)
|
||||
.iter()
|
||||
.filter_map(|indices| {
|
||||
let candidate: Vec<(f32, f32, f32, f32)> =
|
||||
indices.iter().map(|&i| normalized[i]).collect();
|
||||
if is_chart_bar_cluster(items, &candidate, page) {
|
||||
None
|
||||
} else {
|
||||
detect_table_from_rect_group(items, &candidate, page)
|
||||
.or_else(|| {
|
||||
detect_row_stripe_table_from_cell_rects(
|
||||
items, &candidate, page,
|
||||
)
|
||||
})
|
||||
// Small chart panels can still form
|
||||
// plausible grids from their labels.
|
||||
// Require sustained row evidence; the
|
||||
// motivating table has 17 rows.
|
||||
.filter(|table| {
|
||||
table.rows.len() >= COMPETING_TABLE_MIN_ROWS
|
||||
})
|
||||
}
|
||||
})
|
||||
.max_by_key(|table| table.rows.len() * table.columns.len())
|
||||
})
|
||||
.flatten();
|
||||
if let Some(table) = normalized_table {
|
||||
debug!(
|
||||
"page {}: chart-like cluster normalized from {} to {} rects; accepted {}x{} table hypothesis",
|
||||
page,
|
||||
group_rects.len(),
|
||||
normalized.len(),
|
||||
table.rows.len(),
|
||||
table.columns.len()
|
||||
);
|
||||
// The accepted hypothesis is based on normalized
|
||||
// geometry. Keep the original chart-like cluster out of
|
||||
// the merged fallback: reintroducing its repeated page
|
||||
// fills can manufacture a wider candidate that replaces
|
||||
// this valid narrow table below.
|
||||
merge_excluded_cluster_ids.push(cluster_id);
|
||||
tables.push(table);
|
||||
continue;
|
||||
} else {
|
||||
debug!(
|
||||
"page {}: skipping chart-bar cluster ({} rects)",
|
||||
page,
|
||||
group_rects.len()
|
||||
);
|
||||
merge_excluded_cluster_ids.push(cluster_id);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if let Some(table) = detect_direct_rect_table(items, &group_rects, page) {
|
||||
tables.push(table);
|
||||
} else if let Some((left, right)) = split_wide_cluster(&group_rects, 15.0, 6) {
|
||||
// Cluster was too wide — retry each half independently
|
||||
@@ -419,12 +558,19 @@ pub fn detect_tables_from_rects(
|
||||
// text-based column detection.
|
||||
let only_narrow = !tables.is_empty() && tables.iter().all(|t| t.columns.len() <= 3);
|
||||
if tables.is_empty() || only_narrow {
|
||||
let total_clustered: usize = clusters.iter().map(|c| c.len()).sum();
|
||||
if clusters.len() >= 3 && total_clustered >= 50 {
|
||||
// Chart clusters stay out of the merge as well.
|
||||
let table_clusters: Vec<&Vec<usize>> = clusters
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(id, _)| !merge_excluded_cluster_ids.contains(id))
|
||||
.map(|(_, c)| c)
|
||||
.collect();
|
||||
let total_clustered: usize = table_clusters.iter().map(|c| c.len()).sum();
|
||||
if table_clusters.len() >= 3 && total_clustered >= 50 {
|
||||
debug!(
|
||||
"page {}: trying merged-cluster fallback ({} clusters, {} rects{})",
|
||||
page,
|
||||
clusters.len(),
|
||||
table_clusters.len(),
|
||||
total_clustered,
|
||||
if only_narrow {
|
||||
", replacing narrow tables"
|
||||
@@ -432,7 +578,7 @@ pub fn detect_tables_from_rects(
|
||||
""
|
||||
}
|
||||
);
|
||||
let all_cluster_rects: Vec<(f32, f32, f32, f32)> = clusters
|
||||
let all_cluster_rects: Vec<(f32, f32, f32, f32)> = table_clusters
|
||||
.iter()
|
||||
.flat_map(|idxs| idxs.iter().map(|&i| page_rects[i]))
|
||||
.collect();
|
||||
@@ -491,6 +637,14 @@ pub fn detect_tables_from_rects(
|
||||
}
|
||||
}
|
||||
|
||||
// NOTE: 3-5 box stacks never reach detect_stacked_box_table — the main
|
||||
// loop requires >=6-rect clusters (and a >=6-rect page). This is a
|
||||
// deliberate precision gate: routing smaller clusters through the
|
||||
// detector was tried and regressed four pdf-evals documents (striped
|
||||
// bullet lists, wrapped regulation text, stats-table columns) while
|
||||
// improving nothing — with so few boxes the anti-prose guards have too
|
||||
// little signal to discriminate. See stacked_box_three_rows_below_
|
||||
// cluster_minimum for the pinned behavior.
|
||||
if tables.is_empty() {
|
||||
// When no tables detected but clusters exist, generate XY hint regions
|
||||
// from cluster bounding boxes to scope heuristic table detection.
|
||||
@@ -631,6 +785,240 @@ pub fn detect_tables_from_rects(
|
||||
/// overlap or are close (gap < 50pt). This handles calendar-style layouts where a
|
||||
/// month zone's decorative rects split into 2-3 adjacent clusters with small X gaps.
|
||||
/// Runs iteratively until no more merges occur.
|
||||
/// Detect a single-column table drawn as a vertical stack of boxes, each
|
||||
/// holding one short line of text (framework/step lists on slide-style
|
||||
/// pages). The normal grid path rejects these — one column means only two
|
||||
/// x-edges — so the rows would otherwise flow into surrounding prose as a
|
||||
/// run-on paragraph.
|
||||
fn detect_stacked_box_table(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> Option<Table> {
|
||||
// Candidate row boxes: single-text-line height, substantial width.
|
||||
let cands: Vec<(f32, f32, f32, f32)> = group_rects
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(_, _, w, h)| w >= 100.0 && (8.0..=80.0).contains(&h))
|
||||
.collect();
|
||||
// The row boxes form the largest family of same-width, x-aligned rects
|
||||
// (backgrounds and decor have their own geometry and stay out).
|
||||
let mut boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
|
||||
for &anchor in &cands {
|
||||
let family: Vec<(f32, f32, f32, f32)> = cands
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x, _, w, h)| {
|
||||
(x - anchor.0).abs() <= 12.0
|
||||
&& (w - anchor.2).abs() <= anchor.2 * 0.15
|
||||
&& (h - anchor.3).abs() <= anchor.3 * 0.3
|
||||
})
|
||||
.collect();
|
||||
if family.len() > boxes.len() {
|
||||
boxes = family;
|
||||
}
|
||||
}
|
||||
if boxes.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
// Boxes flanked at the same y-level — by other rects or by text outside
|
||||
// the family's x-range — are one column of a wider structure. Leave
|
||||
// those to the grid/cell-rect paths instead of collapsing to one column.
|
||||
let flanked = boxes
|
||||
.iter()
|
||||
.filter(|&&(bx, by, bw, bh)| {
|
||||
let rect_sibling = group_rects.iter().any(|&(ox, oy, ow, oh)| {
|
||||
let y_overlap = (by + bh).min(oy + oh) - by.max(oy);
|
||||
oh >= 8.0
|
||||
&& y_overlap > bh * 0.5
|
||||
&& (ox + ow <= bx + 2.0 || ox >= bx + bw - 2.0)
|
||||
&& ow >= 30.0
|
||||
});
|
||||
let text_sibling = items.iter().any(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
it.page == page
|
||||
&& it.y >= by - 2.0
|
||||
&& it.y <= by + bh + 2.0
|
||||
&& (cx < bx - 5.0 || cx > bx + bw + 5.0)
|
||||
&& it.width >= 10.0
|
||||
});
|
||||
rect_sibling || text_sibling
|
||||
})
|
||||
.count();
|
||||
if flanked * 3 >= boxes.len() {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} boxes flanked by rects or text",
|
||||
flanked,
|
||||
boxes.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
boxes.sort_by(|a, b| b.1.total_cmp(&a.1)); // top to bottom (descending y)
|
||||
|
||||
// Merge duplicates (border + fill pairs draw the same box twice), then
|
||||
// require a clean vertical stack: no overlaps beyond a small tolerance.
|
||||
boxes.dedup_by(|a, b| (a.1 - b.1).abs() <= 3.0 && (a.3 - b.3).abs() <= 6.0);
|
||||
if boxes.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
for w in boxes.windows(2) {
|
||||
let (upper, lower) = (w[0], w[1]);
|
||||
let upper_bottom = upper.1;
|
||||
let lower_top = lower.1 + lower.3;
|
||||
if lower_top > upper_bottom + 4.0 {
|
||||
return None; // vertical overlap — not a stack
|
||||
}
|
||||
if upper_bottom - lower_top > upper.3.max(lower.3) {
|
||||
return None; // gap larger than a row — unrelated boxes
|
||||
}
|
||||
}
|
||||
|
||||
// Assign items to boxes; every box needs text and cells must stay short
|
||||
// (prose paragraphs inside stacked frames are page decor, not a table).
|
||||
let mut cells: Vec<Vec<String>> = Vec::with_capacity(boxes.len());
|
||||
let mut item_indices: Vec<usize> = Vec::new();
|
||||
let mut multi_run_boxes = 0usize;
|
||||
for &(bx, by, bw, bh) in &boxes {
|
||||
let mut in_box: Vec<(usize, &TextItem)> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, it)| {
|
||||
it.page == page
|
||||
&& it.y >= by - 2.0
|
||||
&& it.y <= by + bh + 2.0
|
||||
&& it.x + it.width / 2.0 >= bx
|
||||
&& it.x + it.width / 2.0 <= bx + bw
|
||||
})
|
||||
.collect();
|
||||
if in_box.is_empty() {
|
||||
return None;
|
||||
}
|
||||
in_box.sort_by(|a, b| {
|
||||
b.1.y
|
||||
.partial_cmp(&a.1.y)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then_with(|| {
|
||||
a.1.x
|
||||
.partial_cmp(&b.1.x)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
})
|
||||
});
|
||||
// Count horizontally separated text runs inside the box. A single
|
||||
// list row flows as one run; two-plus runs across most boxes means
|
||||
// multi-column content (striped prose or a real grid) that must not
|
||||
// collapse into a one-column table. Same-baseline only: boxed
|
||||
// display/diagram rows legitimately scatter segments at mixed
|
||||
// baselines, and those must stay one row.
|
||||
let mut runs = 1usize;
|
||||
for pair in in_box.windows(2) {
|
||||
let (prev, item) = (pair[0].1, pair[1].1);
|
||||
if (prev.y - item.y).abs() <= 2.0 && item.x - (prev.x + prev.width) > 15.0 {
|
||||
runs += 1;
|
||||
}
|
||||
}
|
||||
if runs >= 2 {
|
||||
multi_run_boxes += 1;
|
||||
}
|
||||
let text = in_box
|
||||
.iter()
|
||||
.map(|(_, it)| it.text.trim())
|
||||
.filter(|t| !t.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
if text.is_empty() || text.chars().count() > 120 {
|
||||
return None;
|
||||
}
|
||||
item_indices.extend(in_box.iter().map(|(i, _)| *i));
|
||||
cells.push(vec![text]);
|
||||
}
|
||||
if multi_run_boxes * 2 >= boxes.len() {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} boxes hold multiple text runs",
|
||||
multi_run_boxes,
|
||||
boxes.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
// Reject prose behind per-line stripe rects: sentence fragments flowing
|
||||
// across rows read as long, function-word-dense cells, while genuine
|
||||
// list-table rows are short labels/titles.
|
||||
const PROSE_WORDS: &[&str] = &[
|
||||
"a", "an", "the", "of", "to", "is", "was", "are", "were", "be", "been", "in", "on", "at",
|
||||
"with", "for", "by", "as", "and", "or", "but", "this", "that", "these", "those", "from",
|
||||
"into", "has", "have", "had", "not", "it", "its", "their", "such", "shall", "which",
|
||||
];
|
||||
let total_chars: usize = cells.iter().map(|r| r[0].chars().count()).sum();
|
||||
let mean_chars = total_chars / cells.len().max(1);
|
||||
let prose_cells = cells
|
||||
.iter()
|
||||
.filter(|r| {
|
||||
r[0].to_ascii_lowercase()
|
||||
.split(|c: char| !c.is_ascii_alphabetic() && c != '\'')
|
||||
.any(|w| PROSE_WORDS.contains(&w))
|
||||
})
|
||||
.count();
|
||||
if mean_chars > 60 && prose_cells * 5 >= cells.len() * 2 {
|
||||
debug!(
|
||||
" stacked-box rejected: prose rows (mean {} chars, prose words {}/{})",
|
||||
mean_chars,
|
||||
prose_cells,
|
||||
cells.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
// Sentences wrapping across stripe rects: a row ending with a comma, or
|
||||
// a row without terminal punctuation followed by a row starting
|
||||
// lowercase, is mid-sentence flow — not list rows. Genuine label/title
|
||||
// rows produce none of these, so even a small share is disqualifying.
|
||||
let continuations = cells
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let prev = pair[0][0].trim_end();
|
||||
let next = pair[1][0].trim_start();
|
||||
let prev_open = !prev.ends_with(['.', ':', ';', '!', '?', ')', '"', '%']);
|
||||
let next_lower = next.chars().next().is_some_and(|c| c.is_lowercase());
|
||||
prev.ends_with(',') || (prev_open && next_lower)
|
||||
})
|
||||
.count();
|
||||
if cells.len() >= 2 && (continuations >= 2 || continuations * 4 >= cells.len() - 1) {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} row pairs continue a sentence",
|
||||
continuations,
|
||||
cells.len() - 1
|
||||
);
|
||||
return None;
|
||||
}
|
||||
// Numbered/lettered list items behind decorative stripes stay lists:
|
||||
// "1) content..." / "(ii) content..." / "a. content...".
|
||||
let list_marker = |t: &str| {
|
||||
let t = t.trim_start().strip_prefix('(').unwrap_or(t.trim_start());
|
||||
let marker_len = t.chars().take_while(|c| c.is_ascii_alphanumeric()).count();
|
||||
(1..=3).contains(&marker_len)
|
||||
&& t.chars()
|
||||
.nth(marker_len)
|
||||
.is_some_and(|c| c == ')' || c == '.')
|
||||
};
|
||||
let list_rows = cells.iter().filter(|r| list_marker(&r[0])).count();
|
||||
if list_rows * 2 >= cells.len() {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} rows are numbered list items",
|
||||
list_rows,
|
||||
cells.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
debug!(
|
||||
"page {}: stacked-box table: {} single-column rows",
|
||||
page,
|
||||
cells.len()
|
||||
);
|
||||
let columns = vec![boxes[0].0 + boxes[0].2 / 2.0];
|
||||
let rows: Vec<f32> = boxes.iter().map(|b| b.1 + b.3 / 2.0).collect();
|
||||
Some(Table::new(columns, rows, cells, item_indices))
|
||||
}
|
||||
|
||||
fn merge_overlapping_hints(mut hints: Vec<RectHintRegion>) -> Vec<RectHintRegion> {
|
||||
if hints.len() <= 1 {
|
||||
return hints;
|
||||
@@ -1578,11 +1966,153 @@ fn row_stripe_is_sparse_prose_outline(cells: &[Vec<String>]) -> bool {
|
||||
long_dense_cells * 2 >= dense_count
|
||||
}
|
||||
|
||||
/// Remove repeated page-scale fills from a chart-like cluster so the actual
|
||||
/// cell/bar geometry can be evaluated independently. A small number of
|
||||
/// coincident origin frames may be meaningful table structure, so repetition
|
||||
/// only becomes normalization evidence when it dominates the cluster.
|
||||
fn without_dominant_page_backgrounds(rects: &[(f32, f32, f32, f32)]) -> Vec<(f32, f32, f32, f32)> {
|
||||
let x_max = rects
|
||||
.iter()
|
||||
.map(|&(x, _, width, _)| x + width)
|
||||
.fold(0.0_f32, f32::max);
|
||||
let y_max = rects
|
||||
.iter()
|
||||
.map(|&(_, y, _, height)| y + height)
|
||||
.fold(0.0_f32, f32::max);
|
||||
let is_page_scale = |&(x, y, width, height): &(f32, f32, f32, f32)| {
|
||||
x < 5.0 && y < 5.0 && width >= x_max * 0.9 && height >= y_max * 0.9
|
||||
};
|
||||
|
||||
if rects.iter().filter(|rect| is_page_scale(rect)).count()
|
||||
< DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS
|
||||
{
|
||||
return rects.to_vec();
|
||||
}
|
||||
|
||||
rects
|
||||
.iter()
|
||||
.filter(|rect| !is_page_scale(rect))
|
||||
.copied()
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Detect a table from cell-background rects that failed grid detection.
|
||||
///
|
||||
/// Uses rect Y-edges for row boundaries and text X-position clustering for
|
||||
/// columns. Handles tables with cell backgrounds that don't form a clean
|
||||
/// X-edge grid (variable column widths, decorative fills).
|
||||
/// Chart-bar signature: ≥3 rects sharing an aligned bottom edge (the axis),
|
||||
/// with similar widths (bars) but strongly varying heights (data-driven),
|
||||
/// holding at most a single numeric data label each. Bar charts drawn as
|
||||
/// filled rects otherwise read as cell rects and grid their axis labels
|
||||
/// into a phantom table. The mirrored check catches horizontal bar charts.
|
||||
fn is_chart_bar_cluster(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> bool {
|
||||
let numeric_or_empty = |(rx, ry, rw, rh): (f32, f32, f32, f32)| {
|
||||
let inside: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
it.page == page && cx >= rx && cx <= rx + rw && it.y >= ry && it.y <= ry + rh
|
||||
})
|
||||
.collect();
|
||||
// Any number of numeric data labels is chart-like; a single run of
|
||||
// word text inside means a table cell.
|
||||
inside.iter().all(|it| {
|
||||
let t = it.text.trim();
|
||||
let data = t
|
||||
.chars()
|
||||
.filter(|c| c.is_ascii_digit() || ",.%-".contains(*c))
|
||||
.count();
|
||||
t.is_empty() || data * 2 >= t.chars().count()
|
||||
})
|
||||
};
|
||||
|
||||
// Bars: the dominant equal-width family, arranged in >=2 spaced columns
|
||||
// (inter-column gap >= half a bar width — table cell rects touch), with
|
||||
// data-driven height variation (checkbox/cell grids are uniform).
|
||||
// Mirrored predicate catches horizontal bar charts.
|
||||
let bar_family = |pos: fn(&(f32, f32, f32, f32)) -> f32,
|
||||
breadth: fn(&(f32, f32, f32, f32)) -> f32,
|
||||
length: fn(&(f32, f32, f32, f32)) -> f32,
|
||||
along: fn(&(f32, f32, f32, f32)) -> f32| {
|
||||
group_rects.iter().any(|anchor| {
|
||||
let bw = breadth(anchor);
|
||||
if bw <= 0.0 {
|
||||
return false;
|
||||
}
|
||||
let family: Vec<&(f32, f32, f32, f32)> = group_rects
|
||||
.iter()
|
||||
.filter(|r| {
|
||||
(breadth(r) - bw).abs() <= (bw * 0.1).max(2.0)
|
||||
&& length(r) > 0.0
|
||||
&& length(r) < bw * 20.0
|
||||
})
|
||||
.collect();
|
||||
if family.len() < 4 {
|
||||
return false;
|
||||
}
|
||||
// Distinct positions along the axis (bar columns).
|
||||
let mut positions: Vec<f32> = Vec::new();
|
||||
for r in &family {
|
||||
let p = pos(r);
|
||||
if !positions.iter().any(|&q| (q - p).abs() <= 2.0) {
|
||||
positions.push(p);
|
||||
}
|
||||
}
|
||||
if positions.len() < 2 {
|
||||
return false;
|
||||
}
|
||||
positions.sort_by(|a, b| a.total_cmp(b));
|
||||
let min_gap = positions
|
||||
.windows(2)
|
||||
.map(|w| w[1] - w[0] - bw)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
if min_gap < bw * 0.5 {
|
||||
return false;
|
||||
}
|
||||
// Data-driven variation along the bar direction.
|
||||
let len_min = family
|
||||
.iter()
|
||||
.map(|r| length(r))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let len_max = family
|
||||
.iter()
|
||||
.map(|r| length(r))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if len_max < len_min * 1.3 {
|
||||
return false;
|
||||
}
|
||||
// Grid rows disguise as bars: a table's cell rects have same-y,
|
||||
// same-height partners in other columns (uniform row heights).
|
||||
// Chart segments start where the previous datum ended, so their
|
||||
// extents rarely pair up across positions.
|
||||
let matched = family
|
||||
.iter()
|
||||
.filter(|r| {
|
||||
family.iter().any(|s| {
|
||||
(pos(s) - pos(r)).abs() > 2.0
|
||||
&& (along(s) - along(r)).abs() <= 3.0
|
||||
&& (length(s) - length(r)).abs() <= 3.0
|
||||
})
|
||||
})
|
||||
.count();
|
||||
if matched * 5 >= family.len() * 3 {
|
||||
return false;
|
||||
}
|
||||
family.iter().filter(|r| numeric_or_empty(***r)).count() * 3 >= family.len() * 2
|
||||
})
|
||||
};
|
||||
|
||||
// vertical bars: position/breadth = x/width, length = height, along = y
|
||||
bar_family(|r| r.0, |r| r.2, |r| r.3, |r| r.1)
|
||||
// horizontal bars: position/breadth = y/height, length = width, along = x
|
||||
|| bar_family(|r| r.1, |r| r.3, |r| r.2, |r| r.0)
|
||||
}
|
||||
|
||||
fn detect_row_stripe_table_from_cell_rects(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
@@ -2367,7 +2897,29 @@ fn detect_merged_cluster_table(
|
||||
/// suitable for rect-backed tables where we already know tabular structure exists
|
||||
/// (no need for anti-paragraph safeguards).
|
||||
fn cluster_x_positions(items: &[(usize, &TextItem)], min_threshold: f32) -> Vec<f32> {
|
||||
let mut x_positions: Vec<f32> = items.iter().map(|(_, i)| i.x).collect();
|
||||
// Column edges come from where text STARTS. An item whose left edge hugs
|
||||
// the previous item's right edge on the same line is a continuation run
|
||||
// (style boundary, script change, underline split) — feeding its x-start
|
||||
// in here fabricates a phantom column mid-cell.
|
||||
let mut sorted: Vec<&TextItem> = items.iter().map(|&(_, i)| i).collect();
|
||||
sorted.sort_by(|a, b| a.y.total_cmp(&b.y).then(a.x.total_cmp(&b.x)));
|
||||
let mut x_positions: Vec<f32> = Vec::with_capacity(sorted.len());
|
||||
for (idx, item) in sorted.iter().enumerate() {
|
||||
let is_continuation = idx > 0 && {
|
||||
let prev = sorted[idx - 1];
|
||||
// Style/underline splits leave runs that TOUCH (gap ~0); real
|
||||
// cell boundaries in even the tightest tables keep a visible
|
||||
// gap. 2pt separates the two without eating dense-table columns.
|
||||
// The negative side is bounded too: text overhanging from an
|
||||
// adjacent cell overlaps by far more than italic kerning ever
|
||||
// does, and must still start its own column.
|
||||
let gap = item.x - (prev.x + prev.width);
|
||||
(prev.y - item.y).abs() <= 2.0 && gap < 2.0 && gap > -4.0 && item.x >= prev.x
|
||||
};
|
||||
if !is_continuation {
|
||||
x_positions.push(item.x);
|
||||
}
|
||||
}
|
||||
x_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if x_positions.is_empty() {
|
||||
@@ -2436,6 +2988,187 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
// --- is_chart_bar_cluster / detect_chart_regions ---
|
||||
|
||||
/// Stacked bar chart: frame + 3 columns of equal-width segments with
|
||||
/// data-driven heights, holding numeric labels.
|
||||
fn chart_rects() -> Vec<PdfRect> {
|
||||
let mut rects = vec![PdfRect {
|
||||
x: 126.0,
|
||||
y: 548.0,
|
||||
width: 396.0,
|
||||
height: 216.0,
|
||||
page: 1,
|
||||
}];
|
||||
let bars = [
|
||||
(208.0, 618.0, 59.0),
|
||||
(208.0, 661.0, 39.0),
|
||||
(208.0, 696.0, 37.0),
|
||||
(313.0, 618.0, 67.0),
|
||||
(313.0, 670.0, 49.0),
|
||||
(313.0, 691.0, 42.0),
|
||||
(419.0, 618.0, 73.0),
|
||||
(419.0, 684.0, 37.0),
|
||||
(419.0, 708.0, 25.0),
|
||||
];
|
||||
for (x, y, h) in bars {
|
||||
rects.push(PdfRect {
|
||||
x,
|
||||
y,
|
||||
width: 46.0,
|
||||
height: h,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
rects
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_bars_produce_region_not_table() {
|
||||
let items: Vec<TextItem> = [
|
||||
("38", 228.0, 638.0),
|
||||
("30", 228.0, 676.0),
|
||||
("46", 333.0, 643.0),
|
||||
("17", 333.0, 679.0),
|
||||
("57", 438.0, 650.0),
|
||||
("20", 438.0, 694.0),
|
||||
]
|
||||
.iter()
|
||||
.map(|&(t, x, y)| make_item(t, x, y, 9.0))
|
||||
.collect();
|
||||
let rects = chart_rects();
|
||||
let regions = detect_chart_regions(&items, &rects, 1);
|
||||
assert_eq!(regions.len(), 1, "expected one chart region");
|
||||
let (tables, hints) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert!(tables.is_empty(), "chart bars must not become a table");
|
||||
assert!(hints.is_empty(), "chart bars must not become a hint region");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_page_backgrounds_are_normalized_only_after_repetition() {
|
||||
let page_fill = (0.0, 0.0, 600.0, 800.0);
|
||||
let cell = (100.0, 500.0, 120.0, 20.0);
|
||||
|
||||
let mut dominant = vec![page_fill; DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS];
|
||||
dominant.push(cell);
|
||||
assert_eq!(without_dominant_page_backgrounds(&dominant), vec![cell]);
|
||||
|
||||
let mut incidental = vec![page_fill; DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS - 1];
|
||||
incidental.push(cell);
|
||||
assert_eq!(without_dominant_page_backgrounds(&incidental), incidental);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn uniform_cell_grid_is_not_a_chart() {
|
||||
// Touching, uniform-height cell rects (a real table) must not match:
|
||||
// no inter-column gap and no bar-length variation.
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for col in 0..3 {
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + col as f32 * 80.0,
|
||||
y: 600.0 - row as f32 * 20.0,
|
||||
width: 80.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
let items: Vec<TextItem> = (0..4)
|
||||
.flat_map(|r| {
|
||||
(0..3).map(move |c| (100.0 + c as f32 * 80.0 + 10.0, 605.0 - r as f32 * 20.0))
|
||||
})
|
||||
.map(|(x, y)| make_item("42", x, y, 9.0))
|
||||
.collect();
|
||||
assert!(detect_chart_regions(&items, &rects, 1).is_empty());
|
||||
}
|
||||
|
||||
// --- detect_stacked_box_table ---
|
||||
|
||||
/// N stacked boxes at x=100, w=300, h=22, top-to-bottom from y=600.
|
||||
fn stacked_boxes(n: usize) -> Vec<(f32, f32, f32, f32)> {
|
||||
(0..n)
|
||||
.map(|i| (100.0, 600.0 - i as f32 * 22.0, 300.0, 22.0))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_list_becomes_single_column_table() {
|
||||
let rects = stacked_boxes(5);
|
||||
let items: Vec<TextItem> = (0..5)
|
||||
.map(|i| make_item("#1: Recycling Basics", 120.0, 605.0 - i as f32 * 22.0, 10.0))
|
||||
.collect();
|
||||
let table = detect_stacked_box_table(&items, &rects, 1).expect("stacked-box table");
|
||||
assert_eq!(table.cells.len(), 5);
|
||||
assert_eq!(table.cells[0].len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_wrapped_sentences() {
|
||||
// Line stripes behind flowing prose: rows continue mid-sentence.
|
||||
let rects = stacked_boxes(4);
|
||||
let texts = [
|
||||
"the provisions of this section apply to",
|
||||
"companies subject to tax under those",
|
||||
"sections, except that the copy of the",
|
||||
"annual statement must be retained.",
|
||||
];
|
||||
let items: Vec<TextItem> = texts
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, t)| make_item(t, 120.0, 605.0 - i as f32 * 22.0, 10.0))
|
||||
.collect();
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_flanking_text() {
|
||||
// A ruled label column with plain-text data columns beside it is one
|
||||
// column of a wider table, not a single-column list.
|
||||
let rects = stacked_boxes(4);
|
||||
let mut items = Vec::new();
|
||||
for i in 0..4 {
|
||||
let y = 605.0 - i as f32 * 22.0;
|
||||
items.push(make_item("Section 1.382", 120.0, y, 10.0));
|
||||
items.push(make_item("removed text", 450.0, y, 10.0)); // beside the box
|
||||
}
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_two_column_content() {
|
||||
// Boxes holding two separated runs are striped multi-column content.
|
||||
let rects = stacked_boxes(4);
|
||||
let mut items = Vec::new();
|
||||
for i in 0..4 {
|
||||
let y = 605.0 - i as f32 * 22.0;
|
||||
let mut left = make_item("left words", 110.0, y, 10.0);
|
||||
left.width = 60.0;
|
||||
let mut right = make_item("right words", 250.0, y, 10.0);
|
||||
right.width = 60.0;
|
||||
items.push(left);
|
||||
items.push(right);
|
||||
}
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_mixed_height_stripes() {
|
||||
// Mixed 13/27pt stripes (redline markup) — height uniformity splits
|
||||
// the family and the gap check rejects the remainder.
|
||||
let mut rects = Vec::new();
|
||||
let mut y = 600.0;
|
||||
for i in 0..8 {
|
||||
let h = if i % 3 == 0 { 27.0 } else { 13.5 };
|
||||
y -= h;
|
||||
rects.push((100.0, y, 300.0, h));
|
||||
}
|
||||
let items: Vec<TextItem> = (0..8)
|
||||
.map(|i| make_item("PART 602 OMB CONTROL", 120.0, 590.0 - i as f32 * 18.0, 10.0))
|
||||
.collect();
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
// --- has_dominant_prose_cell ---
|
||||
|
||||
fn cells_of(rows: &[&[&str]]) -> Vec<Vec<String>> {
|
||||
@@ -3458,6 +4191,49 @@ mod tests {
|
||||
assert!((merged[0].x_right - 340.0).abs() < 0.01);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_three_rows_below_cluster_minimum() {
|
||||
// Pins a deliberate precision gate: a 3-box stack stays below the
|
||||
// main loop's 6-rect cluster minimum and is NOT detected end-to-end.
|
||||
// Routing smaller clusters through detect_stacked_box_table was
|
||||
// tried and regressed four pdf-evals documents (striped bullet
|
||||
// lists, wrapped regulation text, stats-table columns) with no
|
||||
// corpus gains — too few boxes for the anti-prose guards to work.
|
||||
// If this ever becomes worth revisiting, the guards need stronger
|
||||
// signals first; flipping this assertion is the entry point.
|
||||
let mut rects: Vec<PdfRect> = (0..3)
|
||||
.map(|i| PdfRect {
|
||||
x: 100.0,
|
||||
y: 600.0 - i as f32 * 22.0,
|
||||
width: 300.0,
|
||||
height: 22.0,
|
||||
page: 1,
|
||||
})
|
||||
.collect();
|
||||
// Unrelated scattered rects push the page past the 6-rect page gate
|
||||
// so the run reaches clustering, while the 3-box stack itself stays
|
||||
// below the 6-rect cluster minimum.
|
||||
for i in 0..4 {
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + i as f32 * 120.0,
|
||||
y: 100.0,
|
||||
width: 40.0,
|
||||
height: 15.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
let items: Vec<TextItem> = ["Step One: Plan", "Step Two: Build", "Step Three: Ship"]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, t)| make_item(t, 120.0, 605.0 - i as f32 * 22.0, 10.0))
|
||||
.collect();
|
||||
let (tables, _) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"3-box stacks are intentionally below the detection floor"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failed_cluster_generates_hint_with_items() {
|
||||
// A cluster of rects forming an outer border (2 x-edges after snapping)
|
||||
|
||||
+62
-2
@@ -123,6 +123,7 @@ fn format_toc_as_list(cells: &[Vec<String>], footnotes: &[String]) -> String {
|
||||
|
||||
/// True when the cell looks like a page number. Accepts:
|
||||
/// - plain digit tokens: "42", "86 86"
|
||||
/// - canonical roman numerals (front-matter pages): "vii", "ix", "xii"
|
||||
/// - dashed section-page IDs: "5-21", "A-1", "B--3", "TC-2" (common in
|
||||
/// technical manuals)
|
||||
fn is_page_number_cell(cell: &str) -> bool {
|
||||
@@ -138,6 +139,9 @@ fn is_page_number_cell(cell: &str) -> bool {
|
||||
if all_digits {
|
||||
return t.len() <= 4;
|
||||
}
|
||||
if super::canonical_roman_value(t).is_some() {
|
||||
return true;
|
||||
}
|
||||
// Section-page form: uppercase letters, digits, dashes; at least
|
||||
// one digit present.
|
||||
t.chars()
|
||||
@@ -184,6 +188,21 @@ fn starts_with_numbered_label(cell: &str) -> bool {
|
||||
.is_some_and(|c| matches!(c, '.' | ')' | '-' | ':'))
|
||||
}
|
||||
|
||||
fn starts_with_hierarchical_numbered_label(cell: &str) -> bool {
|
||||
let token = cell
|
||||
.split_whitespace()
|
||||
.next()
|
||||
.unwrap_or("")
|
||||
.trim_end_matches(['.', ')', ':', '-']);
|
||||
let levels: Vec<&str> = token.split('.').collect();
|
||||
(2..=4).contains(&levels.len())
|
||||
&& levels.iter().all(|level| {
|
||||
!level.is_empty()
|
||||
&& level.len() <= 3
|
||||
&& level.chars().all(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
fn alpha_word_count(cell: &str) -> usize {
|
||||
cell.split_whitespace()
|
||||
.filter(|word| word.chars().any(|c| c.is_alphabetic()))
|
||||
@@ -337,11 +356,12 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
// mid-sentence/lowercase ("continued text here", "with 3.5%...") or
|
||||
// carry lowercase fragments in the later cells, so keep those mergeable.
|
||||
let looks_like_hierarchical_subrow = first_cell.is_empty()
|
||||
&& row.len() >= 3
|
||||
&& first_non_empty_col == Some(1)
|
||||
&& looks_like_compact_entry_label(first_non_empty_cell)
|
||||
&& ((non_first_cells.len() >= 2 && title_like_later_cells > 0)
|
||||
&& ((row.len() == 2 && starts_with_hierarchical_numbered_label(first_non_empty_cell))
|
||||
|| (row.len() >= 3 && non_first_cells.len() >= 2 && title_like_later_cells > 0)
|
||||
|| (non_first_cells.len() == 1
|
||||
&& row.len() >= 3
|
||||
&& prev_first_cell_empty
|
||||
&& alpha_word_count(first_non_empty_cell) >= 2));
|
||||
let looks_like_new_first_column_entry = !first_cell.is_empty()
|
||||
@@ -684,6 +704,46 @@ mod tests {
|
||||
assert_eq!(cleaned[4][1], "Model training");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_two_column_numbered_subrows_not_merged() {
|
||||
let cells = vec![
|
||||
vec!["Area".into(), "Competence".into()],
|
||||
vec![
|
||||
"1. Embodying sustainability values".into(),
|
||||
"1.1 Valuing sustainability".into(),
|
||||
],
|
||||
vec!["".into(), "1.2 Supporting fairness".into()],
|
||||
vec!["".into(), "1.3 Promoting nature".into()],
|
||||
vec![
|
||||
"2. Embracing complexity".into(),
|
||||
"2.1 Systems thinking".into(),
|
||||
],
|
||||
vec!["".into(), "2.2 Critical thinking".into()],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 6);
|
||||
assert_eq!(cleaned[2], vec!["", "1.2 Supporting fairness"]);
|
||||
assert_eq!(cleaned[3], vec!["", "1.3 Promoting nature"]);
|
||||
assert_eq!(cleaned[5], vec!["", "2.2 Critical thinking"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_two_column_numbered_continuation_merges() {
|
||||
let cells = vec![
|
||||
vec!["Area".into(), "Requirement".into()],
|
||||
vec!["Safety".into(), "The program includes".into()],
|
||||
vec!["".into(), "1. First requirement for every operator".into()],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 2);
|
||||
assert_eq!(
|
||||
cleaned[1][1],
|
||||
"The program includes 1. First requirement for every operator"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_partial_hierarchical_subrow_not_merged() {
|
||||
let cells = vec![
|
||||
|
||||
+57
-2
@@ -4,7 +4,7 @@
|
||||
|
||||
mod detect_heuristic;
|
||||
mod detect_lines;
|
||||
mod detect_rects;
|
||||
pub(crate) mod detect_rects;
|
||||
mod detect_struct;
|
||||
mod financial;
|
||||
mod format;
|
||||
@@ -14,8 +14,9 @@ pub mod structured;
|
||||
pub use detect_heuristic::detect_tables;
|
||||
pub(crate) use detect_heuristic::is_table_of_contents;
|
||||
pub use detect_lines::detect_tables_from_lines;
|
||||
pub(crate) use detect_lines::detect_vector_grid_tables_from_lines;
|
||||
pub(crate) use detect_rects::cluster_rects;
|
||||
pub use detect_rects::{detect_tables_from_rects, RectHintRegion};
|
||||
pub use detect_rects::{detect_chart_regions, detect_tables_from_rects, RectHintRegion};
|
||||
pub use detect_struct::detect_tables_from_struct_tree;
|
||||
pub use format::table_to_markdown;
|
||||
pub use structured::{cells_to_markdown, StructuredCell};
|
||||
@@ -177,6 +178,60 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
))
|
||||
}
|
||||
|
||||
/// Canonical lowercase roman numeral for `n` (the i/v/x/l/c range).
|
||||
pub(super) fn to_roman_lower(mut n: u32) -> String {
|
||||
const TABLE: [(u32, &str); 9] = [
|
||||
(100, "c"),
|
||||
(90, "xc"),
|
||||
(50, "l"),
|
||||
(40, "xl"),
|
||||
(10, "x"),
|
||||
(9, "ix"),
|
||||
(5, "v"),
|
||||
(4, "iv"),
|
||||
(1, "i"),
|
||||
];
|
||||
let mut out = String::new();
|
||||
for (val, sym) in TABLE {
|
||||
while n >= val {
|
||||
out.push_str(sym);
|
||||
n -= val;
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Parse a *canonical* roman numeral (i/v/x/l/c range, ≤8 chars) to its value.
|
||||
/// Returns `None` for non-canonical strings, so ordinary words made of those
|
||||
/// letters — "civil", "mix", "ill" — are not mistaken for numbers. Shared by
|
||||
/// the TOC detector and the TOC formatter so the two stay in sync.
|
||||
pub(super) fn canonical_roman_value(token: &str) -> Option<u32> {
|
||||
let lower = token.trim().to_ascii_lowercase();
|
||||
if lower.is_empty() || lower.len() > 8 || !lower.chars().all(|c| "ivxlc".contains(c)) {
|
||||
return None;
|
||||
}
|
||||
let mut total = 0i32;
|
||||
let mut prev = 0i32;
|
||||
for c in lower.chars().rev() {
|
||||
let v = match c {
|
||||
'i' => 1,
|
||||
'v' => 5,
|
||||
'x' => 10,
|
||||
'l' => 50,
|
||||
'c' => 100,
|
||||
_ => return None,
|
||||
};
|
||||
if v < prev {
|
||||
total -= v;
|
||||
} else {
|
||||
total += v;
|
||||
prev = v;
|
||||
}
|
||||
}
|
||||
let value = u32::try_from(total).ok().filter(|&n| n > 0)?;
|
||||
(to_roman_lower(value) == lower).then_some(value)
|
||||
}
|
||||
|
||||
/// Split a TextItem whose text contains multiple whitespace-separated tokens
|
||||
/// (like "10 11 12 ... 31") into individual TextItems, each assigned to the
|
||||
/// nearest column boundary.
|
||||
|
||||
@@ -7,6 +7,76 @@
|
||||
use crate::types::TextItem;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
/// Return whether text is an explicit page-number expression.
|
||||
///
|
||||
/// This strict form is suitable before layout, where removing one numeric item
|
||||
/// from substantive text such as `Page 42 explains the result` would lose data.
|
||||
pub(crate) fn is_explicit_page_number_expression(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let is_number = |value: &str| {
|
||||
!value.is_empty() && value.chars().all(|character| character.is_ascii_digit())
|
||||
};
|
||||
|
||||
if trimmed.len() <= 4 && is_number(trimmed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if is_number(inner) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
let lowercase = trimmed.to_ascii_lowercase();
|
||||
if let Some(rest) = lowercase.strip_prefix("page") {
|
||||
let words: Vec<&str> = rest.split_whitespace().collect();
|
||||
if words.len() >= 3 && is_number(words[0]) && words[1] == "of" && is_number(words[2]) {
|
||||
return true;
|
||||
}
|
||||
if words.len() >= 2 && words[0] == "of" && is_number(words[1]) {
|
||||
return true;
|
||||
}
|
||||
return match words.as_slice() {
|
||||
[] | ["of"] => true,
|
||||
[number] => is_number(number),
|
||||
["of", total] => is_number(total),
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
};
|
||||
}
|
||||
|
||||
let words: Vec<&str> = lowercase.split_whitespace().collect();
|
||||
match words.as_slice() {
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return whether a completed Markdown line looks like a page number or a
|
||||
/// labeled running header.
|
||||
///
|
||||
/// At this stage the complete line and surrounding breaks are available, so a
|
||||
/// leading `Page N` remains compatible with the existing header cleanup even
|
||||
/// when the PDF appends a chapter or document title.
|
||||
pub(crate) fn is_page_number_line(text: &str) -> bool {
|
||||
if is_explicit_page_number_expression(text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let lowercase = text.trim().to_ascii_lowercase();
|
||||
lowercase.strip_prefix("page").is_some_and(|rest| {
|
||||
rest.trim_start()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
/// Check if a character is CJK (Chinese, Japanese, Korean).
|
||||
/// CJK languages don't use spaces between words, so word-boundary
|
||||
/// heuristics should not apply when CJK characters are involved.
|
||||
|
||||
+23
-10
@@ -4,11 +4,17 @@
|
||||
|
||||
use log::{debug, warn};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::borrow::Cow;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::glyph_names::glyph_to_char;
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
static BUILTIN_CMAPS: include_dir::Dir<'_> =
|
||||
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
|
||||
|
||||
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ToUnicodeCMap {
|
||||
@@ -1149,9 +1155,7 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
|
||||
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
|
||||
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(name);
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&name)?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -1159,13 +1163,14 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
cmap.code_byte_length = 2;
|
||||
debug!(
|
||||
"Built-in CMap {}: char_map={} ranges={}",
|
||||
path.display(),
|
||||
name,
|
||||
cmap.char_map.len(),
|
||||
cmap.ranges.len()
|
||||
);
|
||||
Some(cmap)
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
|
||||
let p = PathBuf::from(dir);
|
||||
@@ -1182,6 +1187,18 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let path = find_bcmaps_dir()?.join(name);
|
||||
std::fs::read(path).ok().map(Cow::Owned)
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let file = BUILTIN_CMAPS.get_file(name)?;
|
||||
Some(Cow::Borrowed(file.contents()))
|
||||
}
|
||||
|
||||
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
|
||||
let mut stream = BinaryCMapStream::new(data);
|
||||
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
|
||||
@@ -1492,9 +1509,7 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
|
||||
}
|
||||
|
||||
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
parse_binary_cmap_encoding(&data).ok()
|
||||
}
|
||||
|
||||
@@ -1779,9 +1794,7 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
|
||||
if !name.ends_with("UCS2") {
|
||||
return None;
|
||||
}
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
|
||||
+280
-4
@@ -8,12 +8,13 @@ use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
process_pdf_mem, process_pdf_mem_with_options, process_pdf_with_options, to_markdown,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
@@ -40,10 +41,11 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [{media_box}] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>"
|
||||
),
|
||||
);
|
||||
|
||||
let content = "BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
@@ -79,6 +81,186 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_recurring_contextual_folio_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R 7 0 R 9 0 R] /Count 4 >>",
|
||||
);
|
||||
for page_index in 0..4 {
|
||||
let page_id = 3 + page_index * 2;
|
||||
let content_id = page_id + 1;
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
page_id,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 11 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
),
|
||||
);
|
||||
let page_number = page_index + 1;
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 1 0 0 1 25 30 Tm ({page_number}) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Body page {page_number}) Tj ET"
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
content_id,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
}
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
11,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_malformed_unselected_page() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R] /Count 2 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
let content = "BT /F1 12 Tf 1 0 0 1 25 30 Tm (1) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Selected page text) Tj 0 -16 Td (More selected text) Tj 0 -16 Td (Still selected text) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 6 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Length 3 >>\nstream\nBI \nendstream",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
make_text_pdf(
|
||||
"BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET",
|
||||
"0 0 612 792",
|
||||
)
|
||||
}
|
||||
|
||||
fn make_digit_run_repro_pdf() -> Vec<u8> {
|
||||
let content = r#"BT
|
||||
/F1 12 Tf
|
||||
1 0 0 1 72 780 Tm (A\)) Tj
|
||||
1 0 0 1 96 780 Tm (The) Tj
|
||||
1 0 0 1 126 780 Tm (total) Tj
|
||||
1 0 0 1 166 780 Tm (of) Tj
|
||||
1 0 0 1 186 780 Tm (730) Tj
|
||||
1 0 0 1 220 780 Tm (seats) Tj
|
||||
1 0 0 1 262 780 Tm (was) Tj
|
||||
1 0 0 1 296 780 Tm (approved.) Tj
|
||||
1 0 0 1 72 755 Tm (B\)) Tj
|
||||
1 0 0 1 96 755 Tm (let) Tj
|
||||
1 0 0 1 120 755 Tm (log) Tj
|
||||
1 0 0 1 150 755 Tm (2) Tj
|
||||
1 0 0 1 164 755 Tm (=) Tj
|
||||
1 0 0 1 180 755 Tm (a) Tj
|
||||
1 0 0 1 72 720 Tm (C\) Control: The total of 730 seats was approved. let log 2 = a) Tj
|
||||
ET"#;
|
||||
make_text_pdf(content, "0 0 595 842")
|
||||
}
|
||||
|
||||
fn truncate_eof_marker(mut pdf: Vec<u8>) -> Vec<u8> {
|
||||
assert!(pdf.ends_with(b"%%EOF"));
|
||||
pdf.pop();
|
||||
@@ -330,6 +512,21 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
assert_eq!(lines[0].text(), "First Second Third");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_digit_only_text_runs_are_preserved_in_markdown() {
|
||||
let pdf = make_digit_run_repro_pdf();
|
||||
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract positioned text");
|
||||
assert!(items.iter().any(|item| item.text == "730"));
|
||||
assert!(items.iter().any(|item| item.text == "2"));
|
||||
|
||||
let result = process_pdf_mem(&pdf).expect("convert PDF to markdown");
|
||||
assert_eq!(
|
||||
result.markdown.expect("markdown output").trim(),
|
||||
"A) The total of 730 seats was approved.\nB) let log 2 = a\nC) Control: The total of 730 seats was approved. let log 2 = a"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MarkdownOptions Tests
|
||||
// ============================================================================
|
||||
@@ -337,6 +534,7 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
#[test]
|
||||
fn test_markdown_options_default() {
|
||||
let opts = MarkdownOptions::default();
|
||||
assert_eq!(opts.profile, pdf_inspector::MarkdownProfile::Fidelity);
|
||||
assert!(opts.detect_headers);
|
||||
assert!(opts.detect_lists);
|
||||
assert!(opts.detect_code);
|
||||
@@ -346,6 +544,7 @@ fn test_markdown_options_default() {
|
||||
#[test]
|
||||
fn test_markdown_options_custom() {
|
||||
let opts = MarkdownOptions {
|
||||
profile: pdf_inspector::MarkdownProfile::Compact,
|
||||
detect_headers: false,
|
||||
detect_lists: true,
|
||||
detect_code: false,
|
||||
@@ -361,6 +560,7 @@ fn test_markdown_options_custom() {
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!opts.detect_headers);
|
||||
assert_eq!(opts.profile, pdf_inspector::MarkdownProfile::Compact);
|
||||
assert!(opts.detect_lists);
|
||||
assert!(!opts.detect_code);
|
||||
assert_eq!(opts.base_font_size, Some(14.0));
|
||||
@@ -544,6 +744,30 @@ fn test_markdown_from_items_page_breaks() {
|
||||
assert!(md.contains("Content on second page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_page_count_overload_includes_trailing_blank_pages_in_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
items.push(make_text_item(value, 25.0, 30.0, 12.0, page));
|
||||
items.push(make_text_item(
|
||||
"Company report footer",
|
||||
41.0,
|
||||
30.0,
|
||||
12.0,
|
||||
page,
|
||||
));
|
||||
}
|
||||
let options = MarkdownOptions {
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = to_markdown_from_items_with_rects_and_page_count(items, options, &[], 20);
|
||||
|
||||
assert!(md.contains("1 Company report footer"));
|
||||
assert!(md.contains("4 Company report footer"));
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Markdown From Lines Tests
|
||||
// ============================================================================
|
||||
@@ -1102,6 +1326,7 @@ fn test_pages_needing_ocr_field_accessible() {
|
||||
title: None,
|
||||
ocr_recommended: false,
|
||||
pages_needing_ocr: Vec::new(),
|
||||
ocr_reasons_by_page: std::collections::BTreeMap::new(),
|
||||
};
|
||||
assert!(detection_result.pages_needing_ocr.is_empty());
|
||||
|
||||
@@ -2911,6 +3136,57 @@ fn test_extract_pages_markdown_basic() {
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = extract_pages_markdown_mem(&pdf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 4);
|
||||
for (index, page) in result.pages.iter().enumerate() {
|
||||
assert!(page.markdown.contains("Company report footer"));
|
||||
assert!(
|
||||
!page
|
||||
.markdown
|
||||
.contains(&format!("{} Company report footer", index + 1)),
|
||||
"recurring contextual folio survived on page {}: {}",
|
||||
index + 1,
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_process_pdf_page_filter_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
|
||||
assert!(markdown.contains("Company report footer"));
|
||||
assert!(!markdown.contains("1 Company report footer"), "{markdown}");
|
||||
assert!(markdown.contains("Body page 1"));
|
||||
assert!(!markdown.contains("Body page 2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_selected_page_ignores_context_only_extraction_failure() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
let pages = extract_pages_markdown_mem(&pdf, Some(&[0])).unwrap();
|
||||
assert_eq!(pages.pages.len(), 1);
|
||||
assert!(pages.pages[0].markdown.contains("Selected page text"));
|
||||
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
assert!(markdown.contains("Selected page text"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_requested_page_extraction_failure_remains_fatal() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
assert!(extract_pages_markdown_mem(&pdf, Some(&[1])).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
@@ -22,7 +22,9 @@ Name and address of employee
|
||||
|
||||
**Publication 1244 (Rev. 7-96)** Cat. No. 44472W
|
||||
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
### Instructions
|
||||
|
||||
You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
|
||||
*(continued on inside of back cover)*
|
||||
|
||||
@@ -54,7 +56,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
Form **4070** Employee’s Report (Rev. July 1996)
|
||||
|
||||
## of Tips to EmployerOMB No. 1545-0065
|
||||
|
||||
@@ -72,7 +74,7 @@ Month or shorter period in which tips were received **4** Net tips (lines **1 +
|
||||
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—**Use this form to report tips you receive to your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See **Pub. 531**, Reporting Tip Income, for more information. You can get additional copies of **Pub. 1244**, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
|
||||
**Instructions** *(continued)*
|
||||
<u>Instructions (continued)</u>
|
||||
|
||||
**Unreported Tips.—**If you received tips of $20 or more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you **must** use Form 1040 and **Form 4137,** Social Security and Medicare Tax on Unreported Tip Income, to report them. You may **not** use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act **cannot** use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—**Get **Pub. 531,** Reporting Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—**If you do not keep a daily record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
|
||||
|
||||
@@ -1,17 +1,8 @@
|
||||
(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows:
|
||||
<u>§1.1563-3T Rules for determining stock ownership (temporary)</u>.
|
||||
|
||||
(a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)
|
||||
through (d)(2)(iii). (iv) <u>Statement</u>. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
|
||||
|
||||
(A) A description of each of the controlled groups in which the corporation
|
||||
could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and
|
||||
|
||||
(B) The following representation: [INSERT NAME AND EMPLOYER
|
||||
IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].
|
||||
|
||||
(v) <u>Election</u>-- (A) <u>Election filed</u>. An election filed under paragraph (d)(2)(iv) of
|
||||
this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in
|
||||
||||(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows: §1.1563-3T Rules for determining stock ownership (temporary). (a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)|
|
||||
|---|---|---|---|
|
||||
||through (d)(2)(iii).|||
|
||||
||(iv)|Statement|. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include-- (A) A description of each of the controlled groups in which the corporation could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and (B) The following representation: [INSERT NAME AND EMPLOYER IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].|
|
||||
||(v)|Election|-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in|
|
||||
|
||||
|termination of membership in the controlled group in which such corporation has||
|
||||
|---|---|
|
||||
@@ -29,7 +20,7 @@ this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of
|
||||
Federal income tax return (including any amended return filed on or before the due date (including extensions) of such original return) timely filed on or after May 30,
|
||||
|
||||
2006.
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
(2) <u>Expiration date</u>. The applicability of this section will expire on May 26,
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: <u>§1.6012-2 Corporations required to make returns of income</u>.
|
||||
* * * * *
|
||||
(c) [Reserved]. For further guidance, see §1.6012-2T(c).
|
||||
|
||||
@@ -6,20 +6,22 @@
|
||||
|
||||
#### Thermodynamic Properties
|
||||
|
||||
**of**
|
||||
|
||||
®
|
||||
**of** ®
|
||||
|
||||
# Freon 12
|
||||
|
||||
**(R-12)** **Technical Information** **Technical Information**
|
||||
##### (R-12)
|
||||
|
||||
##### Technical Information Technical Information
|
||||
|
||||
**®** **Thermodynamic Properties of Freon 12 Refrigerant** **(R-12)** **SI Units**
|
||||
|
||||
Tables of the thermodynamic **Units** properties of R-12 have been developed and are presented here. P = Pressure in kPa. Absolute This information is based on values calculated using the NIST REFPROP T = Temperature in Celcius Database (McLinden, M.O., Klein,
|
||||
|
||||
S.A., Lemmon, E.W., and Peskin, Vf = Fluid (liquid) specific volume
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998). kilograms per cubic meter
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998).
|
||||
kilograms per cubic meter
|
||||
|
||||
##### H = Enthalpy (kJ/kg)
|
||||
|
||||
##### S = Entropy (kJ/kg.K)
|
||||
@@ -43,9 +45,9 @@ l
|
||||
|
||||
**Freon** **®** **12 Saturation Properties-Temperature Table**
|
||||
|
||||
|Temp|Pressure||Volume|||Density||Enthalpy|||Entropy|Temp|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|°C|[kPa]|[m³ Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m³] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|Temp|Pressure||Volume||Density||Enthalpy|||Entropy|Temp|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|°C|[kPa]|[m³ Liquid v f|/kg] Vapour v g|[kg/m³ Liquid d f|] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|
||||
|-100|1.2|0.0006|10.0000|1679.0|0.100|113.3|192.8|306.1|0.6077|1.7210|-100|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
Generated
+1304
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,37 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.3"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "README.md"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[dependencies]
|
||||
console_error_panic_hook = "0.1"
|
||||
js-sys = "0.3"
|
||||
pdf-inspector = { path = ".." }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde-wasm-bindgen = "0.6"
|
||||
wasm-bindgen = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
wasm-bindgen-test = "0.3"
|
||||
|
||||
[profile.release]
|
||||
codegen-units = 1
|
||||
lto = true
|
||||
opt-level = "s"
|
||||
strip = true
|
||||
|
||||
[package.metadata.wasm-pack.profile.release]
|
||||
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
|
||||
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
|
||||
# size, and LTO optimizations above.
|
||||
wasm-opt = false
|
||||
@@ -0,0 +1,58 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
Third-party notices
|
||||
===================
|
||||
|
||||
Adobe CMaps
|
||||
-----------
|
||||
|
||||
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
|
||||
|
||||
Copyright 1990-2009 Adobe Systems Incorporated.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
Neither the name of Adobe Systems Incorporated nor the names of its
|
||||
contributors may be used to endorse or promote products derived from this
|
||||
software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -0,0 +1,60 @@
|
||||
# @firecrawl/pdf-inspector-wasm
|
||||
|
||||
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```ts
|
||||
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
|
||||
|
||||
await init();
|
||||
|
||||
const response = await fetch("/annual-report.pdf");
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
Pass options when you need selected pages or compact Markdown:
|
||||
|
||||
```ts
|
||||
const result = processPdf(pdf, {
|
||||
pages: [1, 3, 5],
|
||||
profile: "compact",
|
||||
includePageMarkers: true,
|
||||
});
|
||||
```
|
||||
|
||||
The package also exports:
|
||||
|
||||
- `detectPdf(pdf, options?)` for detection without extraction.
|
||||
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
|
||||
- `extractText(pdf)` for plain text.
|
||||
- `version()` for the WASM package version.
|
||||
|
||||
## Browser behavior
|
||||
|
||||
- Parsing runs locally. PDF bytes are not uploaded anywhere.
|
||||
- The build is single-threaded and does not require cross-origin isolation.
|
||||
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
|
||||
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
|
||||
- Image-only documents still require a separate OCR step.
|
||||
|
||||
## Build from source
|
||||
|
||||
```bash
|
||||
cargo install wasm-pack --version 0.15.0 --locked
|
||||
wasm-pack build wasm --target web --scope firecrawl --release
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
+440
@@ -0,0 +1,440 @@
|
||||
use pdf_inspector::{
|
||||
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
|
||||
ProcessMode,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wasm_bindgen::prelude::*;
|
||||
|
||||
#[wasm_bindgen(typescript_custom_section)]
|
||||
const TYPESCRIPT_TYPES: &str = r#"
|
||||
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
|
||||
export type MarkdownProfile = "fidelity" | "compact";
|
||||
|
||||
export interface ProcessOptions {
|
||||
/** Restrict extraction to these 1-indexed page numbers. */
|
||||
pages?: number[];
|
||||
/** Password for an encrypted PDF. */
|
||||
password?: string;
|
||||
/** Source-faithful output by default, or compact output for fewer tokens. */
|
||||
profile?: MarkdownProfile;
|
||||
/** Insert `<!-- Page N -->` markers between pages. */
|
||||
includePageMarkers?: boolean;
|
||||
/** Include image placeholders in Markdown output. */
|
||||
includeImages?: boolean;
|
||||
}
|
||||
|
||||
export interface PageOcrReasons {
|
||||
/** 1-indexed page number. */
|
||||
page: number;
|
||||
reasons: string[];
|
||||
}
|
||||
|
||||
export interface LayoutComplexity {
|
||||
isComplex: boolean;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithTables: number[];
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithColumns: number[];
|
||||
}
|
||||
|
||||
export interface PdfProcessResult {
|
||||
pdfType: PdfType;
|
||||
markdown?: string;
|
||||
pageCount: number;
|
||||
processingTimeMs: number;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesNeedingOcr: number[];
|
||||
ocrReasonsByPage: PageOcrReasons[];
|
||||
title?: string;
|
||||
confidence: number;
|
||||
layout: LayoutComplexity;
|
||||
hasEncodingIssues: boolean;
|
||||
}
|
||||
|
||||
export interface PdfClassification {
|
||||
pdfType: PdfType;
|
||||
pageCount: number;
|
||||
/** 0-indexed page numbers, matching the native Node.js API. */
|
||||
pagesNeedingOcr: number[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
|
||||
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
|
||||
export function classifyPdf(data: Uint8Array): PdfClassification;
|
||||
export function extractText(data: Uint8Array): string;
|
||||
export function version(): string;
|
||||
"#;
|
||||
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
|
||||
struct WasmProcessOptions {
|
||||
pages: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
profile: Option<WasmMarkdownProfile>,
|
||||
include_page_markers: Option<bool>,
|
||||
include_images: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
enum WasmMarkdownProfile {
|
||||
Fidelity,
|
||||
Compact,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPageOcrReasons {
|
||||
page: u32,
|
||||
reasons: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<PageOcrReasons> for WasmPageOcrReasons {
|
||||
fn from(value: PageOcrReasons) -> Self {
|
||||
Self {
|
||||
page: value.page,
|
||||
reasons: value.reasons,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmLayoutComplexity {
|
||||
is_complex: bool,
|
||||
pages_with_tables: Vec<u32>,
|
||||
pages_with_columns: Vec<u32>,
|
||||
}
|
||||
|
||||
impl From<LayoutComplexity> for WasmLayoutComplexity {
|
||||
fn from(value: LayoutComplexity) -> Self {
|
||||
Self {
|
||||
is_complex: value.is_complex,
|
||||
pages_with_tables: value.pages_with_tables,
|
||||
pages_with_columns: value.pages_with_columns,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfProcessResult {
|
||||
pdf_type: &'static str,
|
||||
markdown: Option<String>,
|
||||
page_count: u32,
|
||||
processing_time_ms: f64,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
|
||||
title: Option<String>,
|
||||
confidence: f64,
|
||||
layout: WasmLayoutComplexity,
|
||||
has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
impl From<PdfProcessResult> for WasmPdfProcessResult {
|
||||
fn from(value: PdfProcessResult) -> Self {
|
||||
Self {
|
||||
pdf_type: pdf_type_name(value.pdf_type),
|
||||
markdown: value.markdown,
|
||||
page_count: value.page_count,
|
||||
processing_time_ms: value.processing_time_ms as f64,
|
||||
pages_needing_ocr: value.pages_needing_ocr,
|
||||
ocr_reasons_by_page: value
|
||||
.ocr_reasons_by_page
|
||||
.into_iter()
|
||||
.map(Into::into)
|
||||
.collect(),
|
||||
title: value.title,
|
||||
confidence: value.confidence as f64,
|
||||
layout: value.layout.into(),
|
||||
has_encoding_issues: value.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfClassification {
|
||||
pdf_type: &'static str,
|
||||
page_count: u32,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
confidence: f64,
|
||||
}
|
||||
|
||||
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
|
||||
match pdf_type {
|
||||
PdfType::TextBased => "TextBased",
|
||||
PdfType::Scanned => "Scanned",
|
||||
PdfType::ImageBased => "ImageBased",
|
||||
PdfType::Mixed => "Mixed",
|
||||
}
|
||||
}
|
||||
|
||||
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
|
||||
js_sys::Error::new(&format!("{context}: {error}")).into()
|
||||
}
|
||||
|
||||
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
|
||||
if value.is_undefined() || value.is_null() {
|
||||
return Ok(WasmProcessOptions::default());
|
||||
}
|
||||
|
||||
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
|
||||
}
|
||||
|
||||
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
|
||||
let options = deserialize_options(value)?;
|
||||
if options
|
||||
.pages
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(js_error(
|
||||
"invalid options",
|
||||
"pages are 1-indexed; page 0 is invalid",
|
||||
));
|
||||
}
|
||||
|
||||
let mut result = PdfOptions::new().mode(mode);
|
||||
if let Some(pages) = options.pages {
|
||||
result = result.pages(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(profile) = options.profile {
|
||||
result.markdown.profile = match profile {
|
||||
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
|
||||
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
|
||||
};
|
||||
}
|
||||
if let Some(include_page_markers) = options.include_page_markers {
|
||||
result.markdown.include_page_numbers = include_page_markers;
|
||||
}
|
||||
if let Some(include_images) = options.include_images {
|
||||
result.markdown.include_images = include_images;
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
|
||||
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
|
||||
}
|
||||
|
||||
fn initialize() {
|
||||
console_error_panic_hook::set_once();
|
||||
}
|
||||
|
||||
/// Process PDF bytes entirely inside WebAssembly.
|
||||
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
|
||||
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::Full)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("process PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Classify PDF bytes without extracting text or producing Markdown.
|
||||
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
|
||||
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::DetectOnly)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("detect PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Return the lightweight classification shape used by the native Node API.
|
||||
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
|
||||
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
|
||||
serialize(&WasmPdfClassification {
|
||||
pdf_type: pdf_type_name(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes without Markdown conversion.
|
||||
#[wasm_bindgen(js_name = extractText, skip_typescript)]
|
||||
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
|
||||
initialize();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
|
||||
.map_err(|error| js_error("extract text", error))?;
|
||||
Ok(
|
||||
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
|
||||
.into_iter()
|
||||
.map(|line| line.text())
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n"),
|
||||
)
|
||||
}
|
||||
|
||||
/// Return the WebAssembly package version.
|
||||
#[wasm_bindgen(skip_typescript)]
|
||||
pub fn version() -> String {
|
||||
env!("CARGO_PKG_VERSION").to_string()
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_arch = "wasm32"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use js_sys::Reflect;
|
||||
use wasm_bindgen_test::*;
|
||||
|
||||
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
|
||||
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
|
||||
|
||||
fn synthetic_korea1_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
|
||||
// There is deliberately no ToUnicode stream: decoding must use the
|
||||
// embedded predefined CMap rather than lopdf's plain-text fallback.
|
||||
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
|
||||
// the bottom edge so they look exactly like a numeric page footer.
|
||||
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn processes_pdf_to_markdown() {
|
||||
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
|
||||
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn rejects_non_pdf_bytes() {
|
||||
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn classifies_and_extracts_plain_text() {
|
||||
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
|
||||
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let text = extract_text(TEXT_PDF).expect("extract text");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!text.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn extracts_cjk_and_preserves_numeric_page_footer() {
|
||||
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
|
||||
|
||||
assert_eq!(text, "가\n42");
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn opens_encrypted_pdf_with_password() {
|
||||
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
|
||||
|
||||
let options = js_sys::Object::new();
|
||||
Reflect::set(
|
||||
&options,
|
||||
&JsValue::from_str("password"),
|
||||
&JsValue::from_str("secret123"),
|
||||
)
|
||||
.expect("set password");
|
||||
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user