Compare commits
92
Commits
+45
-31
@@ -20,21 +20,14 @@ jobs:
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/bin/
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
target/
|
||||
key: ${{ runner.os }}-cargo-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-
|
||||
uses: Swatinem/rust-cache@v2
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
|
||||
- name: Test developer scripts
|
||||
run: python3 -m unittest discover -s scripts/tests
|
||||
|
||||
fmt:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
@@ -49,6 +42,9 @@ jobs:
|
||||
- name: Check formatting
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Check WASM formatting
|
||||
run: cargo fmt --manifest-path wasm/Cargo.toml -- --check
|
||||
|
||||
clippy:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
@@ -61,17 +57,9 @@ jobs:
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/bin/
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
target/
|
||||
key: ${{ runner.os }}-cargo-clippy-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-clippy-
|
||||
key: clippy
|
||||
|
||||
- name: Run clippy
|
||||
run: cargo clippy -- -D warnings
|
||||
@@ -89,17 +77,43 @@ jobs:
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/bin/
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
target/
|
||||
key: ${{ runner.os }}-cargo-build-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-build-
|
||||
key: build
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
wasm:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
key: wasm
|
||||
|
||||
- name: Check WebAssembly bindings
|
||||
run: cargo check --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown
|
||||
|
||||
- name: Check root package for WebAssembly
|
||||
run: cargo check --target wasm32-unknown-unknown
|
||||
|
||||
- name: Lint WebAssembly bindings
|
||||
run: cargo clippy --manifest-path wasm/Cargo.toml --target wasm32-unknown-unknown -- -D warnings
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Test WebAssembly package
|
||||
run: wasm-pack test --node --release wasm
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
name: Deploy landing page
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['site/**', '.github/workflows/pages.yml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pages: write
|
||||
id-token: write
|
||||
|
||||
# Allow one concurrent deployment; don't cancel an in-progress production deploy.
|
||||
concurrency:
|
||||
group: pages
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
name: Build & deploy to GitHub Pages
|
||||
runs-on: ubuntu-latest
|
||||
environment:
|
||||
name: github-pages
|
||||
url: ${{ steps.deploy.outputs.page_url }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Upload site artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deploy
|
||||
uses: actions/deploy-pages@v4
|
||||
@@ -0,0 +1,103 @@
|
||||
name: Publish Rust crate
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['Cargo.toml']
|
||||
# Manual fallback: retry a publish that failed after the version was
|
||||
# already merged (a plain re-push won't register as a version change).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches: crates.io trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch publishes the current version regardless of the
|
||||
# previous commit; the crates.io check below still prevents
|
||||
# double-publishing an already-released version.
|
||||
echo "manual dispatch: publishing v$NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
|
||||
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
|
||||
"https://crates.io/api/v1/crates/pdf-inspector/$NEW_VERSION")
|
||||
|
||||
case "$HTTP_STATUS" in
|
||||
200)
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
echo "pdf-inspector v$NEW_VERSION is already published"
|
||||
;;
|
||||
404)
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
*)
|
||||
cat /tmp/crate-version.json
|
||||
echo "Unexpected crates.io response: $HTTP_STATUS" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
publish:
|
||||
name: Publish to crates.io
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
environment: crates-io
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
env:
|
||||
CARGO_REGISTRY_TOKEN: ${{ steps.auth.outputs.token }}
|
||||
@@ -0,0 +1,166 @@
|
||||
name: Publish Python package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['pyproject.toml']
|
||||
# Manual fallback: re-publish the current version without a version bump
|
||||
# (e.g. first run after PyPI trusted publishing is configured).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches too: PyPI trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("pyproject.toml").read_text())["project"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch always rebuilds and publishes. Combined with
|
||||
# skip-existing on the publish step, this repairs partial releases
|
||||
# (PyPI's version endpoint returns 200 even when only some of the
|
||||
# expected wheels were uploaded).
|
||||
echo "manual dispatch: publishing v$NEW_VERSION (skip-existing handles uploaded files)"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# .get(): the parent commit may predate the static version field
|
||||
# (pyproject.toml used dynamic = ["version"]) — treat that as a change
|
||||
# so the very first merge of this workflow publishes.
|
||||
OLD_VERSION=$(git show HEAD~1:pyproject.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["project"].get("version", ""))')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/pypi-version.json --write-out "%{http_code}" \
|
||||
"https://pypi.org/pypi/pdf-inspector/$NEW_VERSION/json")
|
||||
|
||||
case "$HTTP_STATUS" in
|
||||
200)
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
echo "pdf-inspector v$NEW_VERSION is already published to PyPI"
|
||||
;;
|
||||
404)
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
*)
|
||||
cat /tmp/pypi-version.json
|
||||
echo "Unexpected PyPI response: $HTTP_STATUS" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
build:
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
name: Build ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
# macos-13 was retired by GitHub; macos-15-intel is the remaining
|
||||
# Intel runner label (available through 2027).
|
||||
- os: macos-15-intel
|
||||
target: x86_64-apple-darwin
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Build wheel
|
||||
uses: PyO3/maturin-action@v1
|
||||
with:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: wheels-${{ matrix.target }}
|
||||
path: dist/*.whl
|
||||
if-no-files-found: error
|
||||
|
||||
sdist:
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
name: Build sdist
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Build sdist
|
||||
uses: PyO3/maturin-action@v1
|
||||
with:
|
||||
command: sdist
|
||||
args: --out dist
|
||||
|
||||
- name: Upload sdist
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: sdist
|
||||
path: dist/*.tar.gz
|
||||
if-no-files-found: error
|
||||
|
||||
publish:
|
||||
name: Publish to PyPI
|
||||
needs: [check-version, build, sdist]
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
|
||||
- name: List artifacts
|
||||
run: ls -la dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: dist
|
||||
# Tolerate already-uploaded files so a manual re-run can complete
|
||||
# a release that previously failed partway through.
|
||||
skip-existing: true
|
||||
@@ -0,0 +1,111 @@
|
||||
name: Publish WebAssembly package
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['wasm/Cargo.toml']
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
package_exists: ${{ steps.check.outputs.package_exists }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check package version
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("wasm/Cargo.toml").read_text())["package"]["version"])')
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
elif git cat-file -e HEAD~1:wasm/Cargo.toml 2>/dev/null; then
|
||||
OLD_VERSION=$(git show HEAD~1:wasm/Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
if ! npm view "@firecrawl/pdf-inspector-wasm" name >/dev/null 2>&1; then
|
||||
echo "package_exists=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
echo "The initial package must be published once before trusted publishing can be configured."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "package_exists=true" >> "$GITHUB_OUTPUT"
|
||||
if npm view "@firecrawl/pdf-inspector-wasm@$NEW_VERSION" version >/dev/null 2>&1; then
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
publish:
|
||||
name: Build and publish
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.package_exists == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install wasm-pack
|
||||
run: cargo install wasm-pack --version 0.15.0 --locked
|
||||
|
||||
- name: Build browser package
|
||||
run: wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release
|
||||
|
||||
- name: Prepare package metadata
|
||||
run: |
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const path = "wasm/pkg/package.json"
|
||||
const pkg = JSON.parse(fs.readFileSync(path, "utf8"))
|
||||
pkg.name = "@firecrawl/pdf-inspector-wasm"
|
||||
pkg.description = "Browser WebAssembly bindings for the pdf-inspector Rust PDF parser"
|
||||
pkg.keywords = ["pdf", "pdf-parser", "webassembly", "wasm", "markdown", "rust", "firecrawl"]
|
||||
pkg.repository = { type: "git", url: "https://github.com/firecrawl/pdf-inspector" }
|
||||
pkg.homepage = "https://github.com/firecrawl/pdf-inspector/tree/main/wasm"
|
||||
pkg.publishConfig = { access: "public" }
|
||||
fs.writeFileSync(path, JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
- name: Inspect package contents
|
||||
run: npm pack --dry-run ./wasm/pkg
|
||||
|
||||
- name: Publish package
|
||||
run: npm publish ./wasm/pkg --provenance --access public
|
||||
@@ -4,6 +4,9 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['napi/package.json']
|
||||
# Manual fallback: retry a publish that failed partway (per-package
|
||||
# already-published checks make re-runs idempotent).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -12,6 +15,10 @@ permissions:
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches: npm trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
@@ -25,11 +32,21 @@ jobs:
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(node -p "require('./napi/package.json').version")
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch rebuilds and publishes the current version; the
|
||||
# per-package already-published checks in the publish job skip
|
||||
# anything that made it out in a previous partial run.
|
||||
echo "manual dispatch: publishing v$NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
OLD_VERSION=$(git show HEAD~1:napi/package.json | node -p "JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')).version")
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
@@ -115,14 +132,81 @@ jobs:
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
- name: Collect binaries and publish
|
||||
- name: Publish platform packages
|
||||
working-directory: napi
|
||||
run: |
|
||||
cp artifacts/bindings-*/*.node .
|
||||
VERSION="${{ needs.check-version.outputs.version }}"
|
||||
|
||||
for node_file in artifacts/bindings-*/pdf-inspector.*.node; do
|
||||
base=$(basename "$node_file")
|
||||
suffix=${base#pdf-inspector.}
|
||||
suffix=${suffix%.node}
|
||||
pkg="@firecrawl/pdf-inspector-$suffix"
|
||||
|
||||
if npm view "$pkg@$VERSION" version >/dev/null 2>&1; then
|
||||
echo "$pkg@$VERSION already published — skipping"
|
||||
continue
|
||||
fi
|
||||
|
||||
dir="npm-dist/$suffix"
|
||||
mkdir -p "$dir"
|
||||
cp "$node_file" "$dir/"
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
process.exit(1)
|
||||
}
|
||||
const pkg = {
|
||||
name: `@firecrawl/pdf-inspector-${suffix}`,
|
||||
version,
|
||||
description: `Prebuilt ${suffix} binary for @firecrawl/pdf-inspector`,
|
||||
main: `pdf-inspector.${suffix}.node`,
|
||||
files: [`pdf-inspector.${suffix}.node`],
|
||||
license: "MIT",
|
||||
engines: { node: ">= 10" },
|
||||
repository: { type: "git", url: "https://github.com/firecrawl/pdf-inspector" },
|
||||
publishConfig: { access: "public" },
|
||||
...meta,
|
||||
}
|
||||
require("fs").writeFileSync(`npm-dist/${suffix}/package.json`, JSON.stringify(pkg, null, 2) + "\n")
|
||||
' "$suffix" "$VERSION"
|
||||
|
||||
echo "=== $pkg@$VERSION ==="
|
||||
ls -la "$dir"
|
||||
(cd "$dir" && npm publish --provenance --access public)
|
||||
done
|
||||
|
||||
- name: Publish main package
|
||||
working-directory: napi
|
||||
run: |
|
||||
VERSION="${{ needs.check-version.outputs.version }}"
|
||||
|
||||
if npm view "@firecrawl/pdf-inspector@$VERSION" version >/dev/null 2>&1; then
|
||||
echo "@firecrawl/pdf-inspector@$VERSION already published — skipping"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
cp artifacts/js-bindings/index.js .
|
||||
cp artifacts/js-bindings/index.d.ts .
|
||||
|
||||
echo "=== Package contents ==="
|
||||
ls -la *.node index.js index.d.ts
|
||||
# Stamp optionalDependencies to this exact version so the platform
|
||||
# pins can never drift from the main package version.
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const pkg = JSON.parse(fs.readFileSync("package.json", "utf8"))
|
||||
for (const dep of Object.keys(pkg.optionalDependencies ?? {})) {
|
||||
pkg.optionalDependencies[dep] = pkg.version
|
||||
}
|
||||
fs.writeFileSync("package.json", JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
echo "=== Main package contents ==="
|
||||
npm pack --dry-run
|
||||
|
||||
npm publish --provenance --access public
|
||||
|
||||
+3
-1
@@ -1,10 +1,13 @@
|
||||
# Rust build artifacts
|
||||
/target/
|
||||
/wasm/target/
|
||||
/wasm/pkg/
|
||||
debug/
|
||||
*.pdb
|
||||
|
||||
# Cargo lock (optional for libraries)
|
||||
Cargo.lock
|
||||
!/wasm/Cargo.lock
|
||||
|
||||
# IDE
|
||||
.idea/
|
||||
@@ -39,4 +42,3 @@ test_output/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
+28
-9
@@ -1,12 +1,25 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
version = "0.1.6"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "docs/rust-api.md"
|
||||
# Explicit allowlist: crates.io caps uploads at 10 MiB and tests/fixtures
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"/src/**",
|
||||
"/external/bcmaps/**",
|
||||
"/docs/rust-api.md",
|
||||
"/LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"/pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
name = "pdf_inspector"
|
||||
@@ -14,20 +27,13 @@ crate-type = ["lib", "cdylib"]
|
||||
|
||||
[dependencies]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "7a05512d831415b1f2b1ce522391d6beab8a1284", features = ["rayon"] }
|
||||
pyo3 = { version = "0.25", features = ["extension-module", "abi3-py38"], optional = true }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
# Parallel processing
|
||||
rayon = "1.10"
|
||||
|
||||
# Logging
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Text processing
|
||||
regex = "1.10"
|
||||
@@ -37,6 +43,19 @@ unicode-normalization = "0.1"
|
||||
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
||||
ttf-parser = "0.25"
|
||||
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.41.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3.3"
|
||||
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -1,6 +1,11 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
[](https://crates.io/crates/pdf-inspector)
|
||||
[](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md), [Node.js](napi/README.md), and [browser WebAssembly](wasm/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -14,24 +19,28 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Browser WebAssembly** — Run the same Rust parser locally in browsers and Web Workers, with embedded CMaps and no server round trip.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only local engines without model-based PDF parsing are shown; OCR was disabled. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | 0.78 | 0.87 | 0.59 | 0.57 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
| pdf-inspector | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus.
|
||||
Results were refreshed on July 16, 2026, on an Apple M4 Pro. Engine versions were pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1, PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Speed is the median of three complete corpus runs.
|
||||
|
||||
**Where we do well:** Speed (fastest of all engines), reading order, table detection vs other direct-text tools.
|
||||
For context, engines that use OCR or model-based document parsing (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the top of that range without either, in 2.8 seconds.
|
||||
|
||||
**Where we lag:** Heading detection trails opendataloader — many PDFs use bold text at body font size for headings, or headings that are only slightly larger than body text. Table detection trails OCR-based engines that can see visual table structure.
|
||||
**Best fit:** Native-text PDFs where speed, reading order, and table structure matter. pdf-inspector delivered the highest overall, reading-order, and table scores, along with the fastest complete run in this benchmark. That makes it a strong local default for reports, research papers, financial documents, invoices, and legal PDFs that need clean, structured Markdown without adding OCR latency or infrastructure.
|
||||
|
||||
Use the [paired benchmark harness](docs/benchmarking.md) to compare two local builds against the exact same corpus and evaluator revision.
|
||||
|
||||
## Quick start
|
||||
|
||||
@@ -69,11 +78,39 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Browser WebAssembly
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
```javascript
|
||||
import init, { processPdf } from '@firecrawl/pdf-inspector-wasm';
|
||||
|
||||
await init();
|
||||
const response = await fetch('/document.pdf');
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
> Full API reference: [wasm/README.md](wasm/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
|
||||
```bash
|
||||
cargo add pdf-inspector
|
||||
```
|
||||
|
||||
Or add it manually:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
pdf-inspector = "0.1"
|
||||
```
|
||||
|
||||
```rust
|
||||
@@ -91,29 +128,40 @@ if let Some(markdown) = &result.markdown {
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
# Install the CLI tools
|
||||
cargo install pdf-inspector
|
||||
|
||||
# Convert PDF to Markdown
|
||||
cargo run --bin pdf2md -- document.pdf
|
||||
pdf2md document.pdf
|
||||
|
||||
# JSON output (for piping)
|
||||
cargo run --bin pdf2md -- document.pdf --json
|
||||
pdf2md document.pdf --json
|
||||
|
||||
# Positioned TextItem JSON, including is_underline metadata
|
||||
pdf2md document.pdf --items-json
|
||||
|
||||
# Raw markdown only (no headers)
|
||||
cargo run --bin pdf2md -- document.pdf --raw
|
||||
pdf2md document.pdf --raw
|
||||
|
||||
# Token-efficient output (collapses long dot leaders and similar source padding)
|
||||
pdf2md document.pdf --compact
|
||||
|
||||
# Insert page break markers (<!-- Page N -->)
|
||||
cargo run --bin pdf2md -- document.pdf --pages
|
||||
pdf2md document.pdf --pages
|
||||
|
||||
# Process only specific pages
|
||||
cargo run --bin pdf2md -- document.pdf --select-pages 1,3,5-10
|
||||
pdf2md document.pdf --select-pages 1,3,5-10
|
||||
|
||||
# Detection only (no extraction)
|
||||
cargo run --bin detect-pdf -- document.pdf
|
||||
cargo run --bin detect-pdf -- document.pdf --json
|
||||
detect-pdf document.pdf
|
||||
detect-pdf document.pdf --json
|
||||
|
||||
# Detection + layout analysis (tables, columns)
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze --json
|
||||
detect-pdf document.pdf --analyze --json
|
||||
```
|
||||
|
||||
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
@@ -161,6 +209,7 @@ src/
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
wasm/ — Browser bindings (wasm-bindgen)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
@@ -223,4 +272,4 @@ See [docs/debugging.md](docs/debugging.md) for `RUST_LOG` environment variable u
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
[MIT](LICENSE)
|
||||
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
# Security Policy
|
||||
|
||||
## Reporting a Vulnerability
|
||||
|
||||
If you believe you've found a security vulnerability in pdf-inspector, please
|
||||
report it privately so we can fix it before public disclosure.
|
||||
|
||||
**Preferred:** Email **help@firecrawl.dev** with:
|
||||
|
||||
- A description of the issue and its impact
|
||||
- Steps to reproduce (a minimal PDF or input that triggers the bug is ideal)
|
||||
- The version or commit hash of pdf-inspector you tested against
|
||||
|
||||
**Alternative:** Use GitHub's private vulnerability reporting under the
|
||||
[Security tab](https://github.com/firecrawl/pdf-inspector/security/advisories/new).
|
||||
|
||||
We'll acknowledge your report in a timely manner and keep you updated on
|
||||
remediation progress. Please do not open a public GitHub issue for security
|
||||
bugs.
|
||||
|
||||
## Scope
|
||||
|
||||
In scope:
|
||||
- Memory-safety issues (panics, OOB reads, UB) reachable from a crafted PDF
|
||||
- Denial-of-service vectors (unbounded allocation, infinite loops) on
|
||||
reasonably-sized inputs
|
||||
- Bugs in the `pdf2md` / `detect-pdf` binaries or the `pdf-inspector` crate
|
||||
that affect downstream consumers
|
||||
|
||||
Out of scope:
|
||||
- Bugs in upstream dependencies (`lopdf`, etc.) — please report those upstream
|
||||
- Extraction quality issues (wrong text, missing tables) — open a regular
|
||||
GitHub issue instead
|
||||
@@ -0,0 +1,59 @@
|
||||
# Benchmarking against OpenDataLoader
|
||||
|
||||
The paired harness runs two `pdf2md` binaries through the same local
|
||||
OpenDataLoader corpus, evaluates both outputs, and reports aggregate and
|
||||
per-document deltas. This avoids comparing results produced from different
|
||||
corpus revisions or evaluator versions.
|
||||
|
||||
Build a candidate and provide a released or worktree build as the baseline:
|
||||
|
||||
```bash
|
||||
cargo build --release
|
||||
python3 scripts/bench_opendataloader.py \
|
||||
--bench-dir ../opendataloader-bench \
|
||||
--baseline ../pdf-inspector-main/target/release/pdf2md \
|
||||
--candidate target/release/pdf2md \
|
||||
--max-document-regression 0.02 \
|
||||
--json-output /tmp/pdf-inspector-benchmark.json
|
||||
```
|
||||
|
||||
Pass `--reference-evaluation path/to/evaluation.json` to report the candidate
|
||||
delta against another evaluation, and add `--require-reference-lead` to make a
|
||||
negative reference delta fail the run. By default, the candidate must not
|
||||
regress the baseline overall score or introduce missing predictions. Use
|
||||
`--min-overall-delta` to require a specific aggregate gain.
|
||||
|
||||
The OpenDataLoader repository is external and keeps its normal
|
||||
`prediction/pdf-inspector` output. Paired evaluation copies each run into a
|
||||
temporary directory before evaluating it, so the baseline and candidate cannot
|
||||
overwrite one another.
|
||||
|
||||
## Published comparison protocol
|
||||
|
||||
The public benchmark table was refreshed on July 16, 2026, on an Apple M4 Pro
|
||||
using pdf-inspector 0.1.6, LiteParse 2.6.0, OpenDataLoader 2.1.1,
|
||||
PyMuPDF4LLM 0.2.0, and MarkItDown 0.1.4. Every engine processed the same 200
|
||||
PDFs with OCR disabled. Reported speed is the median of three complete corpus
|
||||
runs; quality scores come from the benchmark evaluator over all 200 outputs.
|
||||
|
||||
## Optional backend evidence probe
|
||||
|
||||
The evidence probe compares positioned `pdf2md` items with MuPDF structured
|
||||
text on the same pages. It is intended to find deterministic extraction or
|
||||
layout evidence that could justify a future native implementation; it does not
|
||||
merge MuPDF output into Markdown, invoke OCR, or add a runtime dependency.
|
||||
|
||||
Install MuPDF's `mutool`, build `pdf2md`, then run:
|
||||
|
||||
```bash
|
||||
python3 scripts/probe_backend_evidence.py document.pdf \
|
||||
--pdf2md target/release/pdf2md \
|
||||
--json-output /tmp/backend-evidence.json
|
||||
```
|
||||
|
||||
The report flags pages when MuPDF exposes a material net token gain, repeated
|
||||
alignment anchors absent from local evidence, or additional image blocks. The
|
||||
JSON includes bounded token samples and page-level counts so promising cases
|
||||
can be inspected without treating backend disagreement as automatically
|
||||
correct. Thresholds are configurable with `--min-token-gain`,
|
||||
`--min-alternate-only-ratio`, and `--min-anchor-gain`.
|
||||
@@ -0,0 +1,38 @@
|
||||
# Publishing
|
||||
|
||||
The Rust crate is published to [crates.io](https://crates.io/crates/pdf-inspector) with trusted publishing from GitHub Actions. The first release was published manually; future releases publish from `.github/workflows/publish-crate.yml` when a `Cargo.toml` version change lands on `main`.
|
||||
|
||||
## crates.io Trusted Publisher
|
||||
|
||||
Configure the trusted publisher for the `pdf-inspector` crate with:
|
||||
|
||||
- Repository: `firecrawl/pdf-inspector`
|
||||
- Workflow: `publish-crate.yml`
|
||||
- Environment: `crates-io`
|
||||
|
||||
The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC token for a short-lived crates.io token, then passes it to `cargo publish`.
|
||||
|
||||
## Release Steps
|
||||
|
||||
1. Update `version` in `Cargo.toml`.
|
||||
2. Merge the version bump to `main`.
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
|
||||
## Browser WebAssembly package
|
||||
|
||||
The browser package is published as `@firecrawl/pdf-inspector-wasm`. Its version lives in `wasm/Cargo.toml`, and `.github/workflows/publish-wasm.yml` builds the `web` target with `wasm-pack` before publishing the generated package.
|
||||
|
||||
The npm package must exist before a trusted publisher can be configured. For the first release only:
|
||||
|
||||
1. Build with `wasm-pack build wasm --target web --scope firecrawl --out-dir pkg --release`.
|
||||
2. Inspect with `npm pack --dry-run ./wasm/pkg`.
|
||||
3. Publish with `npm publish ./wasm/pkg --access public` from an authorized maintainer session.
|
||||
4. In the package settings on npm, configure the GitHub Actions trusted publisher:
|
||||
- Organization: `firecrawl`
|
||||
- Repository: `pdf-inspector`
|
||||
- Workflow: `publish-wasm.yml`
|
||||
- Allowed action: `npm publish`
|
||||
|
||||
After that one-time bootstrap, bumping the version in `wasm/Cargo.toml` and merging it to `main` publishes through OIDC. Until the package exists, the workflow exits cleanly without attempting an unauthenticated first publish. See npm's [trusted publishing documentation](https://docs.npmjs.com/trusted-publishers/) for the registry-side setup.
|
||||
+75
-10
@@ -1,9 +1,39 @@
|
||||
# Python API
|
||||
# pdf-inspector
|
||||
|
||||
Python bindings via [PyO3](https://pyo3.rs). Requires Rust toolchain for building from source.
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — `text_based` / `scanned` / `image_based` / `mixed` in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core, no ML models, no external services; ships type stubs.
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt wheels cover CPython ≥3.8 on Linux (x86_64, aarch64), macOS (Intel, Apple Silicon), and Windows (x64). Other platforms build from source, which requires a Rust toolchain. For local development in a repo checkout:
|
||||
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
@@ -73,16 +103,51 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
|
||||
## Types
|
||||
|
||||
**`PdfResult` fields:** `pdf_type`, `markdown`, `page_count`, `processing_time_ms`, `pages_needing_ocr`, `title`, `confidence`, `is_complex_layout`, `pages_with_tables`, `pages_with_columns`, `has_encoding_issues`
|
||||
Type stubs (`pdf_inspector.pyi`) ship with the package. Result types at a glance:
|
||||
|
||||
**`PdfClassification` fields:** `pdf_type`, `page_count`, `pages_needing_ocr` (0-indexed), `confidence`
|
||||
```python
|
||||
class PdfResult: # process_pdf / detect_pdf
|
||||
pdf_type: str # "text_based" | "scanned" | "image_based" | "mixed"
|
||||
markdown: str | None # extracted Markdown (None for detect_pdf)
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
title: str | None
|
||||
confidence: float # 0.0 - 1.0
|
||||
is_complex_layout: bool
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool # broken font encodings — consider OCR fallback
|
||||
|
||||
**`TextItem` fields:** `text`, `x`, `y`, `width`, `height`, `font`, `font_size`, `page`, `is_bold`, `is_italic`, `item_type`
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
pages_needing_ocr: list[int] # 0-indexed
|
||||
confidence: float
|
||||
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
class TextItem: # extract_text_with_positions
|
||||
text: str
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
font: str
|
||||
font_size: float
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
class PageRegionTexts: # extract_text_in_regions
|
||||
page: int # 0-indexed
|
||||
regions: list[RegionText] # RegionText: text: str, needs_ocr: bool
|
||||
|
||||
**`PageMarkdown` fields:** `page` (0-indexed), `markdown`, `needs_ocr`
|
||||
|
||||
**`PagesExtractionResult` fields:** `pages` (list of PageMarkdown), `pages_with_tables` (1-indexed), `pages_with_columns` (1-indexed), `pages_needing_ocr` (1-indexed), `is_complex`
|
||||
class PagesExtractionResult: # extract_pages_markdown
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr
|
||||
pages_with_tables: list[int] # 1-indexed
|
||||
pages_with_columns: list[int] # 1-indexed
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
is_complex: bool # any page has tables or multi-column layout
|
||||
```
|
||||
|
||||
+40
-2
@@ -1,12 +1,50 @@
|
||||
# Rust API
|
||||
# pdf-inspector
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Pure Rust, no ML models, no external services; the only PDF dependency is [lopdf](https://crates.io/crates/lopdf). Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — TextBased / Scanned / ImageBased / Mixed in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — pure Rust, no ML models, no external services; single PDF dependency ([lopdf](https://crates.io/crates/lopdf)).
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
cargo add pdf-inspector
|
||||
```
|
||||
|
||||
For the latest unreleased changes, use the git dependency instead:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
The crate also ships CLI binaries — `pdf2md` (PDF → Markdown, with `--json`, `--pages`, `--select-pages`, and the opt-in token-saving `--compact` profile) and `detect-pdf` (classification, with `--analyze --json`):
|
||||
|
||||
```bash
|
||||
cargo install pdf-inspector
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
Generated
+5
-4
@@ -672,8 +672,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.40.0"
|
||||
source = "git+https://github.com/J-F-Liu/lopdf?rev=7a05512d831415b1f2b1ce522391d6beab8a1284#7a05512d831415b1f2b1ce522391d6beab8a1284"
|
||||
version = "0.41.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -829,7 +830,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
version = "0.1.6"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"log",
|
||||
@@ -844,7 +845,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.0"
|
||||
version = "0.2.2"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.0"
|
||||
version = "0.2.2"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
+32
-6
@@ -4,6 +4,28 @@ Fast PDF classification and region-based text extraction for Node.js/Bun. Native
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract text from PDF structure where possible, fall back to OCR only when needed.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — text-based / scanned / image-based / mixed in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
||||
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~5–6 MB platform binary, TypeScript definitions included.
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), local engines without model-based PDF parsing; OCR disabled. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | **0.875** | **0.915** | **0.814** | 0.788 | **2.8s** |
|
||||
| liteparse | 0.870 | 0.908 | 0.693 | **0.811** | 13.9s |
|
||||
| opendataloader | 0.843 | 0.912 | 0.489 | 0.760 | 9.8s |
|
||||
| pymupdf4llm | 0.735 | 0.886 | 0.401 | 0.424 | 15.5s |
|
||||
| markitdown | 0.583 | 0.879 | 0.000 | 0.000 | 6.7s |
|
||||
|
||||
Refreshed July 16, 2026, on Apple M4 Pro; speed is the median of three complete corpus runs. Full methodology and versions are in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
@@ -12,7 +34,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64**, **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -37,7 +59,7 @@ console.log(result.confidence) // 0.875
|
||||
|
||||
Extract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR.
|
||||
|
||||
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues).
|
||||
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues). When the cause is a suspected garbled text layer, `ocrReason` is set to `"suspected_garbled_text"`.
|
||||
|
||||
```typescript
|
||||
import { extractTextInRegions } from '@firecrawl/pdf-inspector'
|
||||
@@ -84,15 +106,19 @@ interface PageRegionTexts {
|
||||
interface RegionText {
|
||||
text: string
|
||||
needsOcr: boolean // true when text is unreliable
|
||||
ocrReason?: string // "suspected_garbled_text" when known
|
||||
}
|
||||
```
|
||||
|
||||
## Platforms
|
||||
|
||||
| Platform | Architecture | Supported |
|
||||
|----------|-------------|-----------|
|
||||
| Linux | x64 | Yes |
|
||||
| macOS | ARM64 | Yes |
|
||||
Prebuilt binaries ship as platform-specific packages installed automatically via `optionalDependencies`:
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
@@ -7,6 +7,11 @@
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
"packages": {
|
||||
|
||||
+7
-6
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.8.5",
|
||||
"version": "1.11.2",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -22,7 +22,6 @@
|
||||
"files": [
|
||||
"index.js",
|
||||
"index.d.ts",
|
||||
"*.node",
|
||||
"bin/",
|
||||
"README.md"
|
||||
],
|
||||
@@ -40,10 +39,7 @@
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
],
|
||||
"package": {
|
||||
"name": "@firecrawl/pdf-inspector-js"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scripts": {
|
||||
"build": "napi build --platform --release",
|
||||
@@ -51,5 +47,10 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.2",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.2",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.2"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
import { readFileSync } from "node:fs";
|
||||
import { createRequire } from "node:module";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const { detectVectorGridInRegion } = require("./index.js");
|
||||
|
||||
const pdfPath =
|
||||
process.argv[2] ?? "/tmp/pdf_inspector_indent_fixtures/cis_edge_benchmark.pdf";
|
||||
const pdf = readFileSync(pdfPath);
|
||||
const dpi = Number(process.argv[3] ?? 200);
|
||||
|
||||
const crops = [
|
||||
{ pageIdx: 29, box: [0, 0, 612, 792], label: "page30-full" },
|
||||
{ pageIdx: 16, box: [0, 0, 612, 792], label: "page17-full" },
|
||||
{ pageIdx: 23, box: [0, 0, 612, 792], label: "page24-full" },
|
||||
];
|
||||
|
||||
for (const { pageIdx, box, label } of crops) {
|
||||
const result = detectVectorGridInRegion(pdf, pageIdx, box, dpi);
|
||||
if (!result) {
|
||||
console.log(`${label}: null`);
|
||||
continue;
|
||||
}
|
||||
const rows = result.structureTokens.filter((token) => token === "<tr>").length;
|
||||
const cols = rows > 0 ? result.cellBboxes.length / rows : 0;
|
||||
console.log(
|
||||
`${label}: cells=${result.cellBboxes.length} rows=${rows} cols=${cols}`,
|
||||
);
|
||||
}
|
||||
@@ -40,6 +40,8 @@ pub struct PdfResult {
|
||||
pub processing_time_ms: u32,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
pub title: Option<String>,
|
||||
pub confidence: f64,
|
||||
pub is_complex_layout: bool,
|
||||
@@ -48,6 +50,13 @@ pub struct PdfResult {
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
/// OCR reasons for a single 1-indexed page.
|
||||
#[napi(object)]
|
||||
pub struct PageOcrReasons {
|
||||
pub page: u32,
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[napi(object)]
|
||||
pub struct PdfClassification {
|
||||
@@ -71,6 +80,12 @@ pub struct TextItem {
|
||||
pub page: u32,
|
||||
pub is_bold: bool,
|
||||
pub is_italic: bool,
|
||||
/// Underline detected geometrically (drawn rule/thin rect under the
|
||||
/// baseline) — PDFs carry no underline font flag.
|
||||
pub is_underline: bool,
|
||||
/// Strikeout detected geometrically (rule crossing the glyphs at mid
|
||||
/// x-height).
|
||||
pub is_strikeout: bool,
|
||||
pub item_type: ItemType,
|
||||
/// URL for link items, `None` for other types.
|
||||
pub link_url: Option<String>,
|
||||
@@ -90,6 +105,8 @@ pub struct RegionText {
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
@@ -126,6 +143,7 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms as u32,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
title: r.title,
|
||||
confidence: r.confidence as f64,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
@@ -135,6 +153,18 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_page_ocr_reasons(
|
||||
reasons: Vec<pdf_inspector::PageOcrReasons>,
|
||||
) -> Vec<PageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PageOcrReasons {
|
||||
page: reason.page,
|
||||
reasons: reason.reasons,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||
@@ -266,6 +296,8 @@ pub fn extract_text_with_positions(
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type,
|
||||
link_url,
|
||||
}
|
||||
@@ -563,6 +595,8 @@ pub struct PageMarkdownResult {
|
||||
pub markdown: String,
|
||||
/// `true` when text on this page is unreliable.
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Combined per-page markdown extraction and layout classification result.
|
||||
@@ -576,6 +610,8 @@ pub struct PagesExtractionResult {
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// 1-indexed pages that need OCR (scanned/image-based).
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
/// True if any page has tables or columns.
|
||||
pub is_complex: bool,
|
||||
}
|
||||
@@ -607,11 +643,13 @@ pub fn extract_pages_markdown(
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
})
|
||||
})
|
||||
@@ -648,6 +686,7 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
|
||||
@@ -38,6 +38,8 @@ class TextItem:
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText:
|
||||
|
||||
+9
-1
@@ -4,8 +4,11 @@ build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.5"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
classifiers = [
|
||||
@@ -17,5 +20,10 @@ classifiers = [
|
||||
"Topic :: Text Processing",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
Repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
Documentation = "https://github.com/firecrawl/pdf-inspector/blob/main/docs/python.md"
|
||||
|
||||
[tool.maturin]
|
||||
features = ["python"]
|
||||
|
||||
@@ -0,0 +1,351 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run a paired pdf-inspector OpenDataLoader benchmark and report deltas."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SCORE_KEYS = (
|
||||
"overall_mean",
|
||||
"nid_mean",
|
||||
"nid_s_mean",
|
||||
"teds_mean",
|
||||
"teds_s_mean",
|
||||
"mhs_mean",
|
||||
"mhs_s_mean",
|
||||
)
|
||||
|
||||
|
||||
def _non_negative_int(value: str) -> int:
|
||||
parsed = int(value)
|
||||
if parsed < 0:
|
||||
raise argparse.ArgumentTypeError("must be non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _non_negative_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed) or parsed < 0.0:
|
||||
raise argparse.ArgumentTypeError("must be finite and non-negative")
|
||||
return parsed
|
||||
|
||||
|
||||
def _finite_float(value: str) -> float:
|
||||
parsed = float(value)
|
||||
if not math.isfinite(parsed):
|
||||
raise argparse.ArgumentTypeError("must be finite")
|
||||
return parsed
|
||||
|
||||
|
||||
def _scores(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
score = evaluation.get("metrics", {}).get("score", {})
|
||||
return {key: float(score[key]) for key in SCORE_KEYS if score.get(key) is not None}
|
||||
|
||||
|
||||
def _documents(evaluation: dict[str, Any]) -> dict[str, float]:
|
||||
documents: dict[str, float] = {}
|
||||
for document in evaluation.get("documents", []):
|
||||
overall = document.get("scores", {}).get("overall")
|
||||
if overall is not None:
|
||||
documents[str(document["document_id"])] = float(overall)
|
||||
return documents
|
||||
|
||||
|
||||
def compare_evaluations(
|
||||
baseline: dict[str, Any],
|
||||
candidate: dict[str, Any],
|
||||
reference: dict[str, Any] | None = None,
|
||||
*,
|
||||
top: int = 10,
|
||||
) -> dict[str, Any]:
|
||||
"""Build aggregate and per-document deltas from evaluator JSON payloads."""
|
||||
baseline_scores = _scores(baseline)
|
||||
candidate_scores = _scores(candidate)
|
||||
metric_deltas = {
|
||||
key: candidate_scores[key] - baseline_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in baseline_scores and key in candidate_scores
|
||||
}
|
||||
|
||||
baseline_documents = _documents(baseline)
|
||||
candidate_documents = _documents(candidate)
|
||||
shared = sorted(baseline_documents.keys() & candidate_documents.keys())
|
||||
document_deltas = [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"baseline": baseline_documents[document_id],
|
||||
"candidate": candidate_documents[document_id],
|
||||
"delta": candidate_documents[document_id] - baseline_documents[document_id],
|
||||
}
|
||||
for document_id in shared
|
||||
]
|
||||
epsilon = 1e-12
|
||||
improvements = sorted(document_deltas, key=lambda item: item["delta"], reverse=True)
|
||||
regressions = sorted(document_deltas, key=lambda item: item["delta"])
|
||||
|
||||
result: dict[str, Any] = {
|
||||
"baseline": baseline_scores,
|
||||
"candidate": candidate_scores,
|
||||
"deltas": metric_deltas,
|
||||
"missing_predictions": {
|
||||
"baseline": int(baseline.get("metrics", {}).get("missing_predictions", 0)),
|
||||
"candidate": int(candidate.get("metrics", {}).get("missing_predictions", 0)),
|
||||
},
|
||||
"documents": {
|
||||
"shared": len(shared),
|
||||
"improved": sum(item["delta"] > epsilon for item in document_deltas),
|
||||
"regressed": sum(item["delta"] < -epsilon for item in document_deltas),
|
||||
"unchanged": sum(abs(item["delta"]) <= epsilon for item in document_deltas),
|
||||
"largest_improvements": [
|
||||
item for item in improvements if item["delta"] > epsilon
|
||||
][:top],
|
||||
"largest_regressions": [
|
||||
item for item in regressions if item["delta"] < -epsilon
|
||||
][:top],
|
||||
"worst_regression": next(
|
||||
(item for item in regressions if item["delta"] < -epsilon), None
|
||||
),
|
||||
},
|
||||
}
|
||||
if reference is not None:
|
||||
reference_scores = _scores(reference)
|
||||
result["reference"] = reference_scores
|
||||
result["candidate_vs_reference"] = {
|
||||
key: candidate_scores[key] - reference_scores[key]
|
||||
for key in SCORE_KEYS
|
||||
if key in candidate_scores and key in reference_scores
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def evaluate_gates(
|
||||
comparison: dict[str, Any],
|
||||
*,
|
||||
min_overall_delta: float,
|
||||
max_document_regression: float | None,
|
||||
max_missing: int,
|
||||
require_reference_lead: bool,
|
||||
) -> list[str]:
|
||||
"""Return human-readable gate failures; an empty list means pass."""
|
||||
failures: list[str] = []
|
||||
overall_delta = comparison["deltas"].get("overall_mean")
|
||||
if overall_delta is None or overall_delta < min_overall_delta:
|
||||
failures.append(
|
||||
f"overall delta {overall_delta!r} is below {min_overall_delta:+.6f}"
|
||||
)
|
||||
candidate_missing = comparison["missing_predictions"]["candidate"]
|
||||
if candidate_missing > max_missing:
|
||||
failures.append(
|
||||
f"candidate has {candidate_missing} missing predictions (maximum {max_missing})"
|
||||
)
|
||||
if max_document_regression is not None:
|
||||
regression = comparison["documents"].get("worst_regression")
|
||||
if regression is not None and regression["delta"] < -max_document_regression:
|
||||
failures.append(
|
||||
"largest document regression "
|
||||
f"{regression['document_id']}={regression['delta']:+.6f} "
|
||||
f"exceeds {-max_document_regression:+.6f}"
|
||||
)
|
||||
if require_reference_lead:
|
||||
reference_delta = comparison.get("candidate_vs_reference", {}).get("overall_mean")
|
||||
if reference_delta is None:
|
||||
failures.append("reference overall score is unavailable")
|
||||
elif reference_delta < 0.0:
|
||||
failures.append(
|
||||
f"candidate trails reference overall by {reference_delta!r}"
|
||||
)
|
||||
return failures
|
||||
|
||||
|
||||
def _run(command: list[str], *, cwd: Path, env: dict[str, str] | None = None) -> None:
|
||||
print("+", " ".join(command), flush=True)
|
||||
subprocess.run(command, cwd=cwd, env=env, check=True)
|
||||
|
||||
|
||||
def _run_engine(
|
||||
*,
|
||||
bench_dir: Path,
|
||||
python: Path,
|
||||
binary: Path,
|
||||
label: str,
|
||||
scratch_root: Path,
|
||||
) -> dict[str, Any]:
|
||||
env = os.environ.copy()
|
||||
env["PDF_INSPECTOR_BINARY"] = str(binary)
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
if source.exists():
|
||||
if source.is_dir():
|
||||
shutil.rmtree(source)
|
||||
else:
|
||||
source.unlink()
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/pdf_parser.py",
|
||||
"--engine",
|
||||
"pdf-inspector",
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
env=env,
|
||||
)
|
||||
|
||||
if not source.is_dir():
|
||||
raise RuntimeError(f"parser did not produce predictions: {source}")
|
||||
destination = scratch_root / label
|
||||
shutil.copytree(source, destination)
|
||||
_run(
|
||||
[
|
||||
str(python),
|
||||
"src/evaluator.py",
|
||||
"--prediction-root",
|
||||
str(scratch_root),
|
||||
"--engine",
|
||||
label,
|
||||
"--log-level",
|
||||
"WARNING",
|
||||
],
|
||||
cwd=bench_dir,
|
||||
)
|
||||
with (destination / "evaluation.json").open(encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
|
||||
|
||||
def _print_report(comparison: dict[str, Any]) -> None:
|
||||
print("\nMetric baseline candidate delta")
|
||||
print("-------------------- ---------- ---------- ----------")
|
||||
for key in SCORE_KEYS:
|
||||
if key not in comparison["deltas"]:
|
||||
continue
|
||||
print(
|
||||
f"{key:<20} {comparison['baseline'][key]:>10.6f} "
|
||||
f"{comparison['candidate'][key]:>10.6f} "
|
||||
f"{comparison['deltas'][key]:>+10.6f}"
|
||||
)
|
||||
if "reference" in comparison:
|
||||
delta = comparison["candidate_vs_reference"].get("overall_mean")
|
||||
reference = comparison["reference"].get("overall_mean")
|
||||
reference_display = f"{reference:.6f}" if reference is not None else "n/a"
|
||||
delta_display = f"{delta:+.6f}" if delta is not None else "n/a"
|
||||
print(f"\nReference overall: {reference_display}; candidate delta: {delta_display}")
|
||||
|
||||
documents = comparison["documents"]
|
||||
print(
|
||||
"\nDocuments: "
|
||||
f"{documents['improved']} improved, {documents['regressed']} regressed, "
|
||||
f"{documents['unchanged']} unchanged ({documents['shared']} shared)"
|
||||
)
|
||||
for heading, key in (
|
||||
("Largest improvements", "largest_improvements"),
|
||||
("Largest regressions", "largest_regressions"),
|
||||
):
|
||||
print(f"\n{heading}:")
|
||||
rows = documents[key]
|
||||
if not rows:
|
||||
print(" none")
|
||||
for row in rows:
|
||||
print(
|
||||
f" {row['document_id']}: {row['delta']:+.6f} "
|
||||
f"({row['baseline']:.6f} -> {row['candidate']:.6f})"
|
||||
)
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--bench-dir", type=Path, required=True)
|
||||
parser.add_argument("--baseline", type=Path, required=True)
|
||||
parser.add_argument("--candidate", type=Path, required=True)
|
||||
parser.add_argument("--python", type=Path)
|
||||
parser.add_argument("--reference-evaluation", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--top", type=_non_negative_int, default=10)
|
||||
parser.add_argument("--min-overall-delta", type=_finite_float, default=0.0)
|
||||
parser.add_argument("--max-document-regression", type=_non_negative_float)
|
||||
parser.add_argument("--max-missing", type=_non_negative_int, default=0)
|
||||
parser.add_argument("--require-reference-lead", action="store_true")
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
bench_dir = args.bench_dir.resolve()
|
||||
baseline = args.baseline.resolve()
|
||||
candidate = args.candidate.resolve()
|
||||
# Keep the virtualenv launcher path intact. Resolving its symlink would
|
||||
# invoke the underlying system interpreter without the benchmark's site
|
||||
# packages.
|
||||
python = (args.python or bench_dir / ".venv" / "bin" / "python").absolute()
|
||||
for path, description in (
|
||||
(bench_dir / "src" / "pdf_parser.py", "OpenDataLoader parser"),
|
||||
(bench_dir / "src" / "evaluator.py", "OpenDataLoader evaluator"),
|
||||
(baseline, "baseline binary"),
|
||||
(candidate, "candidate binary"),
|
||||
(python, "Python interpreter"),
|
||||
):
|
||||
if not path.exists():
|
||||
raise SystemExit(f"{description} not found: {path}")
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="pdf-inspector-opendataloader-") as temporary:
|
||||
scratch_root = Path(temporary)
|
||||
baseline_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=baseline,
|
||||
label="baseline",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
candidate_evaluation = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=python,
|
||||
binary=candidate,
|
||||
label="candidate",
|
||||
scratch_root=scratch_root,
|
||||
)
|
||||
|
||||
reference = None
|
||||
if args.reference_evaluation is not None:
|
||||
with args.reference_evaluation.resolve().open(encoding="utf-8") as handle:
|
||||
reference = json.load(handle)
|
||||
|
||||
comparison = compare_evaluations(
|
||||
baseline_evaluation,
|
||||
candidate_evaluation,
|
||||
reference,
|
||||
top=args.top,
|
||||
)
|
||||
|
||||
_print_report(comparison)
|
||||
if args.json_output is not None:
|
||||
args.json_output.resolve().write_text(
|
||||
json.dumps(comparison, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=args.min_overall_delta,
|
||||
max_document_regression=args.max_document_regression,
|
||||
max_missing=args.max_missing,
|
||||
require_reference_lead=args.require_reference_lead,
|
||||
)
|
||||
if failures:
|
||||
print("\nBenchmark gate failed:", file=sys.stderr)
|
||||
for failure in failures:
|
||||
print(f" - {failure}", file=sys.stderr)
|
||||
return 1
|
||||
print("\nBenchmark gate passed.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,351 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare pdf-inspector evidence with optional MuPDF structured text.
|
||||
|
||||
This is an experiment and diagnostic tool, not an extraction fallback. It runs
|
||||
MuPDF's deterministic ``stext.json`` backend without OCR and highlights pages
|
||||
where that backend exposes materially different text or layout evidence.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from collections import Counter
|
||||
import json
|
||||
from pathlib import Path
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any, Iterable
|
||||
|
||||
|
||||
TOKEN_PATTERN = re.compile(r"[^\W_]+(?:[\u2019'][^\W_]+)*", re.UNICODE)
|
||||
|
||||
|
||||
def _tokens(texts: Iterable[str]) -> Counter[str]:
|
||||
tokens: Counter[str] = Counter()
|
||||
for text in texts:
|
||||
for token in TOKEN_PATTERN.findall(text.casefold()):
|
||||
# Lone letters are frequently bullets, chart labels, or fragmented
|
||||
# glyphs. Digits remain useful even when they are one character.
|
||||
if len(token) > 1 or token.isdigit():
|
||||
tokens[token] += 1
|
||||
return tokens
|
||||
|
||||
|
||||
def _repeated_x_anchors(xs: Iterable[float], *, tolerance: float = 4.0) -> int:
|
||||
buckets = Counter(round(float(x) / tolerance) for x in xs)
|
||||
return sum(count >= 3 for count in buckets.values())
|
||||
|
||||
|
||||
def local_pages(payload: dict[str, Any]) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize positioned ``pdf2md --items-json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
for item in payload.get("items", []):
|
||||
page_number = int(item["page"])
|
||||
page = pages.setdefault(
|
||||
page_number,
|
||||
{"texts": [], "xs": [], "text_items": 0, "image_items": 0},
|
||||
)
|
||||
if item.get("item_type") == "image":
|
||||
page["image_items"] += 1
|
||||
continue
|
||||
text = str(item.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
page["xs"].append(float(item.get("x", 0.0)))
|
||||
page["text_items"] += 1
|
||||
return pages
|
||||
|
||||
|
||||
def alternate_pages(
|
||||
payload: dict[str, Any] | list[dict[str, Any]],
|
||||
) -> dict[int, dict[str, Any]]:
|
||||
"""Summarize MuPDF ``stext.json`` evidence by page."""
|
||||
pages: dict[int, dict[str, Any]] = {}
|
||||
raw_pages = payload if isinstance(payload, list) else payload.get("pages", [])
|
||||
for index, raw_page in enumerate(raw_pages, start=1):
|
||||
page_number = int(raw_page.get("number", index))
|
||||
page = {
|
||||
"texts": [],
|
||||
"xs": [],
|
||||
"text_blocks": 0,
|
||||
"text_lines": 0,
|
||||
"image_blocks": 0,
|
||||
}
|
||||
for block in raw_page.get("blocks", []):
|
||||
if block.get("type") == "image":
|
||||
page["image_blocks"] += 1
|
||||
continue
|
||||
if block.get("type") != "text":
|
||||
continue
|
||||
page["text_blocks"] += 1
|
||||
for line in block.get("lines", []):
|
||||
text = str(line.get("text", ""))
|
||||
if text.strip():
|
||||
page["texts"].append(text)
|
||||
bbox = line.get("bbox", {})
|
||||
page["xs"].append(float(bbox.get("x", line.get("x", 0.0))))
|
||||
page["text_lines"] += 1
|
||||
pages[page_number] = page
|
||||
return pages
|
||||
|
||||
|
||||
def compare_page(
|
||||
local: dict[str, Any],
|
||||
alternate: dict[str, Any],
|
||||
*,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
"""Compare semantic and coarse layout evidence for one page."""
|
||||
local_tokens = _tokens(local.get("texts", []))
|
||||
alternate_tokens = _tokens(alternate.get("texts", []))
|
||||
shared = local_tokens & alternate_tokens
|
||||
alternate_only = alternate_tokens - local_tokens
|
||||
local_only = local_tokens - alternate_tokens
|
||||
local_total = sum(local_tokens.values())
|
||||
alternate_total = sum(alternate_tokens.values())
|
||||
shared_total = sum(shared.values())
|
||||
alternate_only_total = sum(alternate_only.values())
|
||||
local_only_total = sum(local_only.values())
|
||||
net_token_gain = alternate_total - local_total
|
||||
alternate_only_ratio = alternate_only_total / max(alternate_total, 1)
|
||||
|
||||
local_anchors = _repeated_x_anchors(local.get("xs", []))
|
||||
alternate_anchors = _repeated_x_anchors(alternate.get("xs", []))
|
||||
anchor_gain = alternate_anchors - local_anchors
|
||||
image_gain = int(alternate.get("image_blocks", 0)) - int(
|
||||
local.get("image_items", 0)
|
||||
)
|
||||
|
||||
reasons: list[str] = []
|
||||
if local_total == 0 and alternate_total >= max(5, min_token_gain // 2):
|
||||
reasons.append("local_text_empty")
|
||||
elif (
|
||||
net_token_gain >= min_token_gain
|
||||
and alternate_only_ratio >= min_alternate_only_ratio
|
||||
):
|
||||
reasons.append("alternate_has_more_text")
|
||||
if anchor_gain >= min_anchor_gain:
|
||||
reasons.append("alternate_has_more_alignment_anchors")
|
||||
if image_gain > 0:
|
||||
reasons.append("alternate_has_more_image_blocks")
|
||||
|
||||
if reasons:
|
||||
classification = "investigate_alternate_evidence"
|
||||
elif local_total - alternate_total >= min_token_gain:
|
||||
classification = "local_has_more_text"
|
||||
elif alternate_only_total + local_only_total:
|
||||
classification = "different_segmentation_or_decoding"
|
||||
else:
|
||||
classification = "equivalent_text_evidence"
|
||||
|
||||
return {
|
||||
"classification": classification,
|
||||
"reasons": reasons,
|
||||
"tokens": {
|
||||
"local": local_total,
|
||||
"alternate": alternate_total,
|
||||
"shared": shared_total,
|
||||
"net_alternate_gain": net_token_gain,
|
||||
"alternate_only": alternate_only_total,
|
||||
"local_only": local_only_total,
|
||||
"alternate_only_ratio": alternate_only_ratio,
|
||||
"alternate_only_sample": sorted(alternate_only)[:12],
|
||||
"local_only_sample": sorted(local_only)[:12],
|
||||
},
|
||||
"layout": {
|
||||
"local_text_items": int(local.get("text_items", 0)),
|
||||
"local_image_items": int(local.get("image_items", 0)),
|
||||
"local_repeated_x_anchors": local_anchors,
|
||||
"alternate_text_blocks": int(alternate.get("text_blocks", 0)),
|
||||
"alternate_text_lines": int(alternate.get("text_lines", 0)),
|
||||
"alternate_image_blocks": int(alternate.get("image_blocks", 0)),
|
||||
"alternate_repeated_x_anchors": alternate_anchors,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def compare_documents(
|
||||
local_payload: dict[str, Any],
|
||||
alternate_payload: dict[str, Any] | list[dict[str, Any]],
|
||||
*,
|
||||
min_token_gain: int = 20,
|
||||
min_alternate_only_ratio: float = 0.15,
|
||||
min_anchor_gain: int = 2,
|
||||
) -> dict[str, Any]:
|
||||
"""Return a page-level evidence report for already extracted payloads."""
|
||||
local = local_pages(local_payload)
|
||||
alternate = alternate_pages(alternate_payload)
|
||||
page_numbers = sorted(local.keys() | alternate.keys())
|
||||
pages = []
|
||||
for page_number in page_numbers:
|
||||
result = compare_page(
|
||||
local.get(page_number, {}),
|
||||
alternate.get(page_number, {}),
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
result["page"] = page_number
|
||||
pages.append(result)
|
||||
|
||||
flagged = [
|
||||
page
|
||||
for page in pages
|
||||
if page["classification"] == "investigate_alternate_evidence"
|
||||
]
|
||||
return {
|
||||
"summary": {
|
||||
"pages": len(pages),
|
||||
"flagged_pages": len(flagged),
|
||||
"flagged_page_numbers": [page["page"] for page in flagged],
|
||||
"local_tokens": sum(page["tokens"]["local"] for page in pages),
|
||||
"alternate_tokens": sum(page["tokens"]["alternate"] for page in pages),
|
||||
"alternate_only_tokens": sum(
|
||||
page["tokens"]["alternate_only"] for page in pages
|
||||
),
|
||||
},
|
||||
"pages": pages,
|
||||
}
|
||||
|
||||
|
||||
def _json_command(command: list[str]) -> Any:
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
check=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
)
|
||||
except subprocess.CalledProcessError as error:
|
||||
detail = error.stderr.strip() or error.stdout.strip() or "no diagnostic output"
|
||||
raise RuntimeError(f"command failed: {' '.join(command)}\n{detail}") from error
|
||||
try:
|
||||
return json.loads(completed.stdout)
|
||||
except json.JSONDecodeError as error:
|
||||
raise RuntimeError(
|
||||
f"command did not return JSON: {' '.join(command)}: {error}"
|
||||
) from error
|
||||
|
||||
|
||||
def probe_pdf(
|
||||
pdf: Path,
|
||||
*,
|
||||
pdf2md: Path,
|
||||
mutool: Path,
|
||||
min_token_gain: int,
|
||||
min_alternate_only_ratio: float,
|
||||
min_anchor_gain: int,
|
||||
) -> dict[str, Any]:
|
||||
local_payload = _json_command([str(pdf2md), str(pdf), "--items-json"])
|
||||
# `stext.json` is MuPDF's native structured text output. The OCR formats
|
||||
# are intentionally not used so this remains a deterministic no-model
|
||||
# comparison.
|
||||
alternate_payload = _json_command(
|
||||
[str(mutool), "draw", "-q", "-F", "stext.json", "-o", "-", str(pdf)]
|
||||
)
|
||||
report = compare_documents(
|
||||
local_payload,
|
||||
alternate_payload,
|
||||
min_token_gain=min_token_gain,
|
||||
min_alternate_only_ratio=min_alternate_only_ratio,
|
||||
min_anchor_gain=min_anchor_gain,
|
||||
)
|
||||
report["pdf"] = str(pdf)
|
||||
return report
|
||||
|
||||
|
||||
def _arguments(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("pdf", type=Path, nargs="+")
|
||||
parser.add_argument("--pdf2md", type=Path, default=Path("target/release/pdf2md"))
|
||||
parser.add_argument("--mutool", type=Path)
|
||||
parser.add_argument("--json-output", type=Path)
|
||||
parser.add_argument("--min-token-gain", type=int, default=20)
|
||||
parser.add_argument("--min-alternate-only-ratio", type=float, default=0.15)
|
||||
parser.add_argument("--min-anchor-gain", type=int, default=2)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def _print_report(result: dict[str, Any]) -> None:
|
||||
summary = result["summary"]
|
||||
print(f"\n{result['pdf']}")
|
||||
print(
|
||||
f" {summary['flagged_pages']}/{summary['pages']} pages flagged; "
|
||||
f"tokens local={summary['local_tokens']} alternate={summary['alternate_tokens']} "
|
||||
f"alternate-only={summary['alternate_only_tokens']}"
|
||||
)
|
||||
for page in result["pages"]:
|
||||
if page["classification"] != "investigate_alternate_evidence":
|
||||
continue
|
||||
reasons = ", ".join(page["reasons"])
|
||||
tokens = page["tokens"]
|
||||
print(
|
||||
f" page {page['page']}: {reasons}; "
|
||||
f"net tokens={tokens['net_alternate_gain']:+d}, "
|
||||
f"alternate-only={tokens['alternate_only']}"
|
||||
)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = _arguments(argv)
|
||||
pdf2md = args.pdf2md.absolute()
|
||||
mutool = args.mutool or (Path(found) if (found := shutil.which("mutool")) else None)
|
||||
if not pdf2md.is_file():
|
||||
print(f"error: pdf2md binary not found: {pdf2md}", file=sys.stderr)
|
||||
return 2
|
||||
if mutool is None or not mutool.is_file():
|
||||
print("error: mutool not found; install MuPDF or pass --mutool", file=sys.stderr)
|
||||
return 2
|
||||
if (
|
||||
args.min_token_gain < 0
|
||||
or args.min_anchor_gain < 0
|
||||
or not 0.0 <= args.min_alternate_only_ratio <= 1.0
|
||||
):
|
||||
print("error: thresholds must be non-negative and ratio must be in [0, 1]", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
results = []
|
||||
for pdf in args.pdf:
|
||||
path = pdf.absolute()
|
||||
if not path.is_file():
|
||||
print(f"error: PDF not found: {path}", file=sys.stderr)
|
||||
return 2
|
||||
try:
|
||||
result = probe_pdf(
|
||||
path,
|
||||
pdf2md=pdf2md,
|
||||
mutool=mutool,
|
||||
min_token_gain=args.min_token_gain,
|
||||
min_alternate_only_ratio=args.min_alternate_only_ratio,
|
||||
min_anchor_gain=args.min_anchor_gain,
|
||||
)
|
||||
except RuntimeError as error:
|
||||
print(f"error: {error}", file=sys.stderr)
|
||||
return 1
|
||||
results.append(result)
|
||||
_print_report(result)
|
||||
|
||||
payload = {
|
||||
"schema_version": 1,
|
||||
"experiment": "optional_mupdf_stext_evidence",
|
||||
"ocr": False,
|
||||
"thresholds": {
|
||||
"min_token_gain": args.min_token_gain,
|
||||
"min_alternate_only_ratio": args.min_alternate_only_ratio,
|
||||
"min_anchor_gain": args.min_anchor_gain,
|
||||
},
|
||||
"documents": results,
|
||||
}
|
||||
if args.json_output:
|
||||
args.json_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.json_output.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,203 @@
|
||||
import io
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from contextlib import redirect_stderr, redirect_stdout
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from bench_opendataloader import (
|
||||
_arguments,
|
||||
_print_report,
|
||||
_run_engine,
|
||||
compare_evaluations,
|
||||
evaluate_gates,
|
||||
)
|
||||
|
||||
|
||||
def evaluation(overall, documents, *, missing=0):
|
||||
return {
|
||||
"metrics": {
|
||||
"score": {
|
||||
"overall_mean": overall,
|
||||
"nid_mean": overall + 0.01,
|
||||
},
|
||||
"missing_predictions": missing,
|
||||
},
|
||||
"documents": [
|
||||
{
|
||||
"document_id": document_id,
|
||||
"scores": {"overall": score},
|
||||
}
|
||||
for document_id, score in documents.items()
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class ComparisonTests(unittest.TestCase):
|
||||
def test_reports_metric_and_document_deltas(self):
|
||||
baseline = evaluation(0.80, {"a": 0.8, "b": 0.6, "c": 0.7})
|
||||
candidate = evaluation(0.82, {"a": 0.9, "b": 0.5, "c": 0.7})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, top=1)
|
||||
|
||||
self.assertAlmostEqual(result["deltas"]["overall_mean"], 0.02)
|
||||
self.assertEqual(result["documents"]["improved"], 1)
|
||||
self.assertEqual(result["documents"]["regressed"], 1)
|
||||
self.assertEqual(result["documents"]["unchanged"], 1)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_improvements"][0]["document_id"], "a"
|
||||
)
|
||||
self.assertEqual(
|
||||
result["documents"]["largest_regressions"][0]["document_id"], "b"
|
||||
)
|
||||
|
||||
def test_reference_delta_is_reported(self):
|
||||
baseline = evaluation(0.80, {})
|
||||
candidate = evaluation(0.82, {})
|
||||
reference = evaluation(0.81, {})
|
||||
|
||||
result = compare_evaluations(baseline, candidate, reference)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
result["candidate_vs_reference"]["overall_mean"], 0.01
|
||||
)
|
||||
|
||||
def test_gates_cover_aggregate_document_missing_and_reference(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.79, {"a": 0.7}, missing=1),
|
||||
evaluation(0.81, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 4)
|
||||
|
||||
def test_regression_gate_is_independent_of_report_limit(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {"a": 0.8}),
|
||||
evaluation(0.80, {"a": 0.7}),
|
||||
top=0,
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=0.05,
|
||||
max_missing=0,
|
||||
require_reference_lead=False,
|
||||
)
|
||||
|
||||
self.assertEqual(len(failures), 1)
|
||||
self.assertIn("largest document regression", failures[0])
|
||||
|
||||
def test_report_handles_reference_without_overall_score(self):
|
||||
result = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
{"metrics": {"score": {"nid_mean": 0.81}}},
|
||||
)
|
||||
|
||||
output = io.StringIO()
|
||||
with redirect_stdout(output):
|
||||
_print_report(result)
|
||||
|
||||
self.assertIn("Reference overall: n/a; candidate delta: n/a", output.getvalue())
|
||||
|
||||
def test_reference_gate_reports_missing_score_as_unavailable(self):
|
||||
comparison = compare_evaluations(
|
||||
evaluation(0.80, {}),
|
||||
evaluation(0.82, {}),
|
||||
)
|
||||
|
||||
failures = evaluate_gates(
|
||||
comparison,
|
||||
min_overall_delta=0.0,
|
||||
max_document_regression=None,
|
||||
max_missing=0,
|
||||
require_reference_lead=True,
|
||||
)
|
||||
|
||||
self.assertEqual(failures, ["reference overall score is unavailable"])
|
||||
|
||||
def test_arguments_reject_negative_counts_and_allow_zero_top(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
self.assertEqual(_arguments(required + ["--top", "0"]).top, 0)
|
||||
for option in ("--top", "--max-document-regression", "--max-missing"):
|
||||
with self.subTest(option=option), redirect_stderr(io.StringIO()):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, "-1"])
|
||||
|
||||
def test_arguments_reject_nonfinite_float_thresholds(self):
|
||||
required = [
|
||||
"--bench-dir",
|
||||
".",
|
||||
"--baseline",
|
||||
"baseline",
|
||||
"--candidate",
|
||||
"candidate",
|
||||
]
|
||||
for option in ("--min-overall-delta", "--max-document-regression"):
|
||||
for value in ("nan", "inf", "-inf"):
|
||||
with self.subTest(option=option, value=value), redirect_stderr(
|
||||
io.StringIO()
|
||||
):
|
||||
with self.assertRaises(SystemExit):
|
||||
_arguments(required + [option, value])
|
||||
|
||||
def test_run_engine_clears_stale_predictions_before_parser(self):
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
bench_dir = root / "bench"
|
||||
source = bench_dir / "prediction" / "pdf-inspector"
|
||||
source.mkdir(parents=True)
|
||||
(source / "stale.md").write_text("stale", encoding="utf-8")
|
||||
scratch = root / "scratch"
|
||||
scratch.mkdir()
|
||||
|
||||
def fake_run(command, *, cwd, env=None):
|
||||
if any(part.endswith("pdf_parser.py") for part in command):
|
||||
self.assertFalse(source.exists())
|
||||
(source / "markdown").mkdir(parents=True)
|
||||
(source / "markdown" / "new.md").write_text(
|
||||
"new", encoding="utf-8"
|
||||
)
|
||||
else:
|
||||
destination = scratch / "candidate"
|
||||
(destination / "evaluation.json").write_text(
|
||||
json.dumps(evaluation(0.82, {})), encoding="utf-8"
|
||||
)
|
||||
|
||||
with patch("bench_opendataloader._run", side_effect=fake_run):
|
||||
result = _run_engine(
|
||||
bench_dir=bench_dir,
|
||||
python=Path("python"),
|
||||
binary=Path("pdf2md"),
|
||||
label="candidate",
|
||||
scratch_root=scratch,
|
||||
)
|
||||
|
||||
self.assertEqual(result["metrics"]["score"]["overall_mean"], 0.82)
|
||||
self.assertFalse((source / "stale.md").exists())
|
||||
self.assertFalse((scratch / "candidate" / "stale.md").exists())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,103 @@
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from probe_backend_evidence import compare_documents
|
||||
|
||||
|
||||
def local_payload(items):
|
||||
return {"items": items}
|
||||
|
||||
|
||||
def item(page, text, x=10, item_type="text"):
|
||||
return {"page": page, "text": text, "x": x, "item_type": item_type}
|
||||
|
||||
|
||||
def alternate_payload(pages):
|
||||
return {"pages": pages}
|
||||
|
||||
|
||||
def page(lines, *, images=0):
|
||||
blocks = [
|
||||
{
|
||||
"type": "text",
|
||||
"lines": [
|
||||
{"text": text, "bbox": {"x": x, "y": index * 10, "w": 80, "h": 8}}
|
||||
for index, (text, x) in enumerate(lines)
|
||||
],
|
||||
}
|
||||
]
|
||||
blocks.extend({"type": "image"} for _ in range(images))
|
||||
return {"blocks": blocks}
|
||||
|
||||
|
||||
class EvidenceComparisonTests(unittest.TestCase):
|
||||
def test_accepts_real_top_level_page_array(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = [page([("alpha beta gamma", 10)])]
|
||||
|
||||
result = compare_documents(local, alternate)["pages"][0]
|
||||
|
||||
self.assertEqual(result["tokens"]["alternate"], 3)
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 1)
|
||||
|
||||
def test_flags_material_alternate_text_gain(self):
|
||||
local = local_payload([item(1, "alpha beta")])
|
||||
alternate = alternate_payload(
|
||||
[page([("alpha beta gamma delta epsilon zeta", 10)])]
|
||||
)
|
||||
|
||||
report = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=3,
|
||||
min_alternate_only_ratio=0.2,
|
||||
)
|
||||
|
||||
result = report["pages"][0]
|
||||
self.assertEqual(result["classification"], "investigate_alternate_evidence")
|
||||
self.assertIn("alternate_has_more_text", result["reasons"])
|
||||
self.assertEqual(result["tokens"]["net_alternate_gain"], 4)
|
||||
|
||||
def test_repeated_alignment_and_image_evidence_are_reported(self):
|
||||
local = local_payload([item(1, "one two", 10)])
|
||||
alternate = alternate_payload(
|
||||
[
|
||||
page(
|
||||
[
|
||||
("one two", 10),
|
||||
("row three", 100),
|
||||
("row four", 100),
|
||||
("row five", 100),
|
||||
],
|
||||
images=1,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
result = compare_documents(
|
||||
local,
|
||||
alternate,
|
||||
min_token_gain=99,
|
||||
min_anchor_gain=1,
|
||||
)["pages"][0]
|
||||
|
||||
self.assertIn("alternate_has_more_alignment_anchors", result["reasons"])
|
||||
self.assertIn("alternate_has_more_image_blocks", result["reasons"])
|
||||
self.assertEqual(result["layout"]["alternate_repeated_x_anchors"], 1)
|
||||
|
||||
def test_token_segmentation_difference_does_not_imply_more_evidence(self):
|
||||
local = local_payload([item(1, "Revenue 2025")])
|
||||
alternate = alternate_payload([page([("Revenue 2024", 10)])])
|
||||
|
||||
result = compare_documents(local, alternate, min_token_gain=2)["pages"][0]
|
||||
|
||||
self.assertEqual(result["classification"], "different_segmentation_or_decoding")
|
||||
self.assertEqual(result["reasons"], [])
|
||||
self.assertEqual(result["tokens"]["alternate_only_sample"], ["2024"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,3 @@
|
||||
<svg width="200" height="284" viewBox="0 0 200 284" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M166.862 90.7716C155.812 94.0514 147.483 101.471 141.383 109.53C140.073 111.26 137.343 109.96 137.863 107.841C149.543 59.8136 134.113 19.896 86.0157 0.247269C83.5758 -0.752669 81.0359 1.43719 81.6759 3.99704C103.555 91.8416 11.5294 84.432 23.1588 184.016C23.3588 185.726 21.4389 186.896 20.039 185.896C15.6792 182.766 10.8095 176.236 7.46963 171.647C6.48968 170.297 4.36978 170.677 3.9198 172.287C1.25994 181.906 0 190.965 0 199.965C0 234.963 17.9891 265.771 45.2177 283.63C46.7777 284.65 48.7776 283.19 48.2476 281.4C46.8477 276.7 46.0577 271.74 45.9977 266.611C45.9977 263.461 46.1977 260.241 46.6877 257.241C47.8276 249.702 50.4475 242.522 54.8473 235.983C69.9365 213.334 100.185 191.455 95.3552 161.747C95.0453 159.867 97.2651 158.627 98.6651 159.917C119.974 179.386 124.194 205.575 120.694 229.063C120.394 231.103 122.954 232.193 124.244 230.593C127.504 226.513 131.483 222.933 135.813 220.244C136.893 219.574 138.333 220.084 138.743 221.284C141.153 228.293 144.733 234.873 148.113 241.452C152.152 249.362 154.302 258.391 153.962 267.951C153.792 272.6 153.022 277.1 151.732 281.38C151.182 283.19 153.162 284.7 154.752 283.66C182.001 265.801 200 234.993 200 199.975C200 187.806 197.87 175.876 193.84 164.697C185.391 141.248 163.952 123.64 169.372 93.0815C169.632 91.6216 168.282 90.3517 166.862 90.7716Z" fill="#FA5D19" style="fill:#FA5D19;fill:color(display-p3 0.9816 0.3634 0.0984);fill-opacity:1;"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 1.5 KiB |
@@ -0,0 +1,12 @@
|
||||
<svg width="172" height="40" viewBox="0 0 172 40" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M23.3606 12.8281C21.8137 13.2873 20.6476 14.3261 19.7936 15.4544C19.6102 15.6966 19.228 15.5146 19.3008 15.2178C20.936 8.49401 18.7759 2.90556 12.0422 0.154735C11.7006 0.0147436 11.345 0.321324 11.4346 0.679702C14.4977 12.9779 1.61412 11.9406 3.24224 25.8823C3.27024 26.1217 3.00145 26.2855 2.80546 26.1455C2.19509 25.7073 1.51332 24.7932 1.04575 24.1506C0.908555 23.9616 0.611769 24.0148 0.548773 24.2402C0.176391 25.5869 0 26.8553 0 28.1152C0 33.0149 2.51847 37.328 6.33048 39.8283C6.54887 39.9711 6.82886 39.7667 6.75466 39.5161C6.55867 38.8581 6.44808 38.1638 6.43968 37.4456C6.43968 37.0046 6.46768 36.5539 6.53627 36.1339C6.69587 35.0784 7.06265 34.0732 7.67862 33.1577C9.79111 29.9869 14.0259 26.9239 13.3497 22.7647C13.3063 22.5015 13.6171 22.328 13.8131 22.5085C16.7964 25.2342 17.3871 28.9005 16.8972 32.1889C16.8552 32.4745 17.2135 32.6271 17.3941 32.4031C17.8505 31.832 18.4077 31.3308 19.0138 30.9542C19.165 30.8604 19.3666 30.9318 19.424 31.0998C19.7614 32.0811 20.2626 33.0023 20.7358 33.9234C21.3013 35.0308 21.6023 36.2949 21.5547 37.6332C21.5309 38.2842 21.4231 38.9141 21.2425 39.5133C21.1655 39.7667 21.4427 39.9781 21.6653 39.8325C25.4801 37.3322 28 33.0191 28 28.1166C28 26.4129 27.7018 24.7428 27.1376 23.1777C25.9547 19.8949 22.9533 17.4297 23.712 13.1515C23.7484 12.9471 23.5594 12.7693 23.3606 12.8281Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M41 34.0521V10.9618H55.7586V14.3264H44.7969V21.0226H53.8436V24.2882H44.7969V34.0521H41Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M59.9569 14.7882C58.7352 14.7882 57.7777 13.8976 57.7777 12.6441C57.7777 11.3906 58.7352 10.5 59.9569 10.5C61.1785 10.5 62.136 11.3906 62.136 12.6441C62.136 13.8976 61.1785 14.7882 59.9569 14.7882ZM58.1409 34.0521V17.1632H61.7068V34.0521H58.1409Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M73.5885 17.1632H74.3809V20.4948H72.796C69.6264 20.4948 68.6029 22.9687 68.6029 25.5747V34.0521H65.0371V17.1632H68.2067L68.6029 19.7031C69.4613 18.2847 70.815 17.1632 73.5885 17.1632Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M83.632 34.25C78.3163 34.25 74.9816 30.8194 74.9816 25.6406C74.9816 20.4288 78.3163 16.9653 83.3019 16.9653C88.1884 16.9653 91.457 20.066 91.5561 25.0139C91.5561 25.4427 91.5231 25.9045 91.457 26.3663H78.7125V26.5972C78.8116 29.467 80.6275 31.3472 83.4339 31.3472C85.613 31.3472 87.1979 30.2587 87.6931 28.3785H91.2589C90.6646 31.7101 87.8252 34.25 83.632 34.25ZM78.8446 23.7604H87.8582C87.561 21.2535 85.8112 19.8351 83.3349 19.8351C81.0567 19.8351 79.1087 21.3524 78.8446 23.7604Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M102.033 34.25C96.9151 34.25 93.6465 30.9184 93.6465 25.6406C93.6465 20.4288 97.0142 16.9653 102.132 16.9653C106.49 16.9653 109.197 19.3733 109.891 23.1997H106.16C105.698 21.2205 104.278 20 102.066 20C99.1933 20 97.3113 22.309 97.3113 25.6406C97.3113 28.9392 99.1933 31.2153 102.066 31.2153C104.245 31.2153 105.698 29.9618 106.127 28.0156H109.891C109.23 31.842 106.358 34.25 102.033 34.25Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M121.006 17.1632H121.799V20.4948H120.214C117.044 20.4948 116.021 22.9687 116.021 25.5747V34.0521H112.455V17.1632H115.625L116.021 19.7031C116.879 18.2847 118.233 17.1632 121.006 17.1632Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M130.614 16.9653C135.104 16.9653 137.679 19.1094 137.679 23.1007V34.0521H134.576L134.279 31.6441C133.123 33.1615 131.505 34.25 128.831 34.25C125.133 34.25 122.657 32.4358 122.657 29.3021C122.657 25.8385 125.166 23.8924 129.92 23.8924H134.147V22.8698C134.147 20.9896 132.793 19.8351 130.449 19.8351C128.336 19.8351 126.916 20.8247 126.652 22.309H123.152C123.515 19.0104 126.355 16.9653 130.614 16.9653ZM129.425 31.4792C132.397 31.4792 134.114 29.7309 134.147 27.125V26.5312H129.722C127.51 26.5312 126.289 27.3559 126.289 29.0712C126.289 30.4896 127.477 31.4792 129.425 31.4792Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M144.653 34.0521L139.139 17.1632H142.903L146.766 30.0937L150.629 17.1632H153.897L157.595 30.0937L161.59 17.1632H165.222L159.609 34.0521H155.779L152.214 22.5729L148.516 34.0521H144.653Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
<path d="M166.934 34.0521V10.9618H170.5V34.0521H166.934Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 4.8 KiB |
+1172
File diff suppressed because it is too large
Load Diff
+44
-2
@@ -11,6 +11,37 @@ use std::process;
|
||||
use std::time::Instant;
|
||||
|
||||
/// Escape a string for embedding in a JSON string value.
|
||||
fn format_detector_ocr_reasons(reasons: &std::collections::BTreeMap<u32, Vec<String>>) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|(page, page_reasons)| {
|
||||
let reasons_json = page_reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|entry| {
|
||||
let reasons_json = entry
|
||||
.reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn json_escape(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len() + 16);
|
||||
for ch in s.chars() {
|
||||
@@ -32,6 +63,7 @@ fn json_escape(s: &str) -> String {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
@@ -117,11 +149,13 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"detection_time_ms":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"detection_time_ms":{}}}"#,
|
||||
pdf_type_str(&result.pdf_type),
|
||||
result.page_count,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
@@ -144,6 +178,9 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
println!("Page count: {}", result.page_count);
|
||||
if !result.pages_needing_ocr.is_empty() {
|
||||
println!("Pages needing OCR: {:?}", result.pages_needing_ocr);
|
||||
for entry in &result.ocr_reasons_by_page {
|
||||
println!(" page {}: {}", entry.page, entry.reasons.join(", "));
|
||||
}
|
||||
}
|
||||
println!();
|
||||
if result.layout.is_complex {
|
||||
@@ -184,8 +221,9 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_detector_ocr_reasons(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_sampled":{},"pages_with_text":{},"confidence":{:.2},"title":{},"ocr_recommended":{},"pages_needing_ocr":[{}],"detection_time_ms":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"pages_sampled":{},"pages_with_text":{},"confidence":{:.2},"title":{},"ocr_recommended":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"detection_time_ms":{}}}"#,
|
||||
pdf_type_str(&result.pdf_type),
|
||||
result.page_count,
|
||||
result.pages_sampled,
|
||||
@@ -198,6 +236,7 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
.unwrap_or_else(|| "null".to_string()),
|
||||
result.ocr_recommended,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
elapsed.as_millis()
|
||||
);
|
||||
} else {
|
||||
@@ -232,6 +271,9 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
result.pages_needing_ocr, result.page_count
|
||||
);
|
||||
}
|
||||
for (page, reasons) in &result.ocr_reasons_by_page {
|
||||
println!(" page {}: {}", page, reasons.join(", "));
|
||||
}
|
||||
}
|
||||
if let Some(title) = &result.title {
|
||||
println!("Title: {}", title);
|
||||
|
||||
+149
-3
@@ -1,6 +1,10 @@
|
||||
//! CLI tool for PDF to Markdown conversion
|
||||
|
||||
use pdf_inspector::{process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode};
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
|
||||
PdfType, ProcessMode, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
use std::env;
|
||||
use std::fmt::Write;
|
||||
@@ -31,6 +35,110 @@ fn json_escape(s: &str) -> String {
|
||||
out
|
||||
}
|
||||
|
||||
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|entry| {
|
||||
let reasons_json = entry
|
||||
.reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn item_type_label(item_type: &ItemType) -> &'static str {
|
||||
match item_type {
|
||||
ItemType::Text => "text",
|
||||
ItemType::Image => "image",
|
||||
ItemType::Link(_) => "link",
|
||||
ItemType::FormField => "form_field",
|
||||
}
|
||||
}
|
||||
|
||||
fn format_items_json(items: &[TextItem]) -> String {
|
||||
let underlined_count = items.iter().filter(|item| item.is_underline).count();
|
||||
let items_json = items
|
||||
.iter()
|
||||
.map(|item| {
|
||||
let mcid = item
|
||||
.mcid
|
||||
.map(|value| value.to_string())
|
||||
.unwrap_or_else(|| "null".to_string());
|
||||
let link_url = match &item.item_type {
|
||||
ItemType::Link(url) => format!(r#","url":"{}""#, json_escape(url)),
|
||||
_ => String::new(),
|
||||
};
|
||||
format!(
|
||||
r#"{{"text":"{}","page":{},"x":{:.2},"y":{:.2},"width":{:.2},"height":{:.2},"font":"{}","font_size":{:.2},"is_bold":{},"is_italic":{},"is_underline":{},"is_strikeout":{},"item_type":"{}","mcid":{}{}}}"#,
|
||||
json_escape(&item.text),
|
||||
item.page,
|
||||
item.x,
|
||||
item.y,
|
||||
item.width,
|
||||
item.height,
|
||||
json_escape(&item.font),
|
||||
item.font_size,
|
||||
item.is_bold,
|
||||
item.is_italic,
|
||||
item.is_underline,
|
||||
item.is_strikeout,
|
||||
item_type_label(&item.item_type),
|
||||
mcid,
|
||||
link_url,
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
|
||||
format!(
|
||||
r#"{{"total_items":{},"underlined_count":{},"items":[{}]}}"#,
|
||||
items.len(),
|
||||
underlined_count,
|
||||
items_json
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::format_items_json;
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
#[test]
|
||||
fn items_json_includes_position_and_underline_metadata() {
|
||||
let items = vec![TextItem {
|
||||
text: "A \"quoted\" item".to_string(),
|
||||
x: 12.345,
|
||||
y: 67.891,
|
||||
width: 23.456,
|
||||
height: 9.876,
|
||||
font: "F1".to_string(),
|
||||
font_size: 10.0,
|
||||
page: 2,
|
||||
is_bold: false,
|
||||
is_italic: true,
|
||||
is_underline: true,
|
||||
is_strikeout: true,
|
||||
item_type: ItemType::Text,
|
||||
mcid: Some(7),
|
||||
}];
|
||||
|
||||
let json = format_items_json(&items);
|
||||
|
||||
assert!(json.contains(r#""text":"A \"quoted\" item""#));
|
||||
assert!(json.contains(r#""page":2"#));
|
||||
assert!(json.contains(r#""x":12.35"#));
|
||||
assert!(json.contains(r#""is_underline":true"#));
|
||||
assert!(json.contains(r#""item_type":"text""#));
|
||||
assert!(json.contains(r#""mcid":7"#));
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
fn parse_page_spec(spec: &str) -> Result<HashSet<u32>, String> {
|
||||
let mut pages = HashSet::new();
|
||||
@@ -82,12 +190,14 @@ fn print_layout_info(layout: &LayoutComplexity) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
env_logger::init();
|
||||
let args: Vec<String> = env::args().collect();
|
||||
|
||||
if args.len() < 2 {
|
||||
eprintln!("Usage: {} <pdf_file> [output_file]", args[0]);
|
||||
eprintln!(" {} <pdf_file> --json", args[0]);
|
||||
eprintln!(" {} <pdf_file> --items-json", args[0]);
|
||||
eprintln!(" {} <pdf_file> --raw", args[0]);
|
||||
eprintln!();
|
||||
eprintln!("Converts PDF to Markdown with smart type detection.");
|
||||
@@ -95,9 +205,14 @@ fn main() {
|
||||
eprintln!();
|
||||
eprintln!("Options:");
|
||||
eprintln!(" --json Output result as JSON");
|
||||
eprintln!(" --items-json Output positioned TextItem JSON");
|
||||
eprintln!(" --raw Output only markdown (no headers)");
|
||||
eprintln!(
|
||||
" --compact Collapse token-heavy source formatting such as dot leaders"
|
||||
);
|
||||
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
|
||||
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
||||
eprintln!(" --password PW Password for an encrypted PDF");
|
||||
eprintln!(" --detect-only Only detect PDF type (no extraction)");
|
||||
eprintln!(" --analyze Detect + extract + layout analysis (no markdown)");
|
||||
process::exit(1);
|
||||
@@ -105,11 +220,23 @@ fn main() {
|
||||
|
||||
let pdf_path = &args[1];
|
||||
let json_output = args.iter().any(|a| a == "--json");
|
||||
let items_json_output = args.iter().any(|a| a == "--items-json");
|
||||
let raw_output = args.iter().any(|a| a == "--raw");
|
||||
let compact_output = args.iter().any(|a| a == "--compact");
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
let analyze = args.iter().any(|a| a == "--analyze");
|
||||
|
||||
// Parse --password value
|
||||
let password = args.iter().position(|a| a == "--password").map(|i| {
|
||||
args.get(i + 1)
|
||||
.unwrap_or_else(|| {
|
||||
eprintln!("Error: --password requires a value");
|
||||
process::exit(1);
|
||||
})
|
||||
.clone()
|
||||
});
|
||||
|
||||
// Parse --select-pages value
|
||||
let page_filter = args
|
||||
.iter()
|
||||
@@ -129,6 +256,17 @@ fn main() {
|
||||
})
|
||||
});
|
||||
|
||||
if items_json_output {
|
||||
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
|
||||
Ok(items) => println!("{}", format_items_json(&items)),
|
||||
Err(e) => {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
@@ -143,10 +281,14 @@ fn main() {
|
||||
};
|
||||
|
||||
let mut options = PdfOptions::new().mode(process_mode);
|
||||
if compact_output {
|
||||
options.markdown.profile = pdf_inspector::MarkdownProfile::Compact;
|
||||
}
|
||||
options.markdown.include_page_numbers = page_numbers;
|
||||
if let Some(pages) = page_filter {
|
||||
options.page_filter = Some(pages);
|
||||
}
|
||||
options.password = password;
|
||||
|
||||
match process_pdf_with_options(pdf_path, options) {
|
||||
Ok(result) => {
|
||||
@@ -177,12 +319,14 @@ fn main() {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
|
||||
pdf_type_str,
|
||||
result.page_count,
|
||||
result.processing_time_ms,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
@@ -223,8 +367,9 @@ fn main() {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
|
||||
match result.pdf_type {
|
||||
PdfType::TextBased => "text_based",
|
||||
PdfType::Scanned => "scanned",
|
||||
@@ -236,6 +381,7 @@ fn main() {
|
||||
result.processing_time_ms,
|
||||
result.markdown.as_ref().map(|m| m.len()).unwrap_or(0),
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
|
||||
+111
-2
@@ -60,6 +60,10 @@ pub struct PdfTypeResult {
|
||||
/// 1-indexed page numbers that need OCR (image-only or insufficient text).
|
||||
/// Empty for TextBased. All pages for Scanned/ImageBased. Specific pages for Mixed.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Per-page explanation for `pages_needing_ocr`: 1-indexed page → reason
|
||||
/// codes (`scanned`, `no_text`, `vector_text`, `suspected_garbled_text`).
|
||||
/// Only contains pages that need OCR.
|
||||
pub ocr_reasons_by_page: std::collections::BTreeMap<u32, Vec<String>>,
|
||||
}
|
||||
|
||||
/// Configuration for PDF type detection
|
||||
@@ -382,7 +386,12 @@ pub(crate) fn detect_from_document(
|
||||
let analysis = if let Some(cached) = analysis_cache.get(&page_num) {
|
||||
cached.clone()
|
||||
} else if let Some(&page_id) = pages.get(&page_num) {
|
||||
analyze_page_content(doc, page_id)
|
||||
// Cache the fresh analysis so the reason-classification pass
|
||||
// below sees the real signals (vector_text, etc.) instead of
|
||||
// defaulting to "scanned".
|
||||
let a = analyze_page_content(doc, page_id);
|
||||
analysis_cache.insert(page_num, a.clone());
|
||||
a
|
||||
} else {
|
||||
continue;
|
||||
};
|
||||
@@ -429,6 +438,9 @@ pub(crate) fn detect_from_document(
|
||||
let analysis = analyze_page_content(doc, page_id);
|
||||
if analysis.has_identity_h_no_tounicode || analysis.has_only_type3_fonts {
|
||||
pages_needing_ocr.push(page_num);
|
||||
// Cache so the reason pass reports suspected_garbled_text
|
||||
// rather than defaulting to "scanned".
|
||||
analysis_cache.insert(page_num, analysis);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -436,6 +448,19 @@ pub(crate) fn detect_from_document(
|
||||
pages_needing_ocr.sort();
|
||||
pages_needing_ocr.dedup();
|
||||
|
||||
// Explain each OCR-flagged page. Pages we analyzed get a signal-derived
|
||||
// reason; pages flagged only by whole-document classification (unsampled
|
||||
// pages of a Scanned/ImageBased doc) default to `scanned`.
|
||||
let mut ocr_reasons_by_page: std::collections::BTreeMap<u32, Vec<String>> =
|
||||
std::collections::BTreeMap::new();
|
||||
for &page_num in &pages_needing_ocr {
|
||||
let reasons = match analysis_cache.get(&page_num) {
|
||||
Some(analysis) => page_ocr_reasons(analysis),
|
||||
None => vec![crate::OCR_REASON_SCANNED],
|
||||
};
|
||||
ocr_reasons_by_page.insert(page_num, reasons.into_iter().map(String::from).collect());
|
||||
}
|
||||
|
||||
// Try to get title from metadata
|
||||
let title = get_document_title(doc);
|
||||
|
||||
@@ -448,6 +473,7 @@ pub(crate) fn detect_from_document(
|
||||
title,
|
||||
ocr_recommended,
|
||||
pages_needing_ocr,
|
||||
ocr_reasons_by_page,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -487,7 +513,7 @@ fn distribute_pages(n: u32, total: u32) -> Vec<u32> {
|
||||
}
|
||||
|
||||
/// Page content analysis result
|
||||
#[derive(Clone)]
|
||||
#[derive(Clone, Default)]
|
||||
struct PageAnalysis {
|
||||
text_operator_count: u32,
|
||||
has_images: bool,
|
||||
@@ -523,6 +549,31 @@ struct PageAnalysis {
|
||||
has_decodable_text_fonts: bool,
|
||||
}
|
||||
|
||||
/// Explain *why* a page needs OCR, from its content analysis. Priority:
|
||||
/// undecodable fonts (`suspected_garbled_text`) and vector-outlined text
|
||||
/// (`vector_text`) come first because they persist even when a text layer is
|
||||
/// present; otherwise a page with no extractable text is `scanned` when an
|
||||
/// image backs it or `no_text` when nothing does.
|
||||
fn page_ocr_reasons(a: &PageAnalysis) -> Vec<&'static str> {
|
||||
let mut reasons = Vec::new();
|
||||
if a.has_identity_h_no_tounicode || a.has_only_type3_fonts {
|
||||
reasons.push(crate::OCR_REASON_SUSPECTED_GARBLED_TEXT);
|
||||
}
|
||||
if a.has_vector_text {
|
||||
reasons.push(crate::OCR_REASON_VECTOR_TEXT);
|
||||
}
|
||||
if reasons.is_empty() {
|
||||
let has_extractable_text = a.text_operator_count > 0 && a.unique_text_chars > 0;
|
||||
if !has_extractable_text && !a.has_images && !a.has_template_image {
|
||||
reasons.push(crate::OCR_REASON_NO_TEXT);
|
||||
} else {
|
||||
// Image-backed with no usable text, or too little text to trust.
|
||||
reasons.push(crate::OCR_REASON_SCANNED);
|
||||
}
|
||||
}
|
||||
reasons
|
||||
}
|
||||
|
||||
/// Extracted font information from a Resource dictionary entry.
|
||||
/// Stores the properties needed for decodability/identity-h checks
|
||||
/// without holding a reference to the document.
|
||||
@@ -1809,6 +1860,64 @@ fn get_document_title(doc: &Document) -> Option<String> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn page_ocr_reasons_classify() {
|
||||
// Scanned: no text, full-page image.
|
||||
let scanned = PageAnalysis {
|
||||
has_template_image: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(page_ocr_reasons(&scanned), vec![crate::OCR_REASON_SCANNED]);
|
||||
|
||||
// Image-only page (no template flag, but has an image).
|
||||
let image_only = PageAnalysis {
|
||||
has_images: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&image_only),
|
||||
vec![crate::OCR_REASON_SCANNED]
|
||||
);
|
||||
|
||||
// No text, no image → no_text.
|
||||
let blank = PageAnalysis::default();
|
||||
assert_eq!(page_ocr_reasons(&blank), vec![crate::OCR_REASON_NO_TEXT]);
|
||||
|
||||
// Vector-outlined text.
|
||||
let vector = PageAnalysis {
|
||||
has_vector_text: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&vector),
|
||||
vec![crate::OCR_REASON_VECTOR_TEXT]
|
||||
);
|
||||
|
||||
// Undecodable fonts → garbled, and it wins over the fall-through.
|
||||
let garbled = PageAnalysis {
|
||||
has_identity_h_no_tounicode: true,
|
||||
has_images: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&garbled),
|
||||
vec![crate::OCR_REASON_SUSPECTED_GARBLED_TEXT]
|
||||
);
|
||||
|
||||
// A page with real extractable text and an image is not flagged here
|
||||
// as scanned/no_text (only reached for pages already needing OCR).
|
||||
let text_with_image = PageAnalysis {
|
||||
text_operator_count: 40,
|
||||
unique_text_chars: 120,
|
||||
has_images: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(
|
||||
page_ocr_reasons(&text_with_image),
|
||||
vec![crate::OCR_REASON_SCANNED]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scan_content_operators() {
|
||||
let mut uchars = HashSet::new();
|
||||
|
||||
+592
-38
@@ -14,11 +14,13 @@ use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, descriptor_style_flags,
|
||||
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
FontStyleCache,
|
||||
};
|
||||
use super::underline::UnderlineLine;
|
||||
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
|
||||
use super::{get_number, multiply_matrices};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
/// Strip PDF comments (% to end of line) from content stream bytes.
|
||||
///
|
||||
@@ -74,6 +76,56 @@ fn strip_pdf_comments(data: &[u8]) -> Vec<u8> {
|
||||
result
|
||||
}
|
||||
|
||||
fn transform_path_point(x: f32, y: f32, ctm: &[f32; 6]) -> (f32, f32) {
|
||||
(
|
||||
x * ctm[0] + y * ctm[2] + ctm[4],
|
||||
x * ctm[1] + y * ctm[3] + ctm[5],
|
||||
)
|
||||
}
|
||||
|
||||
fn transformed_stroke_width(
|
||||
line_width: f32,
|
||||
ctm: &[f32; 6],
|
||||
x1: f32,
|
||||
y1: f32,
|
||||
x2: f32,
|
||||
y2: f32,
|
||||
) -> f32 {
|
||||
let user_width = line_width.abs();
|
||||
let dx = x2 - x1;
|
||||
let dy = y2 - y1;
|
||||
let len = (dx * dx + dy * dy).sqrt();
|
||||
if len <= f32::EPSILON {
|
||||
return user_width;
|
||||
}
|
||||
|
||||
// PDF stroke width scales perpendicular to the path direction.
|
||||
let nx = -dy / len;
|
||||
let ny = dx / len;
|
||||
let ndx = nx * ctm[0] + ny * ctm[2];
|
||||
let ndy = nx * ctm[1] + ny * ctm[3];
|
||||
user_width * (ndx * ndx + ndy * ndy).sqrt()
|
||||
}
|
||||
|
||||
/// Text rise (Ts) displaces the glyph origin by (0, rise) in unscaled text
|
||||
/// space — per the rendering-matrix definition it sits left of Tm, so the
|
||||
/// offset maps through the text matrix's y column. Rise never contributes
|
||||
/// to the advance, so callers apply it only to the rendering position and
|
||||
/// keep advancing the unshifted text matrix.
|
||||
fn rise_adjusted(tm: &[f32; 6], rise: f32) -> [f32; 6] {
|
||||
if rise == 0.0 {
|
||||
return *tm;
|
||||
}
|
||||
[
|
||||
tm[0],
|
||||
tm[1],
|
||||
tm[2],
|
||||
tm[3],
|
||||
tm[4] + rise * tm[2],
|
||||
tm[5] + rise * tm[3],
|
||||
]
|
||||
}
|
||||
|
||||
/// Returns `(page_extraction, has_gid_fonts)` where `has_gid_fonts` indicates
|
||||
/// the page uses fonts with unresolvable gid-encoded glyphs.
|
||||
pub(crate) fn extract_page_text_items(
|
||||
@@ -82,6 +134,7 @@ pub(crate) fn extract_page_text_items(
|
||||
page_num: u32,
|
||||
font_cmaps: &FontCMaps,
|
||||
include_invisible: bool,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> Result<(PageExtraction, bool, bool), PdfError> {
|
||||
use lopdf::content::Content;
|
||||
|
||||
@@ -89,6 +142,7 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut rects: Vec<PdfRect> = Vec::new();
|
||||
let mut clip_rects: Vec<PdfRect> = Vec::new();
|
||||
let mut lines: Vec<PdfLine> = Vec::new();
|
||||
let mut underline_lines: Vec<UnderlineLine> = Vec::new();
|
||||
|
||||
// Path construction state for m/l/h → S/s line extraction
|
||||
let mut path_subpath_start: Option<(f32, f32)> = None;
|
||||
@@ -97,12 +151,18 @@ pub(crate) fn extract_page_text_items(
|
||||
// Completed subpaths (each a vec of line segments) for f/f* rect extraction
|
||||
let mut pending_subpaths: Vec<Vec<(f32, f32, f32, f32)>> = Vec::new();
|
||||
let mut fill_rects: Vec<PdfRect> = Vec::new();
|
||||
// `re` rects awaiting a paint operator. Underline detection must only
|
||||
// see painted rects: a `re W n` clip path or `re n` no-op draws nothing
|
||||
// on the page, so treating every `re` as ink would underline text that
|
||||
// merely sits near an invisible clip boundary.
|
||||
let mut pending_re_rects: Vec<PdfRect> = Vec::new();
|
||||
let mut painted_rects: Vec<PdfRect> = Vec::new();
|
||||
|
||||
// Get fonts for encoding
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
|
||||
// Build font encoding maps from Differences arrays
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts);
|
||||
let (font_encodings, has_gid_fonts) = build_font_encodings(doc, &fonts, font_cmaps);
|
||||
|
||||
// Build font width info for accurate text positioning
|
||||
let font_widths = build_font_widths(doc, &fonts);
|
||||
@@ -114,6 +174,8 @@ pub(crate) fn extract_page_text_items(
|
||||
std::collections::HashMap::new();
|
||||
let mut inline_cmaps: std::collections::HashMap<String, crate::tounicode::CMapEntry> =
|
||||
std::collections::HashMap::new();
|
||||
let mut font_style_flags: std::collections::HashMap<String, (bool, bool)> =
|
||||
std::collections::HashMap::new();
|
||||
for (font_name, font_dict) in &fonts {
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
if let Ok(base_font) = font_dict.get(b"BaseFont") {
|
||||
@@ -122,6 +184,12 @@ pub(crate) fn extract_page_text_items(
|
||||
font_base_names.insert(resource_name.clone(), base_name);
|
||||
}
|
||||
}
|
||||
// Descriptor style flags rescue subset fonts whose BaseFont names
|
||||
// are opaque tags the name heuristics can't read.
|
||||
let style = descriptor_style_flags(doc, font_dict, style_cache);
|
||||
if style != (false, false) {
|
||||
font_style_flags.insert(resource_name.clone(), style);
|
||||
}
|
||||
// Track ToUnicode object reference, with FontFile2 fallback for Identity-H/V.
|
||||
// Also handle inline ToUnicode streams.
|
||||
match font_dict.get(b"ToUnicode") {
|
||||
@@ -188,7 +256,20 @@ pub(crate) fn extract_page_text_items(
|
||||
// Graphics state tracking
|
||||
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
|
||||
let mut text_rendering_mode: i32 = 0; // 0=fill, 1=stroke, 2=fill+stroke, 3=invisible
|
||||
let mut gstate_stack: Vec<([f32; 6], i32, f32, f32)> = Vec::new();
|
||||
let mut line_width: f32 = 1.0;
|
||||
#[derive(Clone)]
|
||||
struct SavedGraphicsState {
|
||||
ctm: [f32; 6],
|
||||
text_rendering_mode: i32,
|
||||
line_width: f32,
|
||||
char_spacing: f32,
|
||||
word_spacing: f32,
|
||||
text_rise: f32,
|
||||
text_leading: f32,
|
||||
current_font: String,
|
||||
current_font_size: f32,
|
||||
}
|
||||
let mut gstate_stack: Vec<SavedGraphicsState> = Vec::new();
|
||||
|
||||
// Text state tracking
|
||||
let mut current_font = String::new();
|
||||
@@ -196,6 +277,7 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut text_leading: f32 = 0.0; // TL parameter (in text-space units)
|
||||
let mut char_spacing: f32 = 0.0; // Tc parameter (extra spacing per character, unscaled)
|
||||
let mut word_spacing: f32 = 0.0; // Tw parameter (extra spacing per space char, unscaled)
|
||||
let mut text_rise: f32 = 0.0; // Ts parameter (baseline shift for super/subscripts, unscaled)
|
||||
let mut text_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
let mut line_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
let mut in_text_block = false;
|
||||
@@ -217,6 +299,10 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut suppress_glyph_extraction = false;
|
||||
let mut actual_text_start_tm: Option<[f32; 6]> = None; // text matrix at BDC entry
|
||||
let mut actual_text_glyph_tm: Option<[f32; 6]> = None; // text matrix at first glyph inside BDC
|
||||
// Text rise in effect at each captured matrix — the item must render at
|
||||
// the rise of its GLYPHS, not whatever rise is set by EMC time.
|
||||
let mut actual_text_start_rise: f32 = 0.0;
|
||||
let mut actual_text_glyph_rise: Option<f32> = None;
|
||||
/// Get the innermost MCID from the marked content stack.
|
||||
fn current_mcid(stack: &[MarkedContentEntry]) -> Option<i64> {
|
||||
stack.iter().rev().find_map(|e| e.mcid)
|
||||
@@ -227,15 +313,30 @@ pub(crate) fn extract_page_text_items(
|
||||
match op.operator.as_str() {
|
||||
"q" => {
|
||||
// Save graphics state
|
||||
gstate_stack.push((ctm, text_rendering_mode, char_spacing, word_spacing));
|
||||
gstate_stack.push(SavedGraphicsState {
|
||||
ctm,
|
||||
text_rendering_mode,
|
||||
line_width,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
text_rise,
|
||||
text_leading,
|
||||
current_font: current_font.clone(),
|
||||
current_font_size,
|
||||
});
|
||||
}
|
||||
"Q" => {
|
||||
// Restore graphics state
|
||||
if let Some((saved_ctm, saved_tr, saved_tc, saved_tw)) = gstate_stack.pop() {
|
||||
ctm = saved_ctm;
|
||||
text_rendering_mode = saved_tr;
|
||||
char_spacing = saved_tc;
|
||||
word_spacing = saved_tw;
|
||||
if let Some(saved) = gstate_stack.pop() {
|
||||
ctm = saved.ctm;
|
||||
text_rendering_mode = saved.text_rendering_mode;
|
||||
line_width = saved.line_width;
|
||||
char_spacing = saved.char_spacing;
|
||||
word_spacing = saved.word_spacing;
|
||||
text_rise = saved.text_rise;
|
||||
text_leading = saved.text_leading;
|
||||
current_font = saved.current_font;
|
||||
current_font_size = saved.current_font_size;
|
||||
}
|
||||
}
|
||||
"cm" => {
|
||||
@@ -252,6 +353,11 @@ pub(crate) fn extract_page_text_items(
|
||||
ctm = multiply_matrices(&new_matrix, &ctm);
|
||||
}
|
||||
}
|
||||
"w" => {
|
||||
if let Some(width) = op.operands.first().and_then(get_number) {
|
||||
line_width = width;
|
||||
}
|
||||
}
|
||||
"BT" => {
|
||||
// Begin text block
|
||||
in_text_block = true;
|
||||
@@ -300,6 +406,12 @@ pub(crate) fn extract_page_text_items(
|
||||
word_spacing = tw;
|
||||
}
|
||||
}
|
||||
"Ts" => {
|
||||
// Set text rise (baseline shift for superscripts/subscripts)
|
||||
if let Some(ts) = op.operands.first().and_then(get_number) {
|
||||
text_rise = ts;
|
||||
}
|
||||
}
|
||||
"Td" | "TD" => {
|
||||
// Move text position: TLM = T(tx,ty) × TLM; Tm = TLM
|
||||
// tx,ty are in text space — must be scaled by the text line matrix
|
||||
@@ -358,6 +470,7 @@ pub(crate) fn extract_page_text_items(
|
||||
if suppress_glyph_extraction {
|
||||
if actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
actual_text_glyph_rise = Some(text_rise);
|
||||
}
|
||||
if let Some(w_ts) = w_ts_opt {
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
@@ -387,7 +500,8 @@ pub(crate) fn extract_page_text_items(
|
||||
&mut cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
@@ -409,6 +523,10 @@ pub(crate) fn extract_page_text_items(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
@@ -418,8 +536,10 @@ pub(crate) fn extract_page_text_items(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
@@ -437,6 +557,7 @@ pub(crate) fn extract_page_text_items(
|
||||
// Capture first-glyph position for ActualText
|
||||
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
actual_text_glyph_rise = Some(text_rise);
|
||||
}
|
||||
|
||||
// Compute space threshold based on font metrics when available
|
||||
@@ -557,6 +678,10 @@ pub(crate) fn extract_page_text_items(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
|
||||
for (text, start_w, end_w) in &sub_items {
|
||||
let offset_tm = [
|
||||
@@ -567,7 +692,8 @@ pub(crate) fn extract_page_text_items(
|
||||
text_matrix[4] + start_w * text_matrix[0],
|
||||
text_matrix[5] + start_w * text_matrix[1],
|
||||
];
|
||||
let combined = multiply_matrices(&offset_tm, &ctm);
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&offset_tm, text_rise), &ctm);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = if font_info.is_some() {
|
||||
((end_w - start_w) * scale_x).abs()
|
||||
@@ -583,8 +709,10 @@ pub(crate) fn extract_page_text_items(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
@@ -608,6 +736,26 @@ pub(crate) fn extract_page_text_items(
|
||||
line_matrix[4] += (-tl) * line_matrix[2];
|
||||
line_matrix[5] += (-tl) * line_matrix[3];
|
||||
text_matrix = line_matrix;
|
||||
// Capture first-glyph position for ActualText AFTER the
|
||||
// line move — the BDC-entry matrix is on the previous line.
|
||||
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
actual_text_glyph_rise = Some(text_rise);
|
||||
}
|
||||
// Advance width, as for Tj — without it the item stays
|
||||
// zero-width and geometric underline/strikeout detection
|
||||
// rejects it (`is_underline_candidate` needs width > 0).
|
||||
let w_ts_opt = font_widths.get(¤t_font).and_then(|fi| {
|
||||
op.operands.first().and_then(get_operand_bytes).map(|raw| {
|
||||
compute_string_width_ts(
|
||||
raw,
|
||||
fi,
|
||||
current_font_size,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
)
|
||||
})
|
||||
});
|
||||
if !((text_rendering_mode == 3 && !include_invisible)
|
||||
|| suppress_glyph_extraction
|
||||
|| op.operands.is_empty())
|
||||
@@ -625,7 +773,8 @@ pub(crate) fn extract_page_text_items(
|
||||
&font_widths,
|
||||
) {
|
||||
if !text.trim().is_empty() {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
} else {
|
||||
@@ -633,27 +782,45 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = w_ts_opt
|
||||
.map(|w_ts| {
|
||||
(w_ts * (text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2]))
|
||||
.abs()
|
||||
})
|
||||
.unwrap_or(0.0);
|
||||
let base_font = font_base_names
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
y,
|
||||
width: 0.0,
|
||||
width,
|
||||
height: rendered_size,
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
// Advance regardless of visibility so later show-text
|
||||
// operators on the same line stay positioned (as for Tj).
|
||||
if let Some(w_ts) = w_ts_opt {
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
text_matrix[5] += w_ts * text_matrix[1];
|
||||
}
|
||||
}
|
||||
"Do" => {
|
||||
// XObject invocation - could be an image or form
|
||||
@@ -664,7 +831,31 @@ pub(crate) fn extract_page_text_items(
|
||||
if let Some(xobj_type) = xobjects.get(&xobj_name) {
|
||||
match xobj_type {
|
||||
XObjectType::Image => {
|
||||
// Skip images — text extraction only
|
||||
// Emit a positional placeholder for the image
|
||||
// so downstream consumers (layout-aware
|
||||
// pipelines, figure-OCR routers) can locate
|
||||
// raster figures without parsing the PDF
|
||||
// again. The text field carries the
|
||||
// XObject resource name in the legacy
|
||||
// `[Image: Im0]` format that the markdown
|
||||
// emitter already recognizes.
|
||||
let (x, y, width, height) = image_bbox_from_ctm(&ctm);
|
||||
items.push(TextItem {
|
||||
text: format!("[Image: {}]", xobj_name),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
font: String::new(),
|
||||
font_size: 0.0,
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Image,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
}
|
||||
XObjectType::Form(form_id) => {
|
||||
// Extract text from Form XObject
|
||||
@@ -675,6 +866,7 @@ pub(crate) fn extract_page_text_items(
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
&mut cmap_decisions,
|
||||
style_cache,
|
||||
);
|
||||
items.extend(form_items);
|
||||
}
|
||||
@@ -715,7 +907,9 @@ pub(crate) fn extract_page_text_items(
|
||||
if actual_text.is_some() {
|
||||
suppress_glyph_extraction = true;
|
||||
actual_text_start_tm = Some(text_matrix);
|
||||
actual_text_start_rise = text_rise;
|
||||
actual_text_glyph_tm = None; // reset — will be captured at first Tj/TJ
|
||||
actual_text_glyph_rise = None;
|
||||
}
|
||||
marked_content_stack.push(MarkedContentEntry { actual_text, mcid });
|
||||
}
|
||||
@@ -728,9 +922,11 @@ pub(crate) fn extract_page_text_items(
|
||||
// Tj may have moved the text position to the correct line —
|
||||
// the BDC-entry position can be on the previous line.
|
||||
let glyph_tm = actual_text_glyph_tm.take();
|
||||
let glyph_rise = actual_text_glyph_rise.take();
|
||||
let entry_tm = actual_text_start_tm.take();
|
||||
if let Some(start_tm) = glyph_tm.or(entry_tm) {
|
||||
let combined = multiply_matrices(&start_tm, &ctm);
|
||||
let rise = glyph_rise.unwrap_or(actual_text_start_rise);
|
||||
let combined = multiply_matrices(&rise_adjusted(&start_tm, rise), &ctm);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
} else {
|
||||
@@ -747,6 +943,10 @@ pub(crate) fn extract_page_text_items(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&at),
|
||||
x,
|
||||
@@ -756,8 +956,10 @@ pub(crate) fn extract_page_text_items(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: entry
|
||||
.mcid
|
||||
@@ -782,13 +984,19 @@ pub(crate) fn extract_page_text_items(
|
||||
let y_dev = rx * ctm[1] + ry * ctm[3] + ctm[5];
|
||||
let w_dev = rw * ctm[0];
|
||||
let h_dev = rh * ctm[3];
|
||||
rects.push(PdfRect {
|
||||
let rect = PdfRect {
|
||||
x: x_dev,
|
||||
y: y_dev,
|
||||
width: w_dev,
|
||||
height: h_dev,
|
||||
page: page_num,
|
||||
});
|
||||
};
|
||||
// Underline detection must only see rects that are
|
||||
// actually painted — a `re` used purely as a clip path
|
||||
// (`re W n`) or discarded (`re n`) draws nothing. Hold
|
||||
// the rect as pending until a paint operator confirms it.
|
||||
pending_re_rects.push(rect.clone());
|
||||
rects.push(rect);
|
||||
}
|
||||
}
|
||||
// ── Path construction operators ──────────────────────
|
||||
@@ -838,10 +1046,8 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
for (x1, y1, x2, y2) in pending_lines.drain(..) {
|
||||
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
|
||||
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
|
||||
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
|
||||
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
|
||||
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
|
||||
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
|
||||
lines.push(PdfLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
@@ -849,7 +1055,16 @@ pub(crate) fn extract_page_text_items(
|
||||
y2: y2d,
|
||||
page: page_num,
|
||||
});
|
||||
underline_lines.push(UnderlineLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
x2: x2d,
|
||||
y2: y2d,
|
||||
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
|
||||
page: page_num,
|
||||
});
|
||||
}
|
||||
painted_rects.append(&mut pending_re_rects);
|
||||
pending_subpaths.clear();
|
||||
path_subpath_start = None;
|
||||
path_current = None;
|
||||
@@ -865,10 +1080,8 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
for (x1, y1, x2, y2) in pending_lines.drain(..) {
|
||||
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
|
||||
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
|
||||
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
|
||||
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
|
||||
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
|
||||
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
|
||||
lines.push(PdfLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
@@ -876,7 +1089,16 @@ pub(crate) fn extract_page_text_items(
|
||||
y2: y2d,
|
||||
page: page_num,
|
||||
});
|
||||
underline_lines.push(UnderlineLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
x2: x2d,
|
||||
y2: y2d,
|
||||
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
|
||||
page: page_num,
|
||||
});
|
||||
}
|
||||
painted_rects.append(&mut pending_re_rects);
|
||||
pending_subpaths.clear();
|
||||
path_subpath_start = None;
|
||||
path_current = None;
|
||||
@@ -934,6 +1156,7 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
}
|
||||
painted_rects.append(&mut pending_re_rects);
|
||||
pending_lines.clear();
|
||||
path_subpath_start = None;
|
||||
path_current = None;
|
||||
@@ -999,7 +1222,10 @@ pub(crate) fn extract_page_text_items(
|
||||
// Do NOT clear pending_lines — the following `n` does that
|
||||
}
|
||||
"n" => {
|
||||
// end path (no-op): discard
|
||||
// end path (no-op): discard — including any `re` rects that
|
||||
// were only ever part of a clip path (`re W n`), which draw
|
||||
// no ink and must not feed underline detection.
|
||||
pending_re_rects.clear();
|
||||
pending_lines.clear();
|
||||
pending_subpaths.clear();
|
||||
path_subpath_start = None;
|
||||
@@ -1009,6 +1235,12 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
|
||||
// Underline detection reads only painted ink: `re` rects confirmed by
|
||||
// a paint operator plus filled-subpath rects — never clip-only rects,
|
||||
// which draw nothing.
|
||||
let mut underline_rects = painted_rects;
|
||||
underline_rects.extend(fill_rects.iter().cloned());
|
||||
|
||||
// Only use clip/fill rects when no `re` rects exist on this page.
|
||||
// Clip rects take priority over fill rects, but first we deduplicate
|
||||
// them: some PDFs wrap every text block in a full-page W* clip path,
|
||||
@@ -1038,8 +1270,17 @@ pub(crate) fn extract_page_text_items(
|
||||
// Some PDFs embed landscape content in portrait pages using a rotated text
|
||||
// matrix (e.g. [0, b, -b, 0, tx, ty] for 90° CCW). The layout engine
|
||||
// assumes x=horizontal, y=vertical — so we swap coordinates to match.
|
||||
let (items, rects, lines, coords_rotated) =
|
||||
let (mut items, rects, lines, coords_rotated) =
|
||||
correct_rotated_page(items, rects, lines, &rotation_votes);
|
||||
if coords_rotated {
|
||||
rotate_underline_graphics(&mut underline_rects, &mut underline_lines);
|
||||
}
|
||||
super::underline::mark_underlined_items(
|
||||
&mut items,
|
||||
&underline_rects,
|
||||
&underline_lines,
|
||||
page_num,
|
||||
);
|
||||
|
||||
let items = super::merge_text_items(items);
|
||||
let items = super::merge_subscript_items(items);
|
||||
@@ -1125,6 +1366,27 @@ fn correct_rotated_page(
|
||||
(items, rects, lines, true)
|
||||
}
|
||||
|
||||
fn rotate_underline_graphics(rects: &mut [PdfRect], lines: &mut [UnderlineLine]) {
|
||||
for rect in rects {
|
||||
let new_x = rect.y;
|
||||
let new_y = -(rect.x + rect.width.abs());
|
||||
rect.x = new_x;
|
||||
rect.y = new_y;
|
||||
std::mem::swap(&mut rect.width, &mut rect.height);
|
||||
}
|
||||
|
||||
for line in lines {
|
||||
let new_x1 = line.y1;
|
||||
let new_y1 = -line.x1;
|
||||
let new_x2 = line.y2;
|
||||
let new_y2 = -line.x2;
|
||||
line.x1 = new_x1;
|
||||
line.y1 = new_y1;
|
||||
line.x2 = new_x2;
|
||||
line.y2 = new_y2;
|
||||
}
|
||||
}
|
||||
|
||||
/// Remove near-duplicate rects (same coordinates within 0.5 pt tolerance).
|
||||
/// Some PDFs emit a full-page clip path for every text block, producing
|
||||
/// thousands of identical rects. After dedup these collapse to one rect,
|
||||
@@ -1174,6 +1436,64 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn simple_doc_with_content(content: &[u8]) -> (lopdf::Document, lopdf::ObjectId) {
|
||||
use lopdf::{dictionary, Object, Stream};
|
||||
|
||||
let mut doc = lopdf::Document::new();
|
||||
let widths: Vec<Object> = (0..=255).map(|_| 600.into()).collect();
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"FirstChar" => 0,
|
||||
"LastChar" => 255,
|
||||
"Widths" => Object::Array(widths),
|
||||
});
|
||||
let content_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
content.to_vec(),
|
||||
)));
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Contents" => Object::Reference(content_id),
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(font_id),
|
||||
},
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn extract_simple_items(content: &[u8]) -> Vec<TextItem> {
|
||||
use crate::tounicode::FontCMaps;
|
||||
|
||||
let (doc, page_id) = simple_doc_with_content(content);
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
items
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_dedup_rects_identical() {
|
||||
let mut rects = vec![rect(0.0, 0.0, 612.0, 792.0, 1); 3759];
|
||||
@@ -1223,6 +1543,140 @@ mod tests {
|
||||
assert_eq!(single.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_stroked_rule_does_not_mark_underline() {
|
||||
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (THICK) Tj ET
|
||||
4 w
|
||||
100 498 m 170 498 l S
|
||||
BT /F1 12 Tf 1 0 0 1 100 480 Tm (THIN) Tj ET
|
||||
1 w
|
||||
100 478 m 160 478 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let thick = items.iter().find(|item| item.text == "THICK").unwrap();
|
||||
let thin = items.iter().find(|item| item.text == "THIN").unwrap();
|
||||
|
||||
assert!(!thick.is_underline);
|
||||
assert!(thin.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rotated_page_underline_is_detected_after_coordinate_correction() {
|
||||
let content = b"BT /F1 12 Tf 0 1 -1 0 200 100 Tm (HELLO) Tj ET
|
||||
BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
|
||||
1 w
|
||||
202 100 m 202 170 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let hello = items.iter().find(|item| item.text == "HELLO").unwrap();
|
||||
let world = items.iter().find(|item| item.text == "WORLD").unwrap();
|
||||
|
||||
assert!(hello.is_underline);
|
||||
assert!(!world.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quote_operator_text_carries_advance_width() {
|
||||
// `'` (move-to-next-line-and-show-text) must retain the string's
|
||||
// advance width like Tj — zero-width items are invisible to
|
||||
// geometric underline/strikeout detection.
|
||||
let content = b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (first) Tj (struck) ' ET
|
||||
1 w
|
||||
99 503 m 145 503 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let struck = items.iter().find(|item| item.text == "struck").unwrap();
|
||||
|
||||
// 6 glyphs x 600/1000 x 12pt = 43.2pt, drawn one leading below Tm.
|
||||
assert!((struck.width - 43.2).abs() < 0.1);
|
||||
assert!((struck.y - 500.0).abs() < 0.1);
|
||||
assert!(struck.is_strikeout);
|
||||
assert!(!struck.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quote_operator_advances_text_matrix() {
|
||||
// Text shown after `'` on the same line must start past the shown
|
||||
// string: "CD" lands at x=114.4 (2 glyphs x 600/1000 x 12pt after
|
||||
// x=100), flush against "AB", so the merge pass joins them. Without
|
||||
// the advance "CD" overlaps "AB" at x=100 and the items stay apart.
|
||||
let content = b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (AB) ' (CD) Tj ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let merged = items.iter().find(|item| item.text == "ABCD").unwrap();
|
||||
|
||||
assert!((merged.x - 100.0).abs() < 0.1);
|
||||
assert!((merged.width - 28.8).abs() < 0.1);
|
||||
assert!((merged.y - 500.0).abs() < 0.1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn text_rise_shifts_item_baseline() {
|
||||
// Ts displaces the glyph origin vertically without touching the
|
||||
// advance; the next run at rise 0 must return to the original
|
||||
// baseline and follow the raised run horizontally.
|
||||
let content =
|
||||
b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (base) Tj 5 Ts (super) Tj 0 Ts (after) Tj ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let base = items.iter().find(|item| item.text == "base").unwrap();
|
||||
let raised = items.iter().find(|item| item.text == "super").unwrap();
|
||||
let after = items.iter().find(|item| item.text == "after").unwrap();
|
||||
|
||||
assert!((base.y - 500.0).abs() < 0.1);
|
||||
assert!((raised.y - 505.0).abs() < 0.1);
|
||||
assert!((after.y - 500.0).abs() < 0.1);
|
||||
assert!(after.x > raised.x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn actual_text_item_uses_glyph_rise() {
|
||||
// The ActualText replacement item must render at the rise in
|
||||
// effect when its glyphs were drawn — not the unshifted BDC
|
||||
// baseline, and not whatever rise is set by EMC time.
|
||||
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm \
|
||||
/Span <</ActualText (super) >> BDC 5 Ts (sup) Tj 0 Ts EMC (after) Tj ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let sup = items.iter().find(|item| item.text == "super").unwrap();
|
||||
let after = items.iter().find(|item| item.text == "after").unwrap();
|
||||
|
||||
assert!((sup.y - 505.0).abs() < 0.1);
|
||||
assert!((after.y - 500.0).abs() < 0.1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn actual_text_shown_with_quote_op_uses_moved_risen_baseline() {
|
||||
// When the tagged span's show op is `'`, the glyph position is
|
||||
// only known AFTER its line move — falling back to the BDC-entry
|
||||
// matrix would place the item on the previous line, unrisen.
|
||||
let content = b"BT /F1 12 Tf 14 TL 1 0 0 1 100 500 Tm \
|
||||
/Span <</ActualText (replaced) >> BDC 3 Ts (raw) ' 0 Ts EMC ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let item = items.iter().find(|item| item.text == "replaced").unwrap();
|
||||
|
||||
// Line move: 500 - 14 = 486; rise: +3 -> 489.
|
||||
assert!((item.y - 489.0).abs() < 0.1);
|
||||
assert!(item.width > 0.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strikeout_detected_on_risen_text() {
|
||||
// The rule crosses the glyphs at their risen position; without the
|
||||
// rise in item.y the strike window sits 4pt too low and misses.
|
||||
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm 4 Ts (struck) Tj ET
|
||||
1 w
|
||||
99 507 m 145 507 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let struck = items.iter().find(|item| item.text == "struck").unwrap();
|
||||
|
||||
assert!((struck.y - 504.0).abs() < 0.1);
|
||||
assert!(struck.is_strikeout);
|
||||
assert!(!struck.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_skip_excessive_operations() {
|
||||
use crate::tounicode::FontCMaps;
|
||||
@@ -1257,13 +1711,113 @@ mod tests {
|
||||
doc.add_object(catalog);
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extract_page_text_items(&doc, page_id, 1, &font_cmaps, false).unwrap();
|
||||
let result = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let ((items, rects, lines), _has_gid, _coords_rotated) = result;
|
||||
assert!(items.is_empty());
|
||||
assert!(rects.is_empty());
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_q_restores_current_font_for_text_decoding() {
|
||||
use crate::tounicode::FontCMaps;
|
||||
use lopdf::{dictionary, Object, Stream};
|
||||
|
||||
fn cmap_stream(dst_hex: &str) -> Stream {
|
||||
let cmap = format!(
|
||||
r#"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
/CIDSystemInfo << /Registry (Adobe) /Ordering (UCS) /Supplement 0 >> def
|
||||
/CMapName /Test-UCS def
|
||||
/CMapType 2 def
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
<41> <{dst_hex}>
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end"#
|
||||
);
|
||||
Stream::new(dictionary! {}, cmap.into_bytes())
|
||||
}
|
||||
|
||||
let mut doc = lopdf::Document::new();
|
||||
let f1_cmap = doc.add_object(Object::Stream(cmap_stream("0058"))); // X
|
||||
let f2_cmap = doc.add_object(Object::Stream(cmap_stream("0059"))); // Y
|
||||
let f1 = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"ToUnicode" => Object::Reference(f1_cmap),
|
||||
});
|
||||
let f2 = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"ToUnicode" => Object::Reference(f2_cmap),
|
||||
});
|
||||
|
||||
let content = b"BT /F1 12 Tf 10 700 Tm <41> Tj ET
|
||||
q
|
||||
BT /F2 12 Tf 20 700 Tm <41> Tj ET
|
||||
Q
|
||||
BT 30 700 Tm <41> Tj ET";
|
||||
let content_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
content.to_vec(),
|
||||
)));
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Contents" => Object::Reference(content_id),
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(f1),
|
||||
"F2" => Object::Reference(f2),
|
||||
},
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let text = items
|
||||
.iter()
|
||||
.map(|item| item.text.as_str())
|
||||
.collect::<String>();
|
||||
|
||||
assert_eq!(text, "XYX");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_pdf_comments() {
|
||||
// Basic comment stripping
|
||||
|
||||
+805
-59
File diff suppressed because it is too large
Load Diff
+254
-17
@@ -29,8 +29,12 @@ pub(crate) fn detect_columns(
|
||||
const MIN_ITEMS_PER_COLUMN: usize = 10;
|
||||
const NOISE_FRACTION: f32 = 0.15;
|
||||
|
||||
// Get items for this page
|
||||
let page_items: Vec<&TextItem> = items.iter().filter(|i| i.page == page).collect();
|
||||
// Get items for this page. Strip Image placeholders — an image's left edge
|
||||
// would otherwise count toward the column projection profile.
|
||||
let page_items: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|i| i.page == page && crate::extractor::is_text_layout_item(i))
|
||||
.collect();
|
||||
|
||||
if page_items.is_empty() {
|
||||
return vec![];
|
||||
@@ -670,9 +674,24 @@ fn validate_and_build_columns(
|
||||
page: u32,
|
||||
center_assign: bool,
|
||||
) -> Vec<ColumnRegion> {
|
||||
// Compute Y range of the page
|
||||
let y_min = page_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
let y_max = page_items
|
||||
// Compute the Y range from column-eligible items only — the same items
|
||||
// the histogram counted. Spanning items (full-width captions, titles)
|
||||
// are excluded from the projection, so letting them stretch the page's
|
||||
// vertical extent here would sink the overlap ratio for column regions
|
||||
// that legitimately occupy only part of the page (e.g. two-column text
|
||||
// below a figure).
|
||||
let x_span = page_items
|
||||
.iter()
|
||||
.map(|i| i.x + effective_width(i))
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
- page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let narrow: Vec<&&TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|i| effective_width(i) <= x_span * 0.6)
|
||||
.collect();
|
||||
let span_items: &[&&TextItem] = if narrow.is_empty() { &[] } else { &narrow };
|
||||
let y_min = span_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
let y_max = span_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
@@ -718,6 +737,10 @@ fn validate_and_build_columns(
|
||||
(right_items.len(), left_items.len())
|
||||
};
|
||||
if larger < min_items || smaller < 3 {
|
||||
debug!(
|
||||
" valley rejected: counts smaller={} larger={}",
|
||||
smaller, larger
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -731,6 +754,7 @@ fn validate_and_build_columns(
|
||||
&right_items
|
||||
};
|
||||
if is_list_marker_column(smaller_items) {
|
||||
debug!(" valley rejected: list-marker column");
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -755,6 +779,13 @@ fn validate_and_build_columns(
|
||||
let overlap = (overlap_max - overlap_min).max(0.0);
|
||||
|
||||
if overlap / y_range < min_vertical_span {
|
||||
debug!(
|
||||
" valley rejected: overlap {:.0}/{:.0} = {:.2} < {:.2}",
|
||||
overlap,
|
||||
y_range,
|
||||
overlap / y_range,
|
||||
min_vertical_span
|
||||
);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -1122,6 +1153,22 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds(items, &HashMap::new(), &HashSet::new())
|
||||
}
|
||||
|
||||
/// Group text items into lines without removing numeric page headers or footers.
|
||||
///
|
||||
/// Plain-text extraction uses this path because every extracted item is part of
|
||||
/// the API result. Markdown conversion keeps using [`group_into_lines`], where
|
||||
/// page-number suppression is an intentional presentation cleanup.
|
||||
pub fn group_into_lines_preserving_all_text(items: Vec<TextItem>) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
/// Group text items into lines, using pre-computed per-page adaptive thresholds
|
||||
/// from Canva-style letter-spacing detection. Falls back to computing the
|
||||
/// threshold from item gaps when no pre-computed value is available.
|
||||
@@ -1129,16 +1176,74 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
&HashMap::new(),
|
||||
)
|
||||
}
|
||||
|
||||
/// Like `group_into_lines_with_thresholds`, but items inside chart regions
|
||||
/// are excluded from column detection: chart text scattered across the page
|
||||
/// fills the gutter in the projection histogram, so two-column pages read as
|
||||
/// one column and same-baseline items from both columns fuse into one line
|
||||
/// (headings absorbed into the neighboring column's body text).
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
&HashMap::new(),
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
true,
|
||||
)
|
||||
}
|
||||
|
||||
fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
filter_page_numbers: bool,
|
||||
) -> Vec<TextLine> {
|
||||
if items.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Filter out page numbers (standalone numbers at top/bottom of page)
|
||||
let items: Vec<TextItem> = items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.collect();
|
||||
// Markdown output omits standalone numeric headers/footers. Plain-text
|
||||
// callers opt out because dropping extracted text violates that API.
|
||||
let items = if filter_page_numbers {
|
||||
items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.collect()
|
||||
} else {
|
||||
items
|
||||
};
|
||||
|
||||
// Get unique pages
|
||||
let mut pages: Vec<u32> = items.iter().map(|i| i.page).collect();
|
||||
@@ -1155,8 +1260,67 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
// Non-Canva pages use the default 0.10 threshold.
|
||||
let adaptive_threshold = page_thresholds.get(&page).copied().unwrap_or(0.10);
|
||||
|
||||
// Detect columns for this page
|
||||
let columns = detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
// Image-backed region graphs recover local/asymmetric column flows
|
||||
// that a whole-page projection cannot represent. Charts already have
|
||||
// their own positioned-region ordering and therefore stay on that path.
|
||||
if !chart_regions.contains_key(&page) {
|
||||
let preliminary_columns =
|
||||
detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
let detected_split =
|
||||
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
|
||||
if let Some(band) = image_regions.get(&page).and_then(|regions| {
|
||||
super::reading_order::infer_image_anchored_flow(
|
||||
&page_items,
|
||||
regions,
|
||||
detected_split,
|
||||
)
|
||||
}) {
|
||||
debug!(
|
||||
"page {}: image-anchored region graph split={:.1} y=[{:.1}..{:.1}]",
|
||||
page, band.split_x, band.y_bottom, band.y_top
|
||||
);
|
||||
for node in super::reading_order::build_region_graph(page_items, band) {
|
||||
debug!(
|
||||
"page {}: region node {:?} items={}",
|
||||
page,
|
||||
node.kind,
|
||||
node.items.len()
|
||||
);
|
||||
all_lines.extend(group_single_column(node.items, adaptive_threshold));
|
||||
}
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Detect columns for this page, blind to chart text.
|
||||
debug!(
|
||||
"page {}: grouping chart-aware={} regions={:?}",
|
||||
page,
|
||||
chart_regions.contains_key(&page),
|
||||
chart_regions.get(&page).map(|v| v
|
||||
.iter()
|
||||
.map(|&(a, b, c, d)| (a as i32, b as i32, c as i32, d as i32))
|
||||
.collect::<Vec<_>>())
|
||||
);
|
||||
let columns = match chart_regions.get(&page).filter(|r| !r.is_empty()) {
|
||||
Some(regions) => {
|
||||
let col_input: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Tight bounds: this only blinds the histogram to
|
||||
// chart-internal text; rows adjacent to the chart
|
||||
// belong to the column layout.
|
||||
!regions.iter().any(|&(x0, y0, x1, y1)| {
|
||||
cx >= x0 - 2.0 && cx <= x1 + 2.0 && it.y >= y0 - 2.0 && it.y <= y1 + 2.0
|
||||
})
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
detect_columns(&col_input, page, table_pages.contains(&page))
|
||||
}
|
||||
None => detect_columns(&page_items, page, table_pages.contains(&page)),
|
||||
};
|
||||
|
||||
if columns.len() <= 1 {
|
||||
// Single column - use simple sorting
|
||||
@@ -1230,11 +1394,7 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
ci,
|
||||
item.x,
|
||||
item.y,
|
||||
if item.text.len() > 60 {
|
||||
&item.text[..60]
|
||||
} else {
|
||||
&item.text
|
||||
}
|
||||
super::trace_text_preview(&item.text, 60)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1446,6 +1606,56 @@ fn group_single_column(items: Vec<TextItem>, adaptive_threshold: f32) -> Vec<Tex
|
||||
}
|
||||
}
|
||||
}
|
||||
// Same baseline, but separated by a wide void, with the incoming
|
||||
// run starting alphabetic: the neighboring column's body text
|
||||
// sharing a y with this line, in gutters too narrow for column
|
||||
// detection. Both sides must be multi-word prose — TOC page
|
||||
// numbers, dot leaders, and outline-numbered table cells (which
|
||||
// start with digits) stay joined.
|
||||
if let Some(last_item) = last_line.items.last() {
|
||||
let gap = item.x - (last_item.x + last_item.width);
|
||||
if gap > (item.font_size.max(last_item.font_size) * 3.0).max(30.0)
|
||||
&& item
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|c| c.is_alphabetic())
|
||||
{
|
||||
// The incoming run must be substantial prose; the line
|
||||
// side may be short (a wrapped heading's last words).
|
||||
let incoming_wordy = {
|
||||
let t = item.text.trim();
|
||||
t.split_whitespace().count() >= 3
|
||||
&& t.chars().filter(|c| c.is_alphabetic()).count() >= 10
|
||||
};
|
||||
let line_text = last_line
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| i.text.trim())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let line_wordy = line_text.split_whitespace().count() >= 2
|
||||
&& line_text.chars().filter(|c| c.is_alphabetic()).count() >= 8;
|
||||
// Lowercase starts are mid-sentence continuations and
|
||||
// split on prose signals alone. Uppercase starts also
|
||||
// need a bold-style mismatch between the runs — a bold
|
||||
// heading beside regular body text — otherwise same-style
|
||||
// label rows (feature tiles, legends) would shatter.
|
||||
let starts_lower = item
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|c| c.is_lowercase());
|
||||
// The whole line must be bold (a heading), not merely
|
||||
// its last run — mixed bold-label/value rows stay joined.
|
||||
let style_mismatch = last_line.items.iter().all(|i| i.is_bold) && !item.is_bold;
|
||||
if line_wordy && incoming_wordy && (starts_lower || style_mismatch) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
true
|
||||
});
|
||||
|
||||
@@ -1494,6 +1704,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1517,6 +1729,29 @@ mod tests {
|
||||
items
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_baseline_wide_gap_lowercase_continuation_splits() {
|
||||
// Heading in the left column, mid-sentence body text from the right
|
||||
// column at the same y, separated by a wide void: two lines.
|
||||
let items = vec![
|
||||
make_item(1, 94.0, 242.0, "6.2. Expectations for Re-Hiring Staff"),
|
||||
make_item(1, 380.0, 242.0, "they had no plans to re-hire and more"),
|
||||
];
|
||||
let lines = group_single_column(items, 0.10);
|
||||
assert_eq!(lines.len(), 2, "independent column runs must not fuse");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_baseline_wide_gap_table_label_stays_joined() {
|
||||
// Outline-numbered cell content to the right: table-ish, keep joined.
|
||||
let items = vec![
|
||||
make_item(1, 94.0, 242.0, "2. Embracing complexity in"),
|
||||
make_item(1, 380.0, 242.0, "2.1 Systems thinking and practice"),
|
||||
];
|
||||
let lines = group_single_column(items, 0.10);
|
||||
assert_eq!(lines.len(), 1, "numbered table cells stay on one line");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn three_zone_layout_detected() {
|
||||
// Left months (x=15..330), right months (x=345..660), sidebar (x=675..800)
|
||||
@@ -1623,6 +1858,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
@@ -78,6 +78,8 @@ pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> V
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Link(url),
|
||||
mcid: None,
|
||||
});
|
||||
@@ -316,6 +318,8 @@ pub(crate) fn walk_form_fields(
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::FormField,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
+1006
-25
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,591 @@
|
||||
//! Region-graph evidence for page reading order.
|
||||
//!
|
||||
//! Whole-page column histograms fail when images or spanning captions occupy
|
||||
//! only part of a page. This module turns image geometry and repeated row
|
||||
//! gutters into a small directed acyclic graph: content above a local column
|
||||
//! band, the left flow, the right flow, and content below it. The graph is
|
||||
//! deliberately evidence-gated; ordinary pages keep the established layout
|
||||
//! path.
|
||||
|
||||
use crate::text_utils::{effective_width, is_cjk_char, is_rtl_text};
|
||||
use crate::types::TextItem;
|
||||
|
||||
const MIN_IMAGE_WIDTH: f32 = 60.0;
|
||||
const MIN_IMAGE_HEIGHT: f32 = 40.0;
|
||||
const MIN_ROW_GUTTER: f32 = 8.0;
|
||||
const SPLIT_CLUSTER_TOLERANCE: f32 = 20.0;
|
||||
const MIN_ALIGNED_ROWS: usize = 4;
|
||||
|
||||
pub(crate) type ImageRegion = (f32, f32, f32, f32);
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub(crate) struct ColumnFlowBand {
|
||||
pub(crate) split_x: f32,
|
||||
pub(crate) y_bottom: f32,
|
||||
pub(crate) y_top: f32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum RegionKind {
|
||||
FullWidth,
|
||||
Column,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RegionNode {
|
||||
pub(crate) kind: RegionKind,
|
||||
pub(crate) items: Vec<TextItem>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Row<'a> {
|
||||
y: f32,
|
||||
items: Vec<&'a TextItem>,
|
||||
}
|
||||
|
||||
fn page_x_bounds(items: &[TextItem], images: &[ImageRegion]) -> Option<(f32, f32)> {
|
||||
let text_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let text_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + effective_width(item))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let image_min = images
|
||||
.iter()
|
||||
.map(|region| region.0.min(region.2))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_max = images
|
||||
.iter()
|
||||
.map(|region| region.0.max(region.2))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let x_min = text_min.min(image_min);
|
||||
let x_max = text_max.max(image_max);
|
||||
(x_min.is_finite() && x_max.is_finite() && x_max > x_min).then_some((x_min, x_max))
|
||||
}
|
||||
|
||||
fn group_rows(items: &[TextItem]) -> Vec<Row<'_>> {
|
||||
const Y_TOLERANCE: f32 = 3.0;
|
||||
let mut sorted: Vec<&TextItem> = items.iter().collect();
|
||||
sorted.sort_by(|left, right| right.y.total_cmp(&left.y));
|
||||
let mut rows: Vec<Row<'_>> = Vec::new();
|
||||
for item in sorted {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| (row.y - item.y).abs() <= Y_TOLERANCE)
|
||||
{
|
||||
row.items.push(item);
|
||||
row.y = row.items.iter().map(|member| member.y).sum::<f32>() / row.items.len() as f32;
|
||||
} else {
|
||||
rows.push(Row {
|
||||
y: item.y,
|
||||
items: vec![item],
|
||||
});
|
||||
}
|
||||
}
|
||||
for row in &mut rows {
|
||||
row.items.sort_by(|left, right| left.x.total_cmp(&right.x));
|
||||
}
|
||||
rows
|
||||
}
|
||||
|
||||
fn side_is_prose(items: &[&TextItem]) -> bool {
|
||||
let text = items
|
||||
.iter()
|
||||
.map(|item| item.text.trim())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let alphabetic_count = text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphabetic())
|
||||
.count();
|
||||
let cjk_count = text
|
||||
.chars()
|
||||
.filter(|character| is_cjk_char(*character))
|
||||
.count();
|
||||
(text.split_whitespace().count() >= 3 || cjk_count >= 10) && alphabetic_count >= 10
|
||||
}
|
||||
|
||||
fn aligned_row_split(row: &Row<'_>, x_min: f32, x_max: f32) -> Option<f32> {
|
||||
if row.items.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let page_width = x_max - x_min;
|
||||
let center_low = x_min + page_width * 0.25;
|
||||
let center_high = x_min + page_width * 0.75;
|
||||
row.items
|
||||
.windows(2)
|
||||
.filter_map(|pair| {
|
||||
let left_end = pair[0].x + effective_width(pair[0]);
|
||||
let right_start = pair[1].x;
|
||||
let gap = right_start - left_end;
|
||||
let split_x = (left_end + right_start) / 2.0;
|
||||
if gap < MIN_ROW_GUTTER || split_x < center_low || split_x > center_high {
|
||||
return None;
|
||||
}
|
||||
let left: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 < split_x)
|
||||
.collect();
|
||||
let right: Vec<&TextItem> = row
|
||||
.items
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|item| item.x + effective_width(item) / 2.0 >= split_x)
|
||||
.collect();
|
||||
(side_is_prose(&left) && side_is_prose(&right)).then_some((split_x, gap))
|
||||
})
|
||||
.max_by(|left, right| left.1.total_cmp(&right.1))
|
||||
.map(|candidate| candidate.0)
|
||||
}
|
||||
|
||||
fn local_flow_below_full_width_image(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
let full_width_images: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let width = (x1 - x0).abs();
|
||||
let height = (y1 - y0).abs();
|
||||
width >= page_width * 0.65 && height >= 60.0
|
||||
})
|
||||
.collect();
|
||||
// A local column flow below an image is only unambiguous for a single,
|
||||
// nearly square hero/figure. Wide report banners and full-page artwork
|
||||
// frequently sit above unrelated page furniture whose aligned labels can
|
||||
// mimic prose columns.
|
||||
if full_width_images.len() != 1 {
|
||||
return None;
|
||||
}
|
||||
let (image_x0, _, image_x1, _) = full_width_images[0];
|
||||
let anchor_width = (image_x1 - image_x0).abs();
|
||||
let anchor_height = (full_width_images[0].3 - full_width_images[0].1).abs();
|
||||
if anchor_width < page_width * 0.85
|
||||
|| anchor_height < anchor_width * 0.85
|
||||
|| anchor_height > anchor_width * 1.2
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let image_bottom = full_width_images
|
||||
.iter()
|
||||
.map(|&(_, y0, _, y1)| y0.min(y1))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if !image_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let below: Vec<TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| item.y < image_bottom && item.y >= image_bottom - 220.0)
|
||||
.cloned()
|
||||
.collect();
|
||||
let candidates: Vec<(f32, f32)> = group_rows(&below)
|
||||
.into_iter()
|
||||
.filter_map(|row| aligned_row_split(&row, x_min, x_max).map(|split| (split, row.y)))
|
||||
.collect();
|
||||
if candidates.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut clusters: Vec<Vec<(f32, f32)>> = Vec::new();
|
||||
for candidate in candidates {
|
||||
if let Some(cluster) = clusters.iter_mut().find(|cluster| {
|
||||
let mean = cluster.iter().map(|entry| entry.0).sum::<f32>() / cluster.len() as f32;
|
||||
(mean - candidate.0).abs() <= SPLIT_CLUSTER_TOLERANCE
|
||||
}) {
|
||||
cluster.push(candidate);
|
||||
} else {
|
||||
clusters.push(vec![candidate]);
|
||||
}
|
||||
}
|
||||
let dominant = clusters.into_iter().max_by_key(Vec::len)?;
|
||||
if dominant.len() < MIN_ALIGNED_ROWS {
|
||||
return None;
|
||||
}
|
||||
let split_x = dominant.iter().map(|entry| entry.0).sum::<f32>() / dominant.len() as f32;
|
||||
let y_top = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
let image_gap = image_bottom - y_top;
|
||||
if !(60.0..=120.0).contains(&image_gap) {
|
||||
return None;
|
||||
}
|
||||
let y_bottom = dominant
|
||||
.iter()
|
||||
.map(|entry| entry.1)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if y_top - y_bottom > 130.0 {
|
||||
return None;
|
||||
}
|
||||
log::debug!(
|
||||
"page {}: full-width image flow images={} aligned_rows={} split={:.1} page=[{:.1}..{:.1}] image_bottom={:.1} y=[{:.1}..{:.1}] full_width={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
images.len(),
|
||||
dominant.len(),
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
image_bottom,
|
||||
y_bottom,
|
||||
y_top,
|
||||
full_width_images
|
||||
);
|
||||
Some(ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
})
|
||||
}
|
||||
|
||||
fn paired_column_images(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
split_x: f32,
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
let page_width = x_max - x_min;
|
||||
if split_x < x_min + page_width * 0.4 || split_x > x_min + page_width * 0.6 {
|
||||
return None;
|
||||
}
|
||||
let qualifying: Vec<ImageRegion> = images
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x0, y0, x1, y1)| {
|
||||
let image_left = x0.min(x1);
|
||||
let image_right = x0.max(x1);
|
||||
let confined_to_one_column = image_right <= split_x || image_left >= split_x;
|
||||
confined_to_one_column
|
||||
&& (x1 - x0).abs() >= MIN_IMAGE_WIDTH
|
||||
&& (y1 - y0).abs() >= MIN_IMAGE_HEIGHT
|
||||
})
|
||||
.collect();
|
||||
let wide_images: Vec<ImageRegion> = qualifying
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(x0, _, x1, _)| (x1 - x0).abs() >= page_width * 0.35)
|
||||
.collect();
|
||||
if qualifying.len() < 3 || wide_images.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
let has_left = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 < split_x);
|
||||
let has_right = qualifying
|
||||
.iter()
|
||||
.any(|&(x0, _, x1, _)| (x0 + x1) / 2.0 >= split_x);
|
||||
if !has_left || !has_right {
|
||||
return None;
|
||||
}
|
||||
// A meaningful image-backed column flow spans multiple vertical panels.
|
||||
// Three same-row header/logo images can otherwise satisfy the image count
|
||||
// and send an ordinary asymmetric page through sequential column order.
|
||||
let image_y_min = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.min(region.3))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let image_y_max = wide_images
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let has_vertical_stack = wide_images.iter().enumerate().any(|(index, left)| {
|
||||
wide_images.iter().skip(index + 1).any(|right| {
|
||||
let same_side =
|
||||
((left.0 + left.2) / 2.0 < split_x) == ((right.0 + right.2) / 2.0 < split_x);
|
||||
let left_center = (left.1 + left.3) / 2.0;
|
||||
let right_center = (right.1 + right.3) / 2.0;
|
||||
let left_height = (left.3 - left.1).abs();
|
||||
let right_height = (right.3 - right.1).abs();
|
||||
let vertical_gap = if left.1.max(left.3) < right.1.min(right.3) {
|
||||
right.1.min(right.3) - left.1.max(left.3)
|
||||
} else if right.1.max(right.3) < left.1.min(left.3) {
|
||||
left.1.min(left.3) - right.1.max(right.3)
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
same_side
|
||||
&& (left_center - right_center).abs() >= left_height.min(right_height) * 0.5
|
||||
&& vertical_gap <= left_height.max(right_height) * 0.5
|
||||
})
|
||||
});
|
||||
if image_y_max - image_y_min < page_width * 0.45 || !has_vertical_stack {
|
||||
return None;
|
||||
}
|
||||
let y_top = qualifying
|
||||
.iter()
|
||||
.map(|region| region.1.max(region.3))
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
+ 3.0;
|
||||
// Only column-confined text proves the lower extent of the flow. A
|
||||
// spanning heading or caption below the columns must become the trailing
|
||||
// full-width node rather than stretching the column band to the page foot.
|
||||
let y_bottom = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
let item_right = item.x + effective_width(item);
|
||||
item.y <= y_top && (item_right <= split_x || item.x >= split_x)
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.fold(f32::INFINITY, f32::min)
|
||||
- 3.0;
|
||||
if !y_bottom.is_finite() {
|
||||
return None;
|
||||
}
|
||||
let distinct_rows = |right: bool| {
|
||||
let mut ys: Vec<f32> = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
item.y <= y_top && (item.x + effective_width(item) / 2.0 >= split_x) == right
|
||||
})
|
||||
.map(|item| item.y)
|
||||
.collect();
|
||||
ys.sort_by(|left, right| left.total_cmp(right));
|
||||
ys.dedup_by(|left, right| (*left - *right).abs() <= 3.0);
|
||||
ys.len()
|
||||
};
|
||||
let left_rows = distinct_rows(false);
|
||||
let right_rows = distinct_rows(true);
|
||||
let line_balance = left_rows.min(right_rows) as f32 / left_rows.max(right_rows).max(1) as f32;
|
||||
(left_rows >= 5 && right_rows >= 5 && line_balance < 0.55).then(|| {
|
||||
log::debug!(
|
||||
"page {}: paired-image flow qualifying_images={} rows={}/{} split={:.1} page=[{:.1}..{:.1}] y=[{:.1}..{:.1}] images={:?}",
|
||||
items.first().map_or(0, |item| item.page),
|
||||
qualifying.len(),
|
||||
left_rows,
|
||||
right_rows,
|
||||
split_x,
|
||||
x_min,
|
||||
x_max,
|
||||
y_bottom,
|
||||
y_top,
|
||||
qualifying
|
||||
);
|
||||
ColumnFlowBand {
|
||||
split_x,
|
||||
y_bottom,
|
||||
y_top,
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
pub(crate) fn infer_image_anchored_flow(
|
||||
items: &[TextItem],
|
||||
images: &[ImageRegion],
|
||||
detected_split: Option<f32>,
|
||||
) -> Option<ColumnFlowBand> {
|
||||
if items.is_empty() || images.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let (x_min, x_max) = page_x_bounds(items, images)?;
|
||||
detected_split
|
||||
.and_then(|split_x| paired_column_images(items, images, split_x, x_min, x_max))
|
||||
.or_else(|| local_flow_below_full_width_image(items, images, x_min, x_max))
|
||||
}
|
||||
|
||||
/// Partition a page into the topological order `above -> left -> right -> below`.
|
||||
/// These edges encode the reading-order DAG; empty nodes are omitted.
|
||||
pub(crate) fn build_region_graph(items: Vec<TextItem>, band: ColumnFlowBand) -> Vec<RegionNode> {
|
||||
let mut above = Vec::new();
|
||||
let mut left = Vec::new();
|
||||
let mut right = Vec::new();
|
||||
let mut below = Vec::new();
|
||||
for item in items {
|
||||
if item.y > band.y_top {
|
||||
above.push(item);
|
||||
} else if item.y < band.y_bottom {
|
||||
below.push(item);
|
||||
} else if item.x + effective_width(&item) / 2.0 < band.split_x {
|
||||
left.push(item);
|
||||
} else {
|
||||
right.push(item);
|
||||
}
|
||||
}
|
||||
let rtl = is_rtl_text(left.iter().chain(right.iter()).map(|item| &item.text));
|
||||
let mut ordered = vec![(RegionKind::FullWidth, above)];
|
||||
if rtl {
|
||||
ordered.push((RegionKind::Column, right));
|
||||
ordered.push((RegionKind::Column, left));
|
||||
} else {
|
||||
ordered.push((RegionKind::Column, left));
|
||||
ordered.push((RegionKind::Column, right));
|
||||
}
|
||||
ordered.push((RegionKind::FullWidth, below));
|
||||
ordered
|
||||
.into_iter()
|
||||
.filter_map(|(kind, items)| (!items.is_empty()).then_some(RegionNode { kind, items }))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.into(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 11.0,
|
||||
font: "F1".into(),
|
||||
font_size: 11.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_local_two_column_flow() {
|
||||
let mut items = vec![
|
||||
item("A full width caption", 55.0, 230.0, 430.0),
|
||||
item("A trailing full width heading", 55.0, 80.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
let band = infer_image_anchored_flow(&items, &images, None).unwrap();
|
||||
assert!((band.split_x - 272.5).abs() < 2.0);
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph.len(), 4);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "A trailing full width heading");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_width_image_anchors_cjk_column_flow() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..5 {
|
||||
let y = 170.0 - index as f32 * 14.0;
|
||||
items.push(item("左栏这是没有空格的正文内容", 55.0, y, 210.0));
|
||||
items.push(item("右栏这是没有空格的正文内容", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 250.0, 490.0, 680.0)];
|
||||
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_images_anchor_unbalanced_column_flows() {
|
||||
let mut items = vec![
|
||||
item("running header", 55.0, 700.0, 430.0),
|
||||
item("trailing full width caption", 55.0, 300.0, 430.0),
|
||||
];
|
||||
for index in 0..5 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
500.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 5..12 {
|
||||
items.push(item(
|
||||
"right continuation prose words",
|
||||
280.0,
|
||||
520.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 530.0, 255.0, 680.0),
|
||||
(55.0, 380.0, 255.0, 530.0),
|
||||
(280.0, 560.0, 490.0, 680.0),
|
||||
];
|
||||
let band = infer_image_anchored_flow(&items, &images, Some(270.0)).unwrap();
|
||||
let graph = build_region_graph(items, band);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[1].kind, RegionKind::Column);
|
||||
assert_eq!(graph[2].kind, RegionKind::Column);
|
||||
assert_eq!(graph[3].kind, RegionKind::FullWidth);
|
||||
assert_eq!(graph[3].items[0].text, "trailing full width caption");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rtl_region_graph_reads_right_column_first() {
|
||||
let items = vec![
|
||||
item("A long English report header", 55.0, 250.0, 430.0),
|
||||
item("نص العمود الأيسر", 55.0, 150.0, 180.0),
|
||||
item("نص العمود الأيمن", 300.0, 150.0, 180.0),
|
||||
];
|
||||
let graph = build_region_graph(
|
||||
items,
|
||||
ColumnFlowBand {
|
||||
split_x: 270.0,
|
||||
y_bottom: 100.0,
|
||||
y_top: 200.0,
|
||||
},
|
||||
);
|
||||
|
||||
assert_eq!(graph.len(), 3);
|
||||
assert_eq!(graph[0].kind, RegionKind::FullWidth);
|
||||
assert!(graph[1].items[0].x > graph[2].items[0].x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paired_header_logos_do_not_anchor_page_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
items.push(item(
|
||||
"left prose words",
|
||||
55.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
for index in 0..30 {
|
||||
items.push(item(
|
||||
"right prose words",
|
||||
280.0,
|
||||
700.0 - index as f32 * 14.0,
|
||||
200.0,
|
||||
));
|
||||
}
|
||||
let images = vec![
|
||||
(55.0, 720.0, 205.0, 770.0),
|
||||
(60.0, 718.0, 210.0, 768.0),
|
||||
(280.0, 720.0, 450.0, 770.0),
|
||||
];
|
||||
assert!(infer_image_anchored_flow(&items, &images, Some(270.0)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_banner_does_not_anchor_local_columns() {
|
||||
let mut items = Vec::new();
|
||||
for index in 0..7 {
|
||||
let y = 270.0 - index as f32 * 14.0;
|
||||
items.push(item("left column prose words", 55.0, y, 210.0));
|
||||
items.push(item("right column prose words", 280.0, y, 210.0));
|
||||
}
|
||||
let images = vec![(55.0, 310.0, 490.0, 550.0)];
|
||||
assert!(infer_image_anchored_flow(&items, &images, None).is_none());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,844 @@
|
||||
//! Geometric underline detection.
|
||||
//!
|
||||
//! PDFs have no underline font flag — underlines are drawn as separate
|
||||
//! graphics: stroked horizontal lines (`l`/`S` operators) or thin filled
|
||||
//! rectangles (`re`/`f`). This pass correlates those graphics with text
|
||||
//! items after extraction: an item is underlined when a horizontal
|
||||
//! line/thin rect sits just below its baseline and covers most of its
|
||||
//! horizontal extent.
|
||||
//!
|
||||
//! Repeated same-span rules are treated as table/form rulings rather than
|
||||
//! underlines, which avoids marking every cell in ruled tables.
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::types::{ItemType, PdfRect, TextItem};
|
||||
|
||||
/// Max thickness (pt) for a stroked line / filled rect to count as an
|
||||
/// underline rule rather than a border or decorative band.
|
||||
const MAX_RULE_THICKNESS: f32 = 2.0;
|
||||
|
||||
/// Fraction of the item's width that the rule must cover horizontally.
|
||||
const MIN_X_OVERLAP: f32 = 0.6;
|
||||
|
||||
/// Same-span rules repeated at this many y-levels are usually table/form
|
||||
/// rulings, not semantic underlines.
|
||||
const MIN_REPEATED_RULE_LEVELS: usize = 3;
|
||||
|
||||
/// Vertical tolerance for considering two rules to be on the same row edge.
|
||||
const RULE_Y_DEDUP_EPS: f32 = 2.0;
|
||||
|
||||
/// Horizontal span similarity required when clustering repeated rulings.
|
||||
const RULE_SPAN_OVERLAP_RATIO: f32 = 0.8;
|
||||
const RULE_SPAN_WIDTH_RATIO: f32 = 1.5;
|
||||
|
||||
/// Multiple separated rule segments on one row are usually per-column table
|
||||
/// header/body separators.
|
||||
const MIN_SEGMENTED_ROW_RULES: usize = 3;
|
||||
const MIN_SEGMENTED_ROW_GAPS: usize = 2;
|
||||
const SEGMENTED_ROW_GAP_MIN: f32 = 12.0;
|
||||
|
||||
/// A single rule under several widely separated items is usually a table
|
||||
/// header/body separator, not a sentence underline.
|
||||
const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
pub(crate) y1: f32,
|
||||
pub(crate) x2: f32,
|
||||
pub(crate) y2: f32,
|
||||
pub(crate) stroke_width: f32,
|
||||
pub(crate) page: u32,
|
||||
}
|
||||
|
||||
/// A horizontal rule candidate in page coordinates (PDF y-up).
|
||||
#[derive(Clone)]
|
||||
struct Rule {
|
||||
x1: f32,
|
||||
x2: f32,
|
||||
y: f32,
|
||||
}
|
||||
|
||||
impl Rule {
|
||||
fn width(&self) -> f32 {
|
||||
self.x2 - self.x1
|
||||
}
|
||||
}
|
||||
|
||||
fn rules_from_graphics(rects: &[PdfRect], lines: &[UnderlineLine], page: u32) -> Vec<Rule> {
|
||||
let mut rules: Vec<Rule> = Vec::new();
|
||||
for l in lines {
|
||||
if l.page != page {
|
||||
continue;
|
||||
}
|
||||
// Horizontal stroked line (tolerate slight skew).
|
||||
if l.stroke_width <= MAX_RULE_THICKNESS && (l.y1 - l.y2).abs() <= MAX_RULE_THICKNESS {
|
||||
let (x1, x2) = if l.x1 <= l.x2 {
|
||||
(l.x1, l.x2)
|
||||
} else {
|
||||
(l.x2, l.x1)
|
||||
};
|
||||
if x2 - x1 > 1.0 {
|
||||
rules.push(Rule {
|
||||
x1,
|
||||
x2,
|
||||
y: (l.y1 + l.y2) / 2.0,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
for r in rects {
|
||||
if r.page != page {
|
||||
continue;
|
||||
}
|
||||
// Thin filled rect used as an underline rule. Extents are
|
||||
// normalized first: `re` operands pass through the CTM, so
|
||||
// width/height can be negative (flipped axes / negative scale) —
|
||||
// without normalization negative-width rules are missed and
|
||||
// negative-height bands sneak past the thickness check.
|
||||
let (x1, x2) = if r.width >= 0.0 {
|
||||
(r.x, r.x + r.width)
|
||||
} else {
|
||||
(r.x + r.width, r.x)
|
||||
};
|
||||
if r.height.abs() <= MAX_RULE_THICKNESS && x2 - x1 > 1.0 {
|
||||
rules.push(Rule {
|
||||
x1,
|
||||
x2,
|
||||
y: r.y + r.height / 2.0,
|
||||
});
|
||||
}
|
||||
}
|
||||
rules
|
||||
}
|
||||
|
||||
fn discard_repeated_ruling_rules(
|
||||
rules: Vec<Rule>,
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) -> Vec<Rule> {
|
||||
if rules.len() < MIN_REPEATED_RULE_LEVELS {
|
||||
return rules;
|
||||
}
|
||||
|
||||
rules
|
||||
.iter()
|
||||
.filter(|rule| {
|
||||
// A rule snugly owned by one text line is an underline even when
|
||||
// span-similar rules repeat down the page — documents that
|
||||
// underline many full-width lines (dense CJK business docs) look
|
||||
// exactly like table rulings to the repetition check, which used
|
||||
// to discard every one of them. Table rulings fail snugness:
|
||||
// row separators extend past their cells' text (or have no text
|
||||
// on the baseline above), and multi-column matches are still
|
||||
// culled by the tabular filter afterwards.
|
||||
// Same-row segmented rules (column-header separators) are
|
||||
// always rulings — each segment snugly owns its column label,
|
||||
// so snugness must not override that check.
|
||||
!is_segmented_row_ruling_rule(rule, &rules)
|
||||
&& ((has_snug_text_owner(rule, items)
|
||||
&& !has_flanking_verticals(rule, rects, lines, page))
|
||||
|| !is_repeated_ruling_rule(rule, &rules))
|
||||
})
|
||||
.cloned()
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// True when a single text item both matches the rule vertically (baseline
|
||||
/// window) and horizontally contains it: the rule may not extend past the
|
||||
/// item's span by more than ~0.75em on either side. Underlines are drawn to
|
||||
/// the width of the text they decorate; table/form rulings span cells or
|
||||
/// full table width and overshoot any single item.
|
||||
/// A rule flanked by vertical strokes at its ends is a table/box border
|
||||
/// row edge, not an underline — underlined text lines have no vertical
|
||||
/// rules rising from their ends. Checked against raw stroked lines: a
|
||||
/// near-vertical segment whose x sits at either end of the rule and whose
|
||||
/// y-range covers the rule's row.
|
||||
fn has_flanking_verticals(
|
||||
rule: &Rule,
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) -> bool {
|
||||
// A drawn rect that CONTAINS the rule vetoes rescue only with GRID
|
||||
// EVIDENCE: another drawn rect abutting it vertically (cell rows tile).
|
||||
// Height alone can't separate a table cell from a decorative callout
|
||||
// panel — genuine underlines live inside isolated filled panels, and
|
||||
// multiline table cells can be arbitrarily tall.
|
||||
let norm = |r: &PdfRect| {
|
||||
let (x_lo, x_hi) = if r.width >= 0.0 {
|
||||
(r.x, r.x + r.width)
|
||||
} else {
|
||||
(r.x + r.width, r.x)
|
||||
};
|
||||
let (y_lo, y_hi) = if r.height >= 0.0 {
|
||||
(r.y, r.y + r.height)
|
||||
} else {
|
||||
(r.y + r.height, r.y)
|
||||
};
|
||||
(x_lo, x_hi, y_lo, y_hi)
|
||||
};
|
||||
let page_rects: Vec<(f32, f32, f32, f32)> = rects
|
||||
.iter()
|
||||
.filter(|r| r.page == page && r.height.abs() > 6.0)
|
||||
.map(norm)
|
||||
.collect();
|
||||
let rect_flank = page_rects.iter().any(|&(x_lo, x_hi, y_lo, y_hi)| {
|
||||
let contains = x_lo <= rule.x1 + 2.0
|
||||
&& x_hi >= rule.x2 - 2.0
|
||||
&& y_lo <= rule.y + 2.0
|
||||
&& y_hi >= rule.y - 2.0;
|
||||
if !contains {
|
||||
return false;
|
||||
}
|
||||
// Grid evidence: a vertically abutting neighbor box with x-overlap.
|
||||
page_rects.iter().any(|&(nx_lo, nx_hi, ny_lo, ny_hi)| {
|
||||
let x_overlap = nx_hi.min(x_hi) - nx_lo.max(x_lo);
|
||||
if x_overlap <= 10.0 {
|
||||
return false;
|
||||
}
|
||||
(ny_lo - y_hi).abs() <= 3.0 || (y_lo - ny_hi).abs() <= 3.0
|
||||
})
|
||||
});
|
||||
if rect_flank {
|
||||
return true;
|
||||
}
|
||||
lines.iter().any(|l| {
|
||||
if l.page != page || (l.x1 - l.x2).abs() > 2.0 {
|
||||
return false;
|
||||
}
|
||||
let x = (l.x1 + l.x2) / 2.0;
|
||||
let near_end = (x - rule.x1).abs() <= 6.0 || (x - rule.x2).abs() <= 6.0;
|
||||
if !near_end {
|
||||
return false;
|
||||
}
|
||||
let (y_lo, y_hi) = if l.y1 <= l.y2 {
|
||||
(l.y1, l.y2)
|
||||
} else {
|
||||
(l.y2, l.y1)
|
||||
};
|
||||
y_lo <= rule.y + 2.0 && y_hi >= rule.y - 2.0
|
||||
})
|
||||
}
|
||||
|
||||
fn has_snug_text_owner(rule: &Rule, items: &[TextItem]) -> bool {
|
||||
// Underlines are drawn to the width of the text they decorate, but the
|
||||
// text may be split into several runs (CJK lines mix scripts and font
|
||||
// switches) — so ownership is judged against the UNION of the runs on
|
||||
// the rule's baseline row. Table/form rulings overshoot their row's
|
||||
// text (row separators span cell padding and empty columns), so they
|
||||
// fail either containment or coverage.
|
||||
let matched: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
|
||||
.collect();
|
||||
if matched.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let x1 = matched.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let x2 = matched
|
||||
.iter()
|
||||
.map(|i| i.x + i.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let max_fs = matched.iter().map(|i| i.font_size).fold(0.0, f32::max);
|
||||
let pad = (max_fs * 0.75).max(4.0);
|
||||
if rule.x1 < x1 - pad || rule.x2 > x2 + pad {
|
||||
return false;
|
||||
}
|
||||
let covered: f32 = matched.iter().map(|i| i.width).sum();
|
||||
if covered < rule.width() * 0.6 {
|
||||
return false;
|
||||
}
|
||||
// A table row also unions to the rule's span — but its cells sit apart.
|
||||
// An underlined text line is contiguous runs with word-sized gaps; any
|
||||
// column-sized hole between matched runs means this is a row ruling.
|
||||
let mut sorted = matched;
|
||||
sorted.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
sorted.windows(2).all(|pair| {
|
||||
let gap = pair[1].x - (pair[0].x + pair[0].width);
|
||||
gap <= (max_fs * 2.0).max(12.0)
|
||||
})
|
||||
}
|
||||
|
||||
fn is_repeated_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut y_levels: Vec<f32> = rules
|
||||
.iter()
|
||||
.filter(|other| has_similar_span(rule, other))
|
||||
.map(|other| other.y)
|
||||
.collect();
|
||||
|
||||
y_levels.sort_by(|a, b| a.total_cmp(b));
|
||||
y_levels.dedup_by(|a, b| (*a - *b).abs() <= RULE_Y_DEDUP_EPS);
|
||||
y_levels.len() >= MIN_REPEATED_RULE_LEVELS
|
||||
}
|
||||
|
||||
fn is_segmented_row_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut row_rules: Vec<&Rule> = rules
|
||||
.iter()
|
||||
.filter(|other| (other.y - rule.y).abs() <= RULE_Y_DEDUP_EPS)
|
||||
.collect();
|
||||
|
||||
if row_rules.len() < MIN_SEGMENTED_ROW_RULES {
|
||||
return false;
|
||||
}
|
||||
|
||||
row_rules.sort_by(|a, b| a.x1.total_cmp(&b.x1));
|
||||
let large_gaps = row_rules
|
||||
.windows(2)
|
||||
.filter(|pair| pair[1].x1 - pair[0].x2 > SEGMENTED_ROW_GAP_MIN)
|
||||
.count();
|
||||
|
||||
large_gaps >= MIN_SEGMENTED_ROW_GAPS
|
||||
}
|
||||
|
||||
fn has_similar_span(a: &Rule, b: &Rule) -> bool {
|
||||
let a_width = a.width();
|
||||
let b_width = b.width();
|
||||
if a_width <= 1.0 || b_width <= 1.0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let width_ratio = a_width.max(b_width) / a_width.min(b_width);
|
||||
if width_ratio > RULE_SPAN_WIDTH_RATIO {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = a.x2.min(b.x2) - a.x1.max(b.x1);
|
||||
overlap >= a_width.min(b_width) * RULE_SPAN_OVERLAP_RATIO
|
||||
}
|
||||
|
||||
fn tabular_row_separator_rule_indices(rules: &[Rule], items: &[TextItem]) -> HashSet<usize> {
|
||||
let mut tabular_rules = HashSet::new();
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
let mut matched_items: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
|
||||
.collect();
|
||||
|
||||
if matched_items.len() < MIN_TABULAR_RULE_ITEMS {
|
||||
continue;
|
||||
}
|
||||
|
||||
matched_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let large_gaps = matched_items
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let left = pair[0];
|
||||
let right = pair[1];
|
||||
let gap = right.x - (left.x + left.width);
|
||||
let font_size = left.font_size.max(right.font_size).max(1.0);
|
||||
gap > font_size * TABULAR_RULE_GAP_EM
|
||||
})
|
||||
.count();
|
||||
|
||||
if large_gaps >= MIN_TABULAR_RULE_GAPS {
|
||||
tabular_rules.insert(rule_idx);
|
||||
}
|
||||
}
|
||||
|
||||
tabular_rules
|
||||
}
|
||||
|
||||
fn is_underline_candidate(item: &TextItem) -> bool {
|
||||
matches!(item.item_type, ItemType::Text) && !item.text.trim().is_empty() && item.width > 0.0
|
||||
}
|
||||
|
||||
fn rule_matches_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
// Vertical window: underlines sit at or slightly below the baseline.
|
||||
// Latin fonts draw them at roughly 5-15% of the em below; CJK layouts
|
||||
// put them under the full em box, measured up to ~0.67em below the
|
||||
// baseline (text_dense__underline). Allow 0.72em (min 3pt) below and
|
||||
// 1pt above for rounding.
|
||||
let below = (item.font_size * 0.72).max(3.0);
|
||||
let y_min = item.y - below;
|
||||
let y_max = item.y + 1.0;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let ix1 = item.x;
|
||||
let ix2 = item.x + item.width;
|
||||
let min_overlap = item.width * MIN_X_OVERLAP;
|
||||
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
/// Strikeout window: a rule crossing the glyphs. Strikethroughs sit at
|
||||
/// roughly 20-35% of the em above the baseline (about half the x-height);
|
||||
/// accept a band well inside the glyph body so baseline underlines and
|
||||
/// overlines never qualify.
|
||||
fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
let y_min = item.y + item.font_size * 0.12;
|
||||
let y_max = item.y + item.font_size * 0.55;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let ix1 = item.x;
|
||||
let ix2 = item.x + item.width;
|
||||
let min_overlap = item.width * MIN_X_OVERLAP;
|
||||
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
/// page's extraction output (all in PDF coordinates, y-up, where
|
||||
/// `TextItem::y` is the text baseline).
|
||||
pub(crate) fn mark_underlined_items(
|
||||
items: &mut [TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) {
|
||||
let rules = discard_repeated_ruling_rules(
|
||||
rules_from_graphics(rects, lines, page),
|
||||
items,
|
||||
rects,
|
||||
lines,
|
||||
page,
|
||||
);
|
||||
if rules.is_empty() {
|
||||
return;
|
||||
}
|
||||
let tabular_rules = tabular_row_separator_rule_indices(&rules, items);
|
||||
|
||||
// Math fraction bars are short horizontal lines with the numerator just
|
||||
// above AND the denominator just below — underline geometry from above,
|
||||
// but no underline has text hanging directly beneath it at fraction
|
||||
// distance. Only narrow rules qualify: real underlines under short
|
||||
// labels have their next text line a full line-pitch away.
|
||||
let fraction_rules: HashSet<usize> = rules
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, rule)| {
|
||||
rule.width() <= 60.0
|
||||
&& items.iter().any(|item| {
|
||||
if !is_underline_candidate(item) {
|
||||
return false;
|
||||
}
|
||||
// A denominator HUGS the bar (fraction typesetting
|
||||
// leaves ~0.1-0.2em) and is bar-sized. Both bounds
|
||||
// matter: a short last-line of a paragraph at normal
|
||||
// leading sits further below, and a full next text
|
||||
// line is far wider than the rule.
|
||||
let dy = rule.y - (item.y + item.height);
|
||||
let overlap = rule.x2.min(item.x + item.width) - rule.x1.max(item.x);
|
||||
dy > 0.0
|
||||
&& dy <= item.font_size * 0.3
|
||||
&& overlap > rule.width() * 0.5
|
||||
&& item.width <= rule.width() * 1.5
|
||||
})
|
||||
})
|
||||
.map(|(i, _)| i)
|
||||
.collect();
|
||||
|
||||
for item in items.iter_mut() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
// The fraction guard only gates UNDERLINE marking — a rule that
|
||||
// reads as a fraction bar from below can still legitimately
|
||||
// strike through a line above it.
|
||||
if !fraction_rules.contains(&rule_idx) && rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32, font_size: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: font_size,
|
||||
font: "F1".to_string(),
|
||||
font_size,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn hline(x1: f32, x2: f32, y: f32) -> UnderlineLine {
|
||||
UnderlineLine {
|
||||
x1,
|
||||
y1: y,
|
||||
x2,
|
||||
y2: y,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
fn cell_rect(x: f32, y: f32, width: f32, height: f32) -> PdfRect {
|
||||
PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
fn thin_rect(x: f32, y: f32, width: f32) -> PdfRect {
|
||||
PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 0.8,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stroked_line_under_baseline_marks_underline() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thin_filled_rect_under_baseline_marks_underline() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![thin_rect(100.0, 497.8, 60.0)];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_rule_under_multiple_items_marks_each() {
|
||||
// One underline drawn under a whole sentence: every overlapped
|
||||
// item gets the flag.
|
||||
let mut items = vec![
|
||||
item("first", 100.0, 500.0, 40.0, 10.0),
|
||||
item("second", 145.0, 500.0, 50.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(98.0, 200.0, 498.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
assert!(items[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_far_below_baseline_is_not_an_underline() {
|
||||
// A horizontal rule 30pt below (section divider) must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(90.0, 300.0, 470.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_stroked_line_is_not_an_underline() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let mut line = hline(99.0, 161.0, 498.5);
|
||||
line.stroke_width = 4.0;
|
||||
|
||||
mark_underlined_items(&mut items, &[], &[line], 1);
|
||||
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mid_glyph_rule_marks_strikeout_not_underline() {
|
||||
// Rule at ~30% of the em above the baseline crosses the glyphs.
|
||||
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 503.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_strikeout);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn overline_is_neither_underline_nor_strikeout() {
|
||||
// Rule just above the cap height (overline / next line's rule).
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 507.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thin_filled_rect_at_mid_glyph_marks_strikeout() {
|
||||
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![thin_rect(100.0, 502.6, 60.0)];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_strikeout);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_above_baseline_is_not_an_underline() {
|
||||
// Strikethrough / overline geometry must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(90.0, 300.0, 505.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn insufficient_horizontal_overlap_is_not_an_underline() {
|
||||
// Rule under only a quarter of the item (e.g. neighboring cell
|
||||
// border) must not mark.
|
||||
let mut items = vec![item("wide text item", 100.0, 500.0, 100.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 125.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_width_rect_is_normalized_and_marks_underline() {
|
||||
// A CTM with negative x-scale (or negative `re` operands) produces
|
||||
// rects whose width is negative; the rule extents must normalize.
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 160.0,
|
||||
y: 497.8,
|
||||
width: -60.0,
|
||||
height: 0.8,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_height_band_is_not_an_underline() {
|
||||
// A 14pt band expressed with negative height must not pass the
|
||||
// thickness check via sign trickery.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 95.0,
|
||||
y: 509.0,
|
||||
width: 80.0,
|
||||
height: -14.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_band_is_not_an_underline() {
|
||||
// A highlight bar / filled cell background (tall rect) must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 95.0,
|
||||
y: 495.0,
|
||||
width: 80.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vertical_line_is_not_an_underline() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![UnderlineLine {
|
||||
x1: 120.0,
|
||||
y1: 498.0,
|
||||
x2: 120.0,
|
||||
y2: 400.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn other_pages_graphics_do_not_mark() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let mut line = hline(99.0, 161.0, 498.5);
|
||||
line.page = 2;
|
||||
mark_underlined_items(&mut items, &[], &[line], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_table_row_rules_do_not_mark_cell_text() {
|
||||
let mut items = vec![
|
||||
item("A", 110.0, 500.0, 20.0, 10.0),
|
||||
item("B", 110.0, 480.0, 20.0, 10.0),
|
||||
item("C", 110.0, 460.0, 20.0, 10.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(100.0, 150.0, 498.0),
|
||||
hline(100.0, 150.0, 478.0),
|
||||
hline(100.0, 150.0, 458.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn row_separator_under_spaced_column_labels_is_not_an_underline() {
|
||||
let mut items = vec![
|
||||
item("Date", 100.0, 500.0, 25.0, 10.0),
|
||||
item("Rate", 200.0, 500.0, 25.0, 10.0),
|
||||
item("Yield", 300.0, 500.0, 30.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(90.0, 340.0, 498.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_snug_underlines_survive_ruling_filter() {
|
||||
// Dense docs underline many full-width lines: span-similar rules at
|
||||
// 3+ y-levels used to be discarded wholesale as table rulings.
|
||||
// Each rule here snugly matches one text line, so all must mark.
|
||||
let mut items = vec![
|
||||
item("first underlined line of text", 50.0, 700.0, 300.0, 11.0),
|
||||
item("second underlined line here", 50.0, 650.0, 300.0, 11.0),
|
||||
item("third underlined line as well", 50.0, 600.0, 300.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 350.0, 697.0),
|
||||
hline(50.0, 350.0, 647.0),
|
||||
hline(50.0, 350.0, 597.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rescue_spans_split_runs_on_one_line() {
|
||||
// A single underlined line is often split into several runs (script
|
||||
// or font switches). The union of touching runs owns the rule.
|
||||
let mut items = vec![
|
||||
item("run one", 50.0, 700.0, 100.0, 11.0),
|
||||
item("run two", 150.5, 700.0, 100.0, 11.0),
|
||||
item("run three", 251.0, 700.0, 99.0, 11.0),
|
||||
item("other a", 50.0, 650.0, 300.0, 11.0),
|
||||
item("other b", 50.0, 600.0, 300.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 350.0, 697.0),
|
||||
hline(50.0, 350.0, 647.0),
|
||||
hline(50.0, 350.0, 597.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items[0].is_underline && items[1].is_underline && items[2].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rescue_denied_for_row_with_cell_gaps() {
|
||||
// A full-width rule whose baseline row is several items separated by
|
||||
// column-sized gaps is a table row separator, not an underline —
|
||||
// even when span-similar rules repeat down the page.
|
||||
let mut items = vec![
|
||||
item("cell a", 50.0, 700.0, 60.0, 11.0),
|
||||
item("cell b", 190.0, 700.0, 60.0, 11.0),
|
||||
item("cell c", 330.0, 700.0, 70.0, 11.0),
|
||||
item("cell d", 50.0, 650.0, 60.0, 11.0),
|
||||
item("cell e", 190.0, 650.0, 60.0, 11.0),
|
||||
item("cell f", 330.0, 650.0, 70.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 400.0, 697.0),
|
||||
hline(50.0, 400.0, 647.0),
|
||||
hline(50.0, 400.0, 597.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rescue_denied_inside_cell_box() {
|
||||
// A rule snugly under one text line but enclosed by a drawn cell
|
||||
// box that TILES with vertical neighbors (grid evidence) is a row
|
||||
// ruling of a rect-grid table. Isolated boxes (callout panels) do
|
||||
// not veto — see repeated_snug_underlines_survive_ruling_filter.
|
||||
let mut items = vec![
|
||||
item("one wide cell row", 50.0, 700.0, 300.0, 11.0),
|
||||
item("second wide cell", 50.0, 650.0, 300.0, 11.0),
|
||||
item("third wide cell", 50.0, 600.0, 300.0, 11.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(50.0, 350.0, 697.0),
|
||||
hline(50.0, 350.0, 647.0),
|
||||
hline(50.0, 350.0, 597.0),
|
||||
];
|
||||
let boxes = vec![
|
||||
cell_rect(45.0, 690.0, 320.0, 50.0),
|
||||
cell_rect(45.0, 640.0, 320.0, 50.0),
|
||||
cell_rect(45.0, 590.0, 320.0, 50.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &boxes, &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_row_spaced_rule_segments_do_not_mark_column_labels() {
|
||||
let mut items = vec![
|
||||
item("Date", 100.0, 500.0, 25.0, 10.0),
|
||||
item("Rate", 200.0, 500.0, 25.0, 10.0),
|
||||
item("Yield", 300.0, 500.0, 30.0, 10.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(98.0, 128.0, 498.0),
|
||||
hline(198.0, 228.0, 498.0),
|
||||
hline(298.0, 333.0, 498.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
}
|
||||
+68
-19
@@ -1,5 +1,6 @@
|
||||
//! Form XObject and image XObject extraction.
|
||||
|
||||
use super::fonts::descriptor_style_flags;
|
||||
use crate::text_utils::{effective_font_size, expand_ligatures, is_bold_font, is_italic_font};
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{ItemType, TextItem};
|
||||
@@ -8,9 +9,9 @@ use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache, FontStyleCache,
|
||||
};
|
||||
use super::{get_number, multiply_matrices};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
const MAX_FORM_XOBJECT_DEPTH: u8 = 5;
|
||||
|
||||
@@ -114,6 +115,7 @@ pub(crate) fn extract_form_xobject_text(
|
||||
font_cmaps: &FontCMaps,
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> Vec<TextItem> {
|
||||
extract_form_xobject_text_inner(
|
||||
doc,
|
||||
@@ -122,10 +124,12 @@ pub(crate) fn extract_form_xobject_text(
|
||||
font_cmaps,
|
||||
parent_ctm,
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
0,
|
||||
)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn extract_form_xobject_text_inner(
|
||||
doc: &Document,
|
||||
form_id: ObjectId,
|
||||
@@ -133,6 +137,7 @@ fn extract_form_xobject_text_inner(
|
||||
font_cmaps: &FontCMaps,
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
depth: u8,
|
||||
) -> Vec<TextItem> {
|
||||
use lopdf::content::Content;
|
||||
@@ -157,7 +162,7 @@ fn extract_form_xobject_text_inner(
|
||||
|
||||
// Get fonts from the Form's Resources
|
||||
let form_fonts = get_form_fonts(doc, &stream.dict);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts);
|
||||
let (font_encodings, _has_gid_fonts) = build_font_encodings(doc, &form_fonts, font_cmaps);
|
||||
|
||||
// Build font width info for the form
|
||||
let font_widths = build_font_widths(doc, &form_fonts);
|
||||
@@ -167,6 +172,7 @@ fn extract_form_xobject_text_inner(
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
let mut inline_cmaps: HashMap<String, crate::tounicode::CMapEntry> = HashMap::new();
|
||||
|
||||
let mut font_style_flags: HashMap<String, (bool, bool)> = HashMap::new();
|
||||
for (font_name, font_dict) in &form_fonts {
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
if let Ok(base_font) = font_dict.get(b"BaseFont") {
|
||||
@@ -175,6 +181,10 @@ fn extract_form_xobject_text_inner(
|
||||
font_base_names.insert(resource_name.clone(), base_name);
|
||||
}
|
||||
}
|
||||
let style = descriptor_style_flags(doc, font_dict, style_cache);
|
||||
if style != (false, false) {
|
||||
font_style_flags.insert(resource_name.clone(), style);
|
||||
}
|
||||
match font_dict.get(b"ToUnicode") {
|
||||
Ok(tounicode) => {
|
||||
if let Ok(obj_ref) = tounicode.as_reference() {
|
||||
@@ -262,19 +272,46 @@ fn extract_form_xobject_text_inner(
|
||||
if !op.operands.is_empty() {
|
||||
if let Ok(name) = op.operands[0].as_name() {
|
||||
let xobj_name = String::from_utf8_lossy(name).to_string();
|
||||
if let Some(XObjectType::Form(nested_id)) = form_xobjects.get(&xobj_name) {
|
||||
if depth < MAX_FORM_XOBJECT_DEPTH {
|
||||
let nested_items = extract_form_xobject_text_inner(
|
||||
doc,
|
||||
*nested_id,
|
||||
page_num,
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
cmap_decisions,
|
||||
depth + 1,
|
||||
);
|
||||
items.extend(nested_items);
|
||||
match form_xobjects.get(&xobj_name) {
|
||||
Some(XObjectType::Form(nested_id)) => {
|
||||
if depth < MAX_FORM_XOBJECT_DEPTH {
|
||||
let nested_items = extract_form_xobject_text_inner(
|
||||
doc,
|
||||
*nested_id,
|
||||
page_num,
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
depth + 1,
|
||||
);
|
||||
items.extend(nested_items);
|
||||
}
|
||||
}
|
||||
Some(XObjectType::Image) => {
|
||||
// Mirror the top-level Image-XObject emission
|
||||
// in content_stream.rs so figures embedded
|
||||
// inside Form XObjects (common in print-to-PDF
|
||||
// workflows) aren't silently dropped.
|
||||
let (x, y, width, height) = image_bbox_from_ctm(&ctm);
|
||||
items.push(TextItem {
|
||||
text: format!("[Image: {}]", xobj_name),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
font: String::new(),
|
||||
font_size: 0.0,
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Image,
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
None => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -403,6 +440,10 @@ fn extract_form_xobject_text_inner(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
@@ -412,8 +453,10 @@ fn extract_form_xobject_text_inner(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -534,6 +577,10 @@ fn extract_form_xobject_text_inner(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
|
||||
for (text, start_w, end_w) in &sub_items {
|
||||
let offset_tm = [
|
||||
@@ -560,8 +607,10 @@ fn extract_form_xobject_text_inner(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
+2305
-219
File diff suppressed because it is too large
Load Diff
@@ -130,6 +130,123 @@ pub(crate) fn has_dot_leaders(text: &str) -> bool {
|
||||
dot_groups >= 2
|
||||
}
|
||||
|
||||
/// Detect a table-of-contents entry: a line ending in a page number preceded by
|
||||
/// a dot-leader group (e.g. "Measurement Lab worksheet ... 3"). `has_dot_leaders`
|
||||
/// misses single-group leaders ("..."), but a trailing "<dots> <number>" is a
|
||||
/// strong TOC signal on its own. Such lines must never be promoted to headings.
|
||||
pub(crate) fn is_toc_entry_line(text: &str) -> bool {
|
||||
let trimmed = text.trim_end();
|
||||
let digits = trimmed
|
||||
.chars()
|
||||
.rev()
|
||||
.take_while(|c| c.is_ascii_digit())
|
||||
.count();
|
||||
if digits == 0 || digits > 4 {
|
||||
return false;
|
||||
}
|
||||
let before_number = trimmed[..trimmed.len() - digits].trim_end();
|
||||
let dots = before_number
|
||||
.chars()
|
||||
.rev()
|
||||
.take_while(|c| *c == '.')
|
||||
.count();
|
||||
dots >= 3
|
||||
}
|
||||
|
||||
/// A heading that announces a table of contents ("Contents", "Table of
|
||||
/// Contents"). Lines after it on the same page are ToC entries — section
|
||||
/// titles that look exactly like headings but must not be promoted.
|
||||
pub(crate) fn is_toc_marker_heading(text: &str) -> bool {
|
||||
let t = text.trim().trim_end_matches(':').trim().to_lowercase();
|
||||
matches!(t.as_str(), "contents" | "table of contents")
|
||||
}
|
||||
|
||||
/// Lines that resemble headings structurally but are display-math fragments:
|
||||
/// equations ending in an equation number ("S = kB ln W, (2)") or equation
|
||||
/// lead-ins ("Rearranging Equation (8) gives:"). Both carry an "(N)" equation
|
||||
/// reference — but a trailing "(N)" alone is not enough: real headings end
|
||||
/// with parenthesized numbers too ("Nicaea (325)", appendix numbering), so
|
||||
/// the suffix form additionally requires math evidence — an "=" in the line
|
||||
/// or a comma immediately before the number, both present in every display
|
||||
/// equation and absent from name-plus-number headings. A bare trailing colon
|
||||
/// is NOT a fragment signal either: real headings frequently end with colons
|
||||
/// ("Procedure:", "Steps for Using the Microscope:").
|
||||
pub(crate) fn is_heading_fragment(text: &str) -> bool {
|
||||
let t = text.trim_end();
|
||||
|
||||
// A lowercase-initial one-or-two-word "heading" is a mid-sentence
|
||||
// fragment beside display math ("or inversely", "and therefore") —
|
||||
// real headings that short start uppercase. Measured as spurious
|
||||
// headings on academic docs (fire-pdf ENG-5029 / opendataloader MHS).
|
||||
{
|
||||
let words: Vec<&str> = t.split_whitespace().collect();
|
||||
if words.len() <= 2 {
|
||||
if let Some(first_alpha) = t.chars().find(|c| c.is_alphabetic()) {
|
||||
if first_alpha.is_lowercase() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn is_equation_number(s: &str) -> bool {
|
||||
s.strip_prefix('(')
|
||||
.and_then(|r| r.strip_suffix(')'))
|
||||
.is_some_and(|inner| {
|
||||
!inner.is_empty() && inner.len() <= 3 && inner.chars().all(|c| c.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
// Equation-number suffix with math evidence: "S = kB ln W, (2)"
|
||||
let mut rev = t.rsplit(' ');
|
||||
let last = rev.next().unwrap_or("");
|
||||
if is_equation_number(last) {
|
||||
// Page-of-total running headers: "LIVSMEDELSVERKET PM 2 (10)"
|
||||
if let Some(prev_word) = t.rsplit(' ').nth(1) {
|
||||
if let (Ok(page), Some(total)) = (
|
||||
prev_word.parse::<u32>(),
|
||||
last.trim_start_matches('(')
|
||||
.trim_end_matches(')')
|
||||
.parse::<u32>()
|
||||
.ok(),
|
||||
) {
|
||||
if page <= total {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
let punct_before = rev
|
||||
.next()
|
||||
.is_some_and(|w| w.ends_with(',') || w.ends_with(':'));
|
||||
let has_math_op = t.chars().any(|c| {
|
||||
matches!(
|
||||
c,
|
||||
'=' | '<'
|
||||
| '>'
|
||||
| '≤'
|
||||
| '≥'
|
||||
| '≪'
|
||||
| '≫'
|
||||
| '≈'
|
||||
| '≠'
|
||||
| '±'
|
||||
| '∑'
|
||||
| '∫'
|
||||
| '√'
|
||||
| '∝'
|
||||
)
|
||||
});
|
||||
if punct_before || has_math_op {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// Lead-in: ends with a colon AND references an equation number inline
|
||||
if t.ends_with(':') && t.split_whitespace().any(is_equation_number) {
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Compute the Y-gap threshold for paragraph break detection.
|
||||
///
|
||||
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
|
||||
@@ -257,6 +374,15 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
for line in lines {
|
||||
if let Some(first) = line.items.first() {
|
||||
if first.font_size / base_size >= 1.2 {
|
||||
// Digit-only lines (page numbers, issue numbers) must not
|
||||
// define heading tiers: a large bold folio claims tier 0 and
|
||||
// blocks the bold-size fallback for the document's real
|
||||
// same-size headings.
|
||||
let text = line.text();
|
||||
let t = text.trim();
|
||||
if !t.is_empty() && t.chars().all(|c| !c.is_alphabetic()) {
|
||||
continue;
|
||||
}
|
||||
heading_sizes.push(first.font_size);
|
||||
}
|
||||
}
|
||||
@@ -274,11 +400,45 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
}
|
||||
}
|
||||
|
||||
// Books often set section headings barely above body size (e.g. 11pt
|
||||
// bold over 10pt text). When nothing clears the 1.2x ratio gate, fall
|
||||
// back to bold lines modestly larger than body so those documents still
|
||||
// get an H1 instead of every bold heading defaulting to H2.
|
||||
if tiers.is_empty() {
|
||||
let mut bold_sizes: Vec<f32> = lines
|
||||
.iter()
|
||||
.filter(|line| {
|
||||
let text = line.text();
|
||||
let t = text.trim();
|
||||
!t.is_empty() && t.chars().any(|c| c.is_alphabetic())
|
||||
})
|
||||
.filter_map(|line| line.items.first())
|
||||
.filter(|it| it.is_bold && it.font_size / base_size >= 1.05)
|
||||
.map(|it| it.font_size)
|
||||
.collect();
|
||||
bold_sizes.sort_by(|a, b| b.total_cmp(a));
|
||||
for size in bold_sizes {
|
||||
if !tiers.iter().any(|&t| (t - size).abs() < 0.5) {
|
||||
tiers.push(size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Cap at 4 tiers
|
||||
tiers.truncate(4);
|
||||
tiers
|
||||
}
|
||||
|
||||
/// Boldness of a line judged by character mass, so a heading with an
|
||||
/// unbold section-number prefix ("4. " + bold title) still counts as bold.
|
||||
pub(crate) fn line_is_mostly_bold(line: &TextLine) -> bool {
|
||||
let (bold, total) = line.items.iter().fold((0usize, 0usize), |(b, t), it| {
|
||||
let n = it.text.trim().chars().count();
|
||||
(b + if it.is_bold { n } else { 0 }, t + n)
|
||||
});
|
||||
total > 0 && bold * 2 >= total
|
||||
}
|
||||
|
||||
/// Detect header level from font size using document-specific heading tiers.
|
||||
/// When tiers are available, maps tier 0→H1, tier 1→H2, etc.
|
||||
/// Falls back to ratio-based thresholds when no tiers exist.
|
||||
@@ -286,9 +446,21 @@ pub(crate) fn detect_header_level(
|
||||
font_size: f32,
|
||||
base_size: f32,
|
||||
heading_tiers: &[f32],
|
||||
is_bold: bool,
|
||||
) -> Option<usize> {
|
||||
let ratio = font_size / base_size;
|
||||
|
||||
// Tier matches are trusted below the 1.2x gate (down to 1.05x) only for
|
||||
// bold lines: sub-gate tiers come from the bold fallback, and honoring
|
||||
// them for non-bold text at the same size would promote captions.
|
||||
if (1.05..1.2).contains(&ratio) && is_bold && !heading_tiers.is_empty() {
|
||||
for (i, &tier_size) in heading_tiers.iter().enumerate() {
|
||||
if (font_size - tier_size).abs() < 0.5 {
|
||||
return Some(i + 1); // tier 0 → H1, tier 1 → H2, etc.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if ratio < 1.2 {
|
||||
return None; // Regular text
|
||||
}
|
||||
@@ -320,3 +492,138 @@ pub(crate) fn detect_header_level(
|
||||
Some(4)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn line_of(text: &str, font_size: f32, bold: bool, y: f32) -> crate::types::TextLine {
|
||||
let item = crate::types::TextItem {
|
||||
text: text.into(),
|
||||
x: 72.0,
|
||||
y,
|
||||
width: text.len() as f32 * font_size * 0.5,
|
||||
height: font_size,
|
||||
font: "Test".into(),
|
||||
font_size,
|
||||
page: 1,
|
||||
is_bold: bold,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
};
|
||||
crate::types::TextLine {
|
||||
items: vec![item],
|
||||
y,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn digit_only_lines_do_not_define_tiers() {
|
||||
// A 14pt bold page number must not claim tier 0 — that both demotes
|
||||
// every real heading a level and blocks the bold-size fallback.
|
||||
let lines = vec![
|
||||
line_of("76", 14.0, true, 760.0),
|
||||
line_of("Replace", 11.0, true, 700.0),
|
||||
line_of("body text at eleven points", 11.0, false, 680.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 11.0);
|
||||
assert!(tiers.is_empty(), "page number claimed a tier: {tiers:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_fallback_tiers_when_nothing_clears_ratio_gate() {
|
||||
// 10pt body, 11pt bold section headings (book-style): no size clears
|
||||
// 1.2x, so bold sizes modestly above body form the tiers.
|
||||
let lines = vec![
|
||||
line_of("4. Entropy", 11.0, true, 700.0),
|
||||
line_of("body text about entropy", 10.0, false, 680.0),
|
||||
line_of("5. The dynamics", 11.0, true, 500.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 10.0);
|
||||
assert_eq!(tiers, vec![11.0]);
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, true), Some(1));
|
||||
// Non-bold text at the fallback size must not become a heading.
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, false), None);
|
||||
// Non-tier body text stays regular.
|
||||
assert_eq!(detect_header_level(10.0, 10.0, &tiers, true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_fallback_skipped_when_real_tiers_exist() {
|
||||
let lines = vec![
|
||||
line_of("Chapter One", 18.0, false, 700.0),
|
||||
line_of("bold label", 11.0, true, 600.0),
|
||||
line_of("body", 10.0, false, 580.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 10.0);
|
||||
assert_eq!(tiers, vec![18.0]);
|
||||
// The 11pt bold label does not match any tier and stays non-heading.
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn toc_entry_with_single_dot_group() {
|
||||
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
|
||||
assert!(is_toc_entry_line("Results ........ 12"));
|
||||
assert!(is_toc_entry_line("Appendix B...42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_toc_lines_pass() {
|
||||
assert!(!is_toc_entry_line(
|
||||
"6.2. Expectations for Re-Hiring Employees"
|
||||
));
|
||||
assert!(!is_toc_entry_line("What happened in 2020"));
|
||||
assert!(!is_toc_entry_line("IMPLEMENTATION"));
|
||||
// Ellipsis without a trailing page number
|
||||
assert!(!is_toc_entry_line("and so it goes ..."));
|
||||
// Long numbers are data, not page refs
|
||||
assert!(!is_toc_entry_line("ISBN ... 97814"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn toc_marker_headings() {
|
||||
assert!(is_toc_marker_heading("Contents"));
|
||||
assert!(is_toc_marker_heading("CONTENTS"));
|
||||
assert!(is_toc_marker_heading("Table of Contents"));
|
||||
assert!(is_toc_marker_heading("Table of contents:"));
|
||||
assert!(!is_toc_marker_heading("Contents of the Shipment"));
|
||||
assert!(!is_toc_marker_heading("Introduction"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn heading_fragments() {
|
||||
// Equation lead-ins: colon ending + inline equation reference
|
||||
assert!(is_heading_fragment("or inversely"));
|
||||
assert!(is_heading_fragment("and therefore"));
|
||||
assert!(!is_heading_fragment("Introduction"));
|
||||
assert!(!is_heading_fragment("iPhone Sales Strategy Overview")); // 4 words, exempt
|
||||
assert!(is_heading_fragment("Rearranging Equation (8) gives:"));
|
||||
// Display-equation neighbours ending in an equation number
|
||||
assert!(is_heading_fragment("S = kB ln W, (2)"));
|
||||
assert!(is_heading_fragment("E = mc2 (12)"));
|
||||
assert!(is_heading_fragment("x + y = z, (3)"));
|
||||
// Page-of-total running headers
|
||||
assert!(is_heading_fragment("LIVSMEDELSVERKET PM 2 (10)"));
|
||||
// Comparison-operator evidence and colon-before-number
|
||||
assert!(is_heading_fragment(
|
||||
"PLL\u{fe} PHH\u{226a} PLH\u{fe} PHL: (12)"
|
||||
));
|
||||
// Real headings pass — including name-plus-number and colon-ended ones
|
||||
assert!(!is_heading_fragment("Nicaea (325)"));
|
||||
assert!(!is_heading_fragment(
|
||||
"\u{627}\u{644}\u{645}\u{644}\u{62d}\u{642} \u{631}\u{642}\u{645} (1)"
|
||||
));
|
||||
assert!(!is_heading_fragment("4. Entropy"));
|
||||
assert!(!is_heading_fragment("Procedure:"));
|
||||
assert!(!is_heading_fragment("Steps for Using the Microscope:"));
|
||||
assert!(!is_heading_fragment("Changing objectives:"));
|
||||
assert!(!is_heading_fragment("Sales by Region (2024)"));
|
||||
assert!(!is_heading_fragment("Results (preliminary)"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -131,10 +131,11 @@ pub(crate) fn format_list_item(text: &str) -> String {
|
||||
if let Some(rest) = trimmed.strip_prefix(*bullet) {
|
||||
return format!("- {}", rest.trim_start());
|
||||
}
|
||||
// Bullet inside a leading bold/italic run (e.g. "**● Label:** rest").
|
||||
// The run wraps both the marker and the following label because both
|
||||
// use a bold font in the PDF.
|
||||
for wrapper in ["**", "*"] {
|
||||
// Bullet inside a leading style run (e.g. "**● Label:** rest" or
|
||||
// "<u>● Label</u>"). The run wraps both the marker and the following
|
||||
// label because both carry the style in the PDF. The marker must move
|
||||
// outside the wrapper so markdown still sees a list item.
|
||||
for wrapper in ["**", "*", "<u>"] {
|
||||
if let Some(after_open) = trimmed.strip_prefix(wrapper) {
|
||||
if let Some(rest) = after_open.strip_prefix(*bullet) {
|
||||
return format!("- {}{}", wrapper, rest.trim_start());
|
||||
@@ -235,6 +236,13 @@ mod tests {
|
||||
assert_eq!(format_list_item("• Item"), "- Item");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_underline() {
|
||||
// Fully-underlined bullet line: the marker must move outside the
|
||||
// <u> wrapper so markdown still renders a list item.
|
||||
assert_eq!(format_list_item("<u>● Item text</u>"), "- <u>Item text</u>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_bold() {
|
||||
// PDF that uses bold font for both the marker and the label produces
|
||||
|
||||
+840
-98
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+1238
-86
File diff suppressed because it is too large
Load Diff
+107
-3
@@ -2,12 +2,15 @@
|
||||
|
||||
use regex::Regex;
|
||||
|
||||
use super::MarkdownOptions;
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
// Collapse dot leaders (e.g. TOC entries: "Introduction...............................1")
|
||||
text = collapse_dot_leaders(&text);
|
||||
if options.profile == MarkdownProfile::Compact {
|
||||
// Dot-leader collapse saves tokens but changes source text, so it is
|
||||
// reserved for the explicit compact profile.
|
||||
text = collapse_dot_leaders(&text);
|
||||
}
|
||||
|
||||
// Fix hyphenation first (before other processing)
|
||||
if options.fix_hyphenation {
|
||||
@@ -29,6 +32,8 @@ pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> Str
|
||||
// text item, which combine with gap-based space insertion to produce
|
||||
// double spaces ("Vice President" instead of "Vice President").
|
||||
collapse_consecutive_spaces(&mut text);
|
||||
remove_spaces_before_closing_brackets(&mut text);
|
||||
remove_spaces_before_sentence_punctuation(&mut text);
|
||||
|
||||
// Remove excessive newlines (more than 2 in a row)
|
||||
while text.contains("\n\n\n") {
|
||||
@@ -71,6 +76,46 @@ fn collapse_consecutive_spaces(text: &mut String) {
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Remove spaces before closing square brackets.
|
||||
/// Unit markers and markdown links occasionally pick up a gap-inserted space
|
||||
/// before `]` (e.g. `[kg/m3 ]`), which is cosmetic padding.
|
||||
fn remove_spaces_before_closing_brackets(text: &mut String) {
|
||||
let mut result = String::with_capacity(text.len());
|
||||
for ch in text.chars() {
|
||||
if ch == ']' && result.ends_with(' ') {
|
||||
result.pop();
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Remove a stray space before sentence punctuation ("word ." → "word.").
|
||||
/// Style-boundary item splits (bold/italic/underline runs) can strand a
|
||||
/// trailing period or comma in its own fragment, and several assembly paths
|
||||
/// join fragments with spaces. Only fires when the punctuation ends the
|
||||
/// token (followed by whitespace or end of text), so decimals ("3 .14" stays
|
||||
/// untouched — no such input exists, but the guard is cheap) and dot leaders
|
||||
/// (" ... ") are unaffected.
|
||||
fn remove_spaces_before_sentence_punctuation(text: &mut String) {
|
||||
let chars: Vec<char> = text.chars().collect();
|
||||
let mut result = String::with_capacity(text.len());
|
||||
for (i, &ch) in chars.iter().enumerate() {
|
||||
if matches!(ch, '.' | ',' | ';') && result.ends_with(' ') {
|
||||
let next = chars.get(i + 1);
|
||||
// `|` counts as a token end so table cells get the same fix.
|
||||
let token_ends = next.is_none_or(|c| c.is_whitespace() || *c == '|');
|
||||
// Never touch runs of dots (ellipsis / dot leaders).
|
||||
let in_dot_run = ch == '.' && next == Some(&'.');
|
||||
if token_ends && !in_dot_run {
|
||||
result.pop();
|
||||
}
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Collapse dot leaders (runs of 4+ dots) into " ... "
|
||||
/// Common in tables of contents: "Introduction...............................1" -> "Introduction ... 1"
|
||||
fn collapse_dot_leaders(text: &str) -> String {
|
||||
@@ -313,6 +358,23 @@ fn format_urls(text: &str) -> String {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn fidelity_profile_preserves_dot_leaders() {
|
||||
let input = "Introduction............................1".to_string();
|
||||
let result = clean_markdown(input.clone(), &MarkdownOptions::default());
|
||||
assert_eq!(result, format!("{input}\n"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compact_profile_collapses_dot_leaders() {
|
||||
let input = "Introduction............................1".to_string();
|
||||
let options = MarkdownOptions {
|
||||
profile: MarkdownProfile::Compact,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
assert_eq!(clean_markdown(input, &options), "Introduction ... 1\n");
|
||||
}
|
||||
|
||||
// --- collapse_dot_leaders ---
|
||||
|
||||
#[test]
|
||||
@@ -342,6 +404,48 @@ mod tests {
|
||||
assert!(result.contains("Chapter 2 ... 20"));
|
||||
}
|
||||
|
||||
// --- remove_spaces_before_closing_brackets ---
|
||||
|
||||
#[test]
|
||||
fn test_remove_spaces_before_closing_brackets() {
|
||||
let mut input = "Density [kg/m3 ] and [linked text ](https://example.com)".to_string();
|
||||
remove_spaces_before_closing_brackets(&mut input);
|
||||
assert_eq!(
|
||||
input,
|
||||
"Density [kg/m3] and [linked text](https://example.com)"
|
||||
);
|
||||
}
|
||||
|
||||
// --- remove_spaces_before_sentence_punctuation ---
|
||||
|
||||
#[test]
|
||||
fn strips_space_before_trailing_period() {
|
||||
let mut t = "Foreign insurance companies . The provisions".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "Foreign insurance companies. The provisions");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_space_before_period_at_cell_boundary() {
|
||||
let mut t = "|Applicability date .|This section|".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "|Applicability date.|This section|");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_dot_leaders_and_ellipses() {
|
||||
let mut t = "Introduction ... 1".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "Introduction ... 1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_mid_token_periods() {
|
||||
let mut t = "version 3 .14 released".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "version 3 .14 released");
|
||||
}
|
||||
|
||||
// --- fix_hyphenation ---
|
||||
|
||||
#[test]
|
||||
|
||||
+108
-2
@@ -3,7 +3,7 @@
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use crate::structure_tree::StructRole;
|
||||
use crate::types::TextLine;
|
||||
use crate::types::{TextItem, TextLine};
|
||||
|
||||
use super::analysis::detect_header_level;
|
||||
|
||||
@@ -42,7 +42,12 @@ fn effective_heading_level(
|
||||
|
||||
// Fall back to font-size heuristic
|
||||
let font = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(font, base_size, heading_tiers)
|
||||
detect_header_level(
|
||||
font,
|
||||
base_size,
|
||||
heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
}
|
||||
|
||||
/// Merge consecutive heading lines at the same level into a single line.
|
||||
@@ -87,6 +92,41 @@ pub(crate) fn merge_heading_lines(
|
||||
false
|
||||
};
|
||||
|
||||
// Bold headings at body font size never reach a tier, so wrapped ones
|
||||
// split into two output headings ("…of wood pellets and cost" /
|
||||
// "structure in Japan"). Merge a fully-bold line into the previous
|
||||
// fully-bold line when it reads as a wrap continuation: starts
|
||||
// lowercase, tiny Y gap, and the previous line has no terminal
|
||||
// punctuation. Kept deliberately narrow — bold list labels and bold
|
||||
// sentences start with markers or capitals and are unaffected.
|
||||
let should_merge = should_merge
|
||||
|| if let Some(prev) = result.last() {
|
||||
let all_bold = |l: &TextLine| {
|
||||
!l.items.is_empty() && l.items.iter().all(|i: &TextItem| i.is_bold)
|
||||
};
|
||||
let prev_text = prev.text();
|
||||
let prev_trim = prev_text.trim_end();
|
||||
let curr_text = line.text();
|
||||
let curr_trim = curr_text.trim();
|
||||
let y_gap = prev.y - line.y;
|
||||
// Both lines must be tier-less: a tiered/tagged bold heading
|
||||
// followed by bold body text must not absorb it.
|
||||
line_level.is_none()
|
||||
&& effective_heading_level(prev, base_size, heading_tiers, struct_roles)
|
||||
.is_none()
|
||||
&& prev.page == line.page
|
||||
&& all_bold(prev)
|
||||
&& all_bold(&line)
|
||||
&& y_gap > 0.0
|
||||
&& y_gap < line_font * 1.6
|
||||
&& curr_trim.chars().next().is_some_and(|c| c.is_lowercase())
|
||||
&& !prev_trim.ends_with(['.', ':', ';', '!', '?'])
|
||||
&& prev_trim.split_whitespace().count() + curr_trim.split_whitespace().count()
|
||||
<= 20
|
||||
} else {
|
||||
false
|
||||
};
|
||||
|
||||
if should_merge {
|
||||
// Append this line's items to the previous line
|
||||
let prev = result.last_mut().unwrap();
|
||||
@@ -542,6 +582,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
@@ -683,4 +725,68 @@ mod tests {
|
||||
.unwrap();
|
||||
assert_eq!(first_header.page, 1, "first occurrence should be on page 1");
|
||||
}
|
||||
|
||||
fn make_bold_line(text: &str, page: u32, y: f32) -> TextLine {
|
||||
let mut item = make_item(text, 12.0, None);
|
||||
item.is_bold = true;
|
||||
TextLine {
|
||||
items: vec![item],
|
||||
y,
|
||||
page,
|
||||
adaptive_threshold: 0.10,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_wrapped_bold_heading_lowercase_continuation() {
|
||||
// Bold-at-body-size heading wrapped across two lines: the second line
|
||||
// starts lowercase and must merge into the first.
|
||||
let lines = vec![
|
||||
make_bold_line(
|
||||
"3. Perspective of supply and demand balance and cost",
|
||||
1,
|
||||
700.0,
|
||||
),
|
||||
make_bold_line("structure in Japan", 1, 686.0),
|
||||
make_line("Body text paragraph follows here.", 12.0, 1, 660.0, None),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||
assert_eq!(result.len(), 2, "wrapped bold heading should merge");
|
||||
assert!(result[0].text().contains("cost structure in Japan"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_merge_for_bold_sentences_or_new_headings() {
|
||||
// Second bold line starts with a capital — a new heading or label,
|
||||
// not a wrap continuation.
|
||||
let lines = vec![
|
||||
make_bold_line("Replace", 1, 700.0),
|
||||
make_bold_line("Trash", 1, 686.0),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||
assert_eq!(result.len(), 2, "distinct bold lines must not merge");
|
||||
|
||||
// Previous line ends a sentence — continuation must not merge.
|
||||
let lines = vec![
|
||||
make_bold_line("This is a bold sentence.", 1, 700.0),
|
||||
make_bold_line("another bold line", 1, 686.0),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||
assert_eq!(result.len(), 2, "sentence-final bold line must not merge");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiered_bold_heading_does_not_absorb_bold_body() {
|
||||
// Previous line is a tier-level bold heading (16pt vs 12pt body);
|
||||
// a following lowercase bold body line must NOT merge into it.
|
||||
let mut heading = make_bold_line("Section Title", 1, 700.0);
|
||||
heading.items[0].font_size = 16.0;
|
||||
heading.items[0].height = 16.0;
|
||||
let lines = vec![
|
||||
heading,
|
||||
make_bold_line("emphasized body text continues here", 1, 686.0),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[16.0], None);
|
||||
assert_eq!(result.len(), 2, "tiered heading must not absorb bold body");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,6 +30,9 @@ pub struct PyPdfResult {
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
/// Title from PDF metadata.
|
||||
#[pyo3(get)]
|
||||
pub title: Option<String>,
|
||||
@@ -60,6 +63,28 @@ impl PyPdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reasons for a single 1-indexed page.
|
||||
#[pyclass(name = "PageOcrReasons")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageOcrReasons {
|
||||
/// 1-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Machine-readable OCR reason identifiers.
|
||||
#[pyo3(get)]
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageOcrReasons {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageOcrReasons(page={}, reasons={:?})",
|
||||
self.page, self.reasons
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classification wrapper (lightweight)
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -106,6 +131,9 @@ pub struct PyRegionText {
|
||||
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -160,6 +188,9 @@ pub struct PyPageMarkdown {
|
||||
/// encoding issues, garbage text, or empty extraction).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -190,6 +221,9 @@ pub struct PyPagesExtractionResult {
|
||||
/// 1-indexed pages that need OCR (scanned/image-based or unreliable text).
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
/// True if any page has tables or columns.
|
||||
#[pyo3(get)]
|
||||
pub is_complex: bool,
|
||||
@@ -232,6 +266,10 @@ pub struct PyTextItem {
|
||||
#[pyo3(get)]
|
||||
pub is_italic: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_underline: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_strikeout: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
}
|
||||
|
||||
@@ -268,6 +306,7 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
title: r.title,
|
||||
confidence: r.confidence,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
@@ -277,6 +316,16 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_page_ocr_reasons(reasons: Vec<crate::PageOcrReasons>) -> Vec<PyPageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PyPageOcrReasons {
|
||||
page: reason.page,
|
||||
reasons: reason.reasons,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_py_err(e: crate::PdfError) -> PyErr {
|
||||
PyValueError::new_err(e.to_string())
|
||||
}
|
||||
@@ -304,6 +353,8 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
})
|
||||
.collect()
|
||||
@@ -350,11 +401,13 @@ fn to_py_pages_result(r: crate::PagesExtractionResult) -> PyPagesExtractionResul
|
||||
page: p.page,
|
||||
markdown: p.markdown,
|
||||
needs_ocr: p.needs_ocr,
|
||||
ocr_reason: p.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: r.pages_with_tables,
|
||||
pages_with_columns: r.pages_with_columns,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
is_complex: r.is_complex,
|
||||
}
|
||||
}
|
||||
@@ -370,6 +423,7 @@ fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRe
|
||||
.map(|r| PyRegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
@@ -563,6 +617,7 @@ fn extract_pages_markdown_bytes(
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPdfResult>()?;
|
||||
m.add_class::<PyPageOcrReasons>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
|
||||
@@ -76,6 +76,52 @@ pub enum StructRole {
|
||||
}
|
||||
|
||||
impl StructRole {
|
||||
/// Content roles whose text must never be promoted to a heading by the
|
||||
/// visual heuristic. These carry an explicit non-heading meaning in the
|
||||
/// struct tree (lists, quotes, notes, references, captions, formulas,
|
||||
/// forms, ToC entries), yet their text is often short and visually
|
||||
/// isolated — exactly what the heuristic keys on. Heading roles (H, H1–H6)
|
||||
/// and generic container/flow roles (P, Div, Sect, Span, …) are excluded
|
||||
/// so the heuristic can still fire there.
|
||||
///
|
||||
/// `Figure` is deliberately NOT in this set: cover/banner pages routinely
|
||||
/// tag the document title inside a Figure (alongside a seal or logo), and
|
||||
/// that title is a real heading. `Formula` and `Form` stay — a line
|
||||
/// explicitly tagged as an equation or form field is never a heading.
|
||||
///
|
||||
/// Table roles (Table/TR/TH/TD/THead/TBody/TFoot) are included so that
|
||||
/// when table reconstruction falls back and cells reach the line loop as
|
||||
/// plain text, a short isolated cell — a `TH` column header especially —
|
||||
/// is not promoted to a heading.
|
||||
pub(crate) fn is_non_heading_content(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Self::L
|
||||
| Self::LI
|
||||
| Self::Lbl
|
||||
| Self::LBody
|
||||
| Self::BlockQuote
|
||||
| Self::Quote
|
||||
| Self::Caption
|
||||
| Self::TOC
|
||||
| Self::TOCI
|
||||
| Self::Index
|
||||
| Self::Note
|
||||
| Self::Reference
|
||||
| Self::BibEntry
|
||||
| Self::Code
|
||||
| Self::Formula
|
||||
| Self::Form
|
||||
| Self::Table
|
||||
| Self::TR
|
||||
| Self::TH
|
||||
| Self::TD
|
||||
| Self::THead
|
||||
| Self::TBody
|
||||
| Self::TFoot
|
||||
)
|
||||
}
|
||||
|
||||
fn from_name(name: &str) -> Self {
|
||||
match name {
|
||||
"Document" => Self::Document,
|
||||
@@ -856,6 +902,54 @@ fn contains_bytes(haystack: &[u8], needle: &[u8]) -> bool {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn non_heading_content_roles() {
|
||||
for r in [
|
||||
StructRole::L,
|
||||
StructRole::LI,
|
||||
StructRole::BlockQuote,
|
||||
StructRole::Quote,
|
||||
StructRole::Caption,
|
||||
StructRole::TOC,
|
||||
StructRole::TOCI,
|
||||
StructRole::Index,
|
||||
StructRole::Note,
|
||||
StructRole::Reference,
|
||||
StructRole::BibEntry,
|
||||
StructRole::Code,
|
||||
StructRole::Formula,
|
||||
StructRole::Form,
|
||||
StructRole::Table,
|
||||
StructRole::TR,
|
||||
StructRole::TH,
|
||||
StructRole::TD,
|
||||
StructRole::THead,
|
||||
StructRole::TBody,
|
||||
StructRole::TFoot,
|
||||
] {
|
||||
assert!(
|
||||
r.is_non_heading_content(),
|
||||
"{r:?} should block heading promotion"
|
||||
);
|
||||
}
|
||||
// Heading and generic container/flow roles must NOT block promotion
|
||||
for r in [
|
||||
StructRole::H,
|
||||
StructRole::H1,
|
||||
StructRole::H3,
|
||||
StructRole::P,
|
||||
StructRole::Div,
|
||||
StructRole::Sect,
|
||||
StructRole::Span,
|
||||
StructRole::Figure,
|
||||
] {
|
||||
assert!(
|
||||
!r.is_non_heading_content(),
|
||||
"{r:?} should allow heading promotion"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_struct_role_from_name() {
|
||||
assert_eq!(StructRole::from_name("H1"), StructRole::H1);
|
||||
|
||||
@@ -104,6 +104,8 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
page: first_item.page,
|
||||
is_bold: first_item.is_bold,
|
||||
is_italic: first_item.is_italic,
|
||||
is_underline: first_item.is_underline,
|
||||
is_strikeout: first_item.is_strikeout,
|
||||
item_type: first_item.item_type.clone(),
|
||||
mcid: first_item.mcid,
|
||||
});
|
||||
@@ -912,7 +914,126 @@ fn looks_like_number(s: &str) -> bool {
|
||||
///
|
||||
/// Used by format.rs to render TOCs as flat lists instead of markdown tables.
|
||||
pub fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells)
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells) || is_page_number_toc(cells)
|
||||
}
|
||||
|
||||
/// Parse a page-number-like token: a short arabic integer (≤4 digits) or a
|
||||
/// canonical roman numeral (front-matter pages: i, ii, …, xxxviii). Roman
|
||||
/// parsing is shared with the formatter via `super::canonical_roman_value` so
|
||||
/// the two stay in sync.
|
||||
fn page_number_value(token: &str) -> Option<u32> {
|
||||
let t = token.trim();
|
||||
if t.is_empty() {
|
||||
return None;
|
||||
}
|
||||
if t.chars().all(|c| c.is_ascii_digit()) && t.len() <= 4 {
|
||||
return t.parse().ok();
|
||||
}
|
||||
super::canonical_roman_value(t)
|
||||
}
|
||||
|
||||
/// Page-number-column TOC: title-based contents with no dot leaders and no
|
||||
/// section numbers (e.g. "About the Publisher vii", "Experiment #1 … 3").
|
||||
/// The signature is a text-title first column and a last column that is almost
|
||||
/// entirely page numbers whose values are *mostly non-decreasing* — the
|
||||
/// monotonic run is what separates a real TOC from an incidental 2-column
|
||||
/// numeric data table.
|
||||
pub(super) fn is_page_number_toc(cells: &[Vec<String>]) -> bool {
|
||||
let num_cols = cells.first().map(|r| r.len()).unwrap_or(0);
|
||||
// A page-number TOC is a narrow list (title + page, optionally a leader
|
||||
// column). Wider grids are data tables, not contents.
|
||||
if !(2..=3).contains(&num_cols) || cells.len() < 5 {
|
||||
return false;
|
||||
}
|
||||
let last = num_cols - 1;
|
||||
|
||||
// No header row: a TOC's first row is already an entry, so its last cell is
|
||||
// a page number. A data table's first row is a column header (non-numeric,
|
||||
// or an empty units cell like "Category | ") — the tell that separates
|
||||
// "Mineral | CEC" tables from real contents. Check the actual first row,
|
||||
// not the first non-empty one, so a blank header cell still rejects.
|
||||
let first_last = cells[0].get(last).map(|s| s.trim()).unwrap_or("");
|
||||
if page_number_value(first_last).is_none() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Last column: page numbers on ≥70% of filled rows; collect their values.
|
||||
let mut filled = 0u32;
|
||||
let mut page_vals: Vec<u32> = Vec::new();
|
||||
for row in cells {
|
||||
let cell = row.get(last).map(|s| s.trim()).unwrap_or("");
|
||||
if cell.is_empty() {
|
||||
continue;
|
||||
}
|
||||
filled += 1;
|
||||
if let Some(v) = page_number_value(cell) {
|
||||
page_vals.push(v);
|
||||
}
|
||||
}
|
||||
if filled < 4 || (page_vals.len() as f32) < 0.7 * filled as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// First column: mostly text titles (has alphabetic content). This rejects
|
||||
// numeric-vs-numeric grids.
|
||||
let text_first = cells
|
||||
.iter()
|
||||
.filter(|row| {
|
||||
row.first()
|
||||
.is_some_and(|c| c.chars().any(|ch| ch.is_alphabetic()))
|
||||
})
|
||||
.count();
|
||||
if (text_first as f32) < 0.6 * cells.len() as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Page numbers mostly ascend (allow front-matter→body resets and noise).
|
||||
if page_vals.len() < 2 {
|
||||
return false;
|
||||
}
|
||||
let non_decreasing = page_vals.windows(2).filter(|w| w[1] >= w[0]).count();
|
||||
if (non_decreasing as f32) < 0.7 * (page_vals.len() - 1) as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Stronger TOC signal. Real page numbers SPAN the document — entries skip
|
||||
// (3, 6, 13, 24, …) so their range exceeds the entry count. A rank / ID /
|
||||
// ordinal column is instead a *perfectly dense* consecutive run (1,2,3,… or
|
||||
// 100,101,102,…). Accept anything with page gaps; for a dense run — which a
|
||||
// one-page-per-entry TOC can also produce — fall back to a title signal:
|
||||
// real contents entries are multi-word headings, rank labels are short.
|
||||
let min = *page_vals.iter().min().unwrap();
|
||||
let max = *page_vals.iter().max().unwrap();
|
||||
let span = max.saturating_sub(min);
|
||||
if span > page_vals.len() as u32 {
|
||||
return true;
|
||||
}
|
||||
let dense_consecutive = (span as usize) + 1 == page_vals.len() && {
|
||||
let mut sorted = page_vals.clone();
|
||||
sorted.sort_unstable();
|
||||
sorted.dedup();
|
||||
sorted.len() == page_vals.len()
|
||||
};
|
||||
if !dense_consecutive {
|
||||
// Narrow range but with a gap or repeat — still contents-like.
|
||||
return true;
|
||||
}
|
||||
// Dense counter: only a TOC if the titles read like headings, not the
|
||||
// short single-word labels typical of rank/leaderboard/ID tables.
|
||||
let (total_words, titled_rows) = cells
|
||||
.iter()
|
||||
.filter_map(|row| row.first())
|
||||
.filter(|c| c.chars().any(|ch| ch.is_alphabetic()))
|
||||
.fold((0usize, 0usize), |(w, n), c| {
|
||||
(
|
||||
w + c
|
||||
.split_whitespace()
|
||||
.filter(|t| t.chars().any(|ch| ch.is_alphabetic()))
|
||||
.count(),
|
||||
n + 1,
|
||||
)
|
||||
});
|
||||
titled_rows > 0 && (total_words as f32) / titled_rows as f32 >= 1.8
|
||||
}
|
||||
|
||||
/// Dot-leader TOC: any "Chapter 1 ........ 42" style with explicit leader
|
||||
@@ -1884,4 +2005,166 @@ mod tests {
|
||||
assert!(!starts_with_section_number(""));
|
||||
assert!(!starts_with_section_number("Hello world"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_value_rejects_roman_lookalike_words() {
|
||||
// Ordinary words made only of {i,v,x,l,c} are not page numbers.
|
||||
assert!(page_number_value("civil").is_none());
|
||||
assert!(page_number_value("mix").is_none());
|
||||
assert!(page_number_value("ill").is_none());
|
||||
assert!(page_number_value("lil").is_none());
|
||||
// Canonical roman numerals still parse.
|
||||
assert_eq!(page_number_value("vii"), Some(7));
|
||||
assert_eq!(page_number_value("ix"), Some(9));
|
||||
assert_eq!(page_number_value("xii"), Some(12));
|
||||
assert_eq!(page_number_value("42"), Some(42));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_matches_consecutive_pages_with_titles() {
|
||||
// A short chapter-per-page contents: pages are a dense 1..n run, but
|
||||
// the multi-word titles mark it as a real TOC (recovered by the title
|
||||
// signal rather than rejected for lacking page gaps).
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Introduction to the Study".into(), "1".into()],
|
||||
vec!["Materials and Methods".into(), "2".into()],
|
||||
vec!["Results and Discussion".into(), "3".into()],
|
||||
vec!["Summary of Findings".into(), "4".into()],
|
||||
vec!["References and Notes".into(), "5".into()],
|
||||
];
|
||||
assert!(is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_dense_ordinal_column() {
|
||||
// Headerless title | rank table: values are a consecutive 1..n
|
||||
// sequence (monotonic, no header, text first column) but their range
|
||||
// ~= the row count, so it is data, not a table of contents.
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Alice".into(), "1".into()],
|
||||
vec!["Bob".into(), "2".into()],
|
||||
vec!["Carol".into(), "3".into()],
|
||||
vec!["Dave".into(), "4".into()],
|
||||
vec!["Erin".into(), "5".into()],
|
||||
vec!["Frank".into(), "6".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_blank_header_cell() {
|
||||
// First row is a header whose last cell is blank ("Category | ");
|
||||
// must not be flattened even though later rows look TOC-like.
|
||||
let cells = vec![
|
||||
vec!["Category".into(), "".into()],
|
||||
vec!["Alpha".into(), "3".into()],
|
||||
vec!["Beta".into(), "9".into()],
|
||||
vec!["Gamma".into(), "14".into()],
|
||||
vec!["Delta".into(), "20".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_matches_title_based_contents() {
|
||||
// Title-left, page-number-right, no dot leaders, no section numbers.
|
||||
let cells = vec![
|
||||
vec!["About the Publisher".into(), "vii".into()],
|
||||
vec!["About This Project".into(), "ix".into()],
|
||||
vec!["Acknowledgments".into(), "xi".into()],
|
||||
vec!["Experiment #1: Hydrostatic Pressure".into(), "3".into()],
|
||||
vec!["Experiment #2: Bernoulli's Theorem".into(), "13".into()],
|
||||
vec![
|
||||
"Experiment #3: Energy Loss in Pipe Fittings".into(),
|
||||
"24".into(),
|
||||
],
|
||||
];
|
||||
assert!(is_page_number_toc(&cells));
|
||||
assert!(is_table_of_contents(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_numeric_data_table() {
|
||||
// Real 2-col data table: numeric first column, non-monotonic values.
|
||||
let cells = vec![
|
||||
vec!["101".into(), "45".into()],
|
||||
vec!["102".into(), "12".into()],
|
||||
vec!["103".into(), "88".into()],
|
||||
vec!["104".into(), "7".into()],
|
||||
vec!["105".into(), "63".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_non_monotonic_pages() {
|
||||
// Text labels but the "page" column jumps around — a small data table,
|
||||
// not a contents listing. 5 rows so the row-count guard passes and the
|
||||
// monotonicity check is what does the rejecting.
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Apples".into(), "42".into()],
|
||||
vec!["Oranges".into(), "7".into()],
|
||||
vec!["Pears".into(), "91".into()],
|
||||
vec!["Plums".into(), "3".into()],
|
||||
vec!["Grapes".into(), "60".into()],
|
||||
];
|
||||
// Sanity: this input clears the row-count and header guards, so a
|
||||
// failure here is genuinely the monotonicity check.
|
||||
assert!(cells.len() >= 5 && page_number_value(cells[0][1].trim()).is_some());
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_header_row_data_table() {
|
||||
// Real 2-col data table with a header row ("Mineral | CEC") and
|
||||
// ascending values that mimic page numbers — the header tells us it
|
||||
// is data, not contents.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Mineral or colloid type".into(),
|
||||
"CEC of pure colloid".into(),
|
||||
],
|
||||
vec!["kaolinite".into(), "10".into()],
|
||||
vec!["illite".into(), "30".into()],
|
||||
vec!["montmorillonite".into(), "100".into()],
|
||||
vec!["vermiculite".into(), "150".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_wide_data_grid() {
|
||||
// A 4-column regional data table must not be read as a TOC even with a
|
||||
// text first column and integer last column.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"REGIONS".into(),
|
||||
"2007".into(),
|
||||
"2010".into(),
|
||||
"2016".into(),
|
||||
],
|
||||
vec![
|
||||
"National Capital Region".into(),
|
||||
"9".into(),
|
||||
"8".into(),
|
||||
"5".into(),
|
||||
],
|
||||
vec!["Cordillera".into(), "1".into(), "2".into(), "1".into()],
|
||||
vec!["Ilocos Region".into(), "1".into(), "5".into(), "4".into()],
|
||||
vec!["Cagayan Valley".into(), "1".into(), "3".into(), "5".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_needs_page_number_last_column() {
|
||||
// Last column is prose, not page numbers.
|
||||
let cells = vec![
|
||||
vec!["Section A".into(), "see appendix".into()],
|
||||
vec!["Section B".into(), "see notes".into()],
|
||||
vec!["Section C".into(), "later".into()],
|
||||
vec!["Section D".into(), "TBD".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
}
|
||||
|
||||
+2371
-47
File diff suppressed because it is too large
Load Diff
+1512
-33
File diff suppressed because it is too large
Load Diff
@@ -586,6 +586,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
|
||||
@@ -108,6 +108,8 @@ pub(crate) fn try_split_financial_item(item: &TextItem) -> Option<Vec<TextItem>>
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item.item_type.clone(),
|
||||
mcid: item.mcid,
|
||||
});
|
||||
|
||||
+329
-7
@@ -123,6 +123,7 @@ fn format_toc_as_list(cells: &[Vec<String>], footnotes: &[String]) -> String {
|
||||
|
||||
/// True when the cell looks like a page number. Accepts:
|
||||
/// - plain digit tokens: "42", "86 86"
|
||||
/// - canonical roman numerals (front-matter pages): "vii", "ix", "xii"
|
||||
/// - dashed section-page IDs: "5-21", "A-1", "B--3", "TC-2" (common in
|
||||
/// technical manuals)
|
||||
fn is_page_number_cell(cell: &str) -> bool {
|
||||
@@ -138,6 +139,9 @@ fn is_page_number_cell(cell: &str) -> bool {
|
||||
if all_digits {
|
||||
return t.len() <= 4;
|
||||
}
|
||||
if super::canonical_roman_value(t).is_some() {
|
||||
return true;
|
||||
}
|
||||
// Section-page form: uppercase letters, digits, dashes; at least
|
||||
// one digit present.
|
||||
t.chars()
|
||||
@@ -160,6 +164,96 @@ fn starts_with_uppercase_word(cell: &str) -> bool {
|
||||
.is_some_and(|c| c.is_uppercase())
|
||||
}
|
||||
|
||||
fn starts_with_uppercase_alpha(cell: &str) -> bool {
|
||||
cell.chars()
|
||||
.find(|c| c.is_alphabetic())
|
||||
.is_some_and(|c| c.is_uppercase())
|
||||
}
|
||||
|
||||
fn starts_with_lowercase_alpha(cell: &str) -> bool {
|
||||
cell.chars()
|
||||
.find(|c| c.is_alphabetic())
|
||||
.is_some_and(|c| c.is_lowercase())
|
||||
}
|
||||
|
||||
fn starts_with_numbered_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim_start();
|
||||
let digit_count = trimmed.chars().take_while(|c| c.is_ascii_digit()).count();
|
||||
|
||||
digit_count > 0
|
||||
&& digit_count <= 3
|
||||
&& trimmed
|
||||
.chars()
|
||||
.nth(digit_count)
|
||||
.is_some_and(|c| matches!(c, '.' | ')' | '-' | ':'))
|
||||
}
|
||||
|
||||
fn starts_with_hierarchical_numbered_label(cell: &str) -> bool {
|
||||
let token = cell
|
||||
.split_whitespace()
|
||||
.next()
|
||||
.unwrap_or("")
|
||||
.trim_end_matches(['.', ')', ':', '-']);
|
||||
let levels: Vec<&str> = token.split('.').collect();
|
||||
(2..=4).contains(&levels.len())
|
||||
&& levels.iter().all(|level| {
|
||||
!level.is_empty()
|
||||
&& level.len() <= 3
|
||||
&& level.chars().all(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
fn alpha_word_count(cell: &str) -> usize {
|
||||
cell.split_whitespace()
|
||||
.filter(|word| word.chars().any(|c| c.is_alphabetic()))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn looks_like_compact_entry_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 3 || trimmed.len() > 80 {
|
||||
return false;
|
||||
}
|
||||
|
||||
if !starts_with_uppercase_alpha(trimmed) && !starts_with_numbered_label(trimmed) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if trimmed.ends_with(['.', ',', ';', ':']) {
|
||||
return false;
|
||||
}
|
||||
|
||||
let words = alpha_word_count(trimmed);
|
||||
(1..=6).contains(&words)
|
||||
}
|
||||
|
||||
fn looks_like_plain_section_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 4 || trimmed.len() > 40 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.ends_with(['.', ',', ';', ':']) || trimmed.contains(|ch: char| ch.is_ascii_digit()) {
|
||||
return false;
|
||||
}
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|ch| !ch.is_lowercase()) {
|
||||
return false;
|
||||
}
|
||||
trimmed
|
||||
.chars()
|
||||
.all(|ch| ch.is_alphabetic() || ch.is_whitespace() || matches!(ch, '&' | '/' | '-'))
|
||||
&& starts_with_uppercase_alpha(trimmed)
|
||||
&& (1..=4).contains(&alpha_word_count(trimmed))
|
||||
}
|
||||
|
||||
fn ends_like_incomplete_phrase(cell: &str) -> bool {
|
||||
let lower = cell.trim_end().to_ascii_lowercase();
|
||||
lower.ends_with(" and")
|
||||
|| lower.ends_with(" or")
|
||||
|| lower.ends_with(',')
|
||||
|| lower.ends_with('-')
|
||||
|| lower.ends_with('/')
|
||||
}
|
||||
|
||||
/// Clean up table cells: merge continuation rows, extract footnotes, remove empty rows
|
||||
fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
let mut cleaned: Vec<Vec<String>> = Vec::new();
|
||||
@@ -185,6 +279,9 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let num_cols = row.len();
|
||||
let filled_cells = row.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
|
||||
// Check if this is a continuation row (first column is empty but others have content).
|
||||
// A row with only 1 short non-empty cell (besides the first) is more likely a
|
||||
// section sub-header (e.g. "JAN", "FEB") than overflow text — don't merge it.
|
||||
@@ -222,31 +319,77 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
.iter()
|
||||
.filter(|cell| starts_with_uppercase_word(cell))
|
||||
.count();
|
||||
let first_non_empty_col = row.iter().position(|c| !c.trim().is_empty());
|
||||
let first_non_empty_cell = first_non_empty_col
|
||||
.and_then(|idx| row.get(idx))
|
||||
.map(|c| c.trim())
|
||||
.unwrap_or("");
|
||||
let title_like_later_cells = first_non_empty_col
|
||||
.map(|idx| {
|
||||
row.iter()
|
||||
.skip(idx + 1)
|
||||
.map(|c| c.trim())
|
||||
.filter(|c| !c.is_empty() && starts_with_uppercase_alpha(c))
|
||||
.count()
|
||||
})
|
||||
.unwrap_or(0);
|
||||
let prev_first_cell_empty = cleaned
|
||||
.last()
|
||||
.and_then(|r| r.first())
|
||||
.is_some_and(|c| c.trim().is_empty());
|
||||
let prev_first_cell = cleaned
|
||||
.last()
|
||||
.and_then(|r| r.first())
|
||||
.map(|c| c.trim())
|
||||
.unwrap_or("");
|
||||
let header_filled = cleaned
|
||||
.first()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(num_cols);
|
||||
let looks_like_spanning_first_column_row = first_cell.is_empty()
|
||||
&& row.len() >= 4
|
||||
&& non_first_cells.len() == row.len().saturating_sub(1)
|
||||
&& uppercase_leading_cells >= non_first_cells.len().saturating_sub(1);
|
||||
// Hierarchical tables often use a row-spanned first column: sub-rows
|
||||
// leave column 0 blank, then start a compact title-like label in
|
||||
// column 1. Wrapped continuations in the existing fixtures start
|
||||
// mid-sentence/lowercase ("continued text here", "with 3.5%...") or
|
||||
// carry lowercase fragments in the later cells, so keep those mergeable.
|
||||
let looks_like_hierarchical_subrow = first_cell.is_empty()
|
||||
&& first_non_empty_col == Some(1)
|
||||
&& looks_like_compact_entry_label(first_non_empty_cell)
|
||||
&& ((row.len() == 2 && starts_with_hierarchical_numbered_label(first_non_empty_cell))
|
||||
|| (row.len() >= 3 && non_first_cells.len() >= 2 && title_like_later_cells > 0)
|
||||
|| (non_first_cells.len() == 1
|
||||
&& row.len() >= 3
|
||||
&& prev_first_cell_empty
|
||||
&& alpha_word_count(first_non_empty_cell) >= 2));
|
||||
let looks_like_new_first_column_entry = !first_cell.is_empty()
|
||||
&& (starts_with_numbered_label(first_cell) || starts_with_uppercase_alpha(first_cell))
|
||||
&& filled_cells >= 2
|
||||
&& non_first_cells
|
||||
.iter()
|
||||
.any(|cell| looks_like_compact_entry_label(cell));
|
||||
let looks_like_section_label_row = !first_cell.is_empty()
|
||||
&& filled_cells == 1
|
||||
&& header_filled >= 3
|
||||
&& looks_like_plain_section_label(first_cell);
|
||||
// Classic continuation: first cell empty, content in other cells
|
||||
let is_classic_continuation = first_cell.is_empty()
|
||||
&& !non_first_cells.is_empty()
|
||||
&& !is_short_subheader
|
||||
&& !looks_like_data_row
|
||||
&& !looks_like_spanning_first_column_row
|
||||
&& !looks_like_hierarchical_subrow
|
||||
&& cleaned.len() > 1;
|
||||
|
||||
// Wrapped-cell continuation: row has fewer filled cells than the header
|
||||
// row, suggesting it's overflow text from the previous row's cells.
|
||||
// Only trigger when the previous row has significantly more filled cells.
|
||||
let num_cols = row.len();
|
||||
let filled_cells = row.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
let prev_filled = cleaned
|
||||
.last()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(0);
|
||||
let header_filled = cleaned
|
||||
.first()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(num_cols);
|
||||
// Merge when the row has significantly fewer filled cells than header.
|
||||
// For wide tables (5+ cols), require ≤50% of header cells.
|
||||
// For narrow tables (2-4 cols), require fewer than header cells.
|
||||
@@ -257,11 +400,18 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
} else {
|
||||
header_filled.saturating_sub(1)
|
||||
};
|
||||
let continues_wrapped_first_column_label = !first_cell.is_empty()
|
||||
&& starts_with_lowercase_alpha(first_cell)
|
||||
&& ends_like_incomplete_phrase(prev_first_cell);
|
||||
let is_wrapped_continuation = cleaned.len() > 1
|
||||
&& filled_cells <= max_filled_for_merge
|
||||
&& prev_filled > filled_cells
|
||||
&& (prev_filled > filled_cells
|
||||
|| (continues_wrapped_first_column_label && prev_filled >= filled_cells))
|
||||
&& !looks_like_data_row
|
||||
&& !looks_like_spanning_first_column_row
|
||||
&& !looks_like_hierarchical_subrow
|
||||
&& !looks_like_new_first_column_entry
|
||||
&& !looks_like_section_label_row
|
||||
&& !is_short_subheader;
|
||||
|
||||
let is_continuation = is_classic_continuation || is_wrapped_continuation;
|
||||
@@ -407,6 +557,44 @@ mod tests {
|
||||
assert!(cleaned[1][1].contains("continued text here"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_first_column_section_label_not_merged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Properties".into(),
|
||||
"Conditions".into(),
|
||||
"Method".into(),
|
||||
"Typical values".into(),
|
||||
"Units".into(),
|
||||
],
|
||||
vec![
|
||||
"Melt Flow Rate".into(),
|
||||
"230 C/2.16 kg".into(),
|
||||
"ASTM D1238".into(),
|
||||
"3.0".into(),
|
||||
"g/10 min".into(),
|
||||
],
|
||||
vec![
|
||||
"Mechanical".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"Tensile Stress at Yield".into(),
|
||||
"50 mm/min".into(),
|
||||
"ASTM D638".into(),
|
||||
"31".into(),
|
||||
"MPa".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 4);
|
||||
assert_eq!(cleaned[2][0], "Mechanical");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_short_subheader_not_merged() {
|
||||
let cells = vec![
|
||||
@@ -459,6 +647,140 @@ mod tests {
|
||||
assert_eq!(cleaned[2][1], "Uncertainty around other copies");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_numbered_hierarchy_rows_not_overmerged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Group".into(),
|
||||
"Task".into(),
|
||||
"Detail".into(),
|
||||
"Benefit".into(),
|
||||
],
|
||||
vec![
|
||||
"1. Group alpha".into(),
|
||||
"Task setup and".into(),
|
||||
"Begin setup".into(),
|
||||
"Faster start".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"management".into(),
|
||||
"recommended profile".into(),
|
||||
"with saved defaults".into(),
|
||||
],
|
||||
vec![
|
||||
"2. Group beta and".into(),
|
||||
"Storage setup".into(),
|
||||
"Provides upload tools".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"fine-tuning".into(),
|
||||
"".into(),
|
||||
"for filtered inputs".into(),
|
||||
"service".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Label workspace".into(),
|
||||
"Creates review sets".into(),
|
||||
"Lets teams review".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Model training".into(),
|
||||
"".into(),
|
||||
"Supports custom model".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 5);
|
||||
assert_eq!(cleaned[1][0], "1. Group alpha");
|
||||
assert_eq!(cleaned[1][1], "Task setup and management");
|
||||
assert_eq!(cleaned[2][0], "2. Group beta and fine-tuning");
|
||||
assert_eq!(cleaned[2][1], "Storage setup");
|
||||
assert_eq!(cleaned[3][1], "Label workspace");
|
||||
assert_eq!(cleaned[4][1], "Model training");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_two_column_numbered_subrows_not_merged() {
|
||||
let cells = vec![
|
||||
vec!["Area".into(), "Competence".into()],
|
||||
vec![
|
||||
"1. Embodying sustainability values".into(),
|
||||
"1.1 Valuing sustainability".into(),
|
||||
],
|
||||
vec!["".into(), "1.2 Supporting fairness".into()],
|
||||
vec!["".into(), "1.3 Promoting nature".into()],
|
||||
vec![
|
||||
"2. Embracing complexity".into(),
|
||||
"2.1 Systems thinking".into(),
|
||||
],
|
||||
vec!["".into(), "2.2 Critical thinking".into()],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 6);
|
||||
assert_eq!(cleaned[2], vec!["", "1.2 Supporting fairness"]);
|
||||
assert_eq!(cleaned[3], vec!["", "1.3 Promoting nature"]);
|
||||
assert_eq!(cleaned[5], vec!["", "2.2 Critical thinking"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_two_column_numbered_continuation_merges() {
|
||||
let cells = vec![
|
||||
vec!["Area".into(), "Requirement".into()],
|
||||
vec!["Safety".into(), "The program includes".into()],
|
||||
vec!["".into(), "1. First requirement for every operator".into()],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 2);
|
||||
assert_eq!(
|
||||
cleaned[1][1],
|
||||
"The program includes 1. First requirement for every operator"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_partial_hierarchical_subrow_not_merged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Group".into(),
|
||||
"Task".into(),
|
||||
"Detail".into(),
|
||||
"Benefit".into(),
|
||||
],
|
||||
vec![
|
||||
"Group A".into(),
|
||||
"Alpha task".into(),
|
||||
"Initial detail".into(),
|
||||
"Initial benefit".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Beta task".into(),
|
||||
"Parallel detail".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"second line".into(),
|
||||
"additional detail".into(),
|
||||
"".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 3);
|
||||
assert_eq!(cleaned[1][1], "Alpha task");
|
||||
assert_eq!(cleaned[2][0], "");
|
||||
assert_eq!(cleaned[2][1], "Beta task second line");
|
||||
assert_eq!(cleaned[2][2], "Parallel detail additional detail");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_full_width_continuation_row_still_merges_when_lowercase() {
|
||||
let cells = vec![
|
||||
|
||||
@@ -369,6 +369,10 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|
||||
let prev_ends_with_hyphen = result.ends_with('-');
|
||||
let curr_is_hyphen = text == "-";
|
||||
let curr_starts_with_hyphen = text.starts_with('-');
|
||||
let prev_ends_with_open_delimiter =
|
||||
result.ends_with('(') || result.ends_with('[') || result.ends_with('{');
|
||||
let curr_starts_with_close_delimiter =
|
||||
text.starts_with(')') || text.starts_with(']') || text.starts_with('}');
|
||||
|
||||
// Detect subscript/superscript: smaller font size and/or Y offset
|
||||
let font_ratio = item.font_size / prev_item.font_size;
|
||||
@@ -385,6 +389,8 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|
||||
|| curr_starts_with_hyphen
|
||||
|| is_sub_super
|
||||
|| was_sub_super
|
||||
|| prev_ends_with_open_delimiter
|
||||
|| curr_starts_with_close_delimiter
|
||||
{
|
||||
result.push_str(text);
|
||||
} else {
|
||||
@@ -514,6 +520,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -727,6 +735,18 @@ mod tests {
|
||||
assert_eq!(join_cell_items(&[&a, &b, &c]), "pre-fix");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_join_cell_items_parenthetical_no_inner_spaces() {
|
||||
let a = make_item("The first sentence", 100.0, 500.0, 10.0);
|
||||
let b = make_item("(", 190.0, 500.0, 10.0);
|
||||
let c = make_item("twice", 195.0, 500.0, 10.0);
|
||||
let d = make_item(")", 220.0, 500.0, 10.0);
|
||||
assert_eq!(
|
||||
join_cell_items(&[&a, &b, &c, &d]),
|
||||
"The first sentence (twice)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_join_cell_items_subscript_no_space() {
|
||||
let a = make_item("H", 100.0, 500.0, 12.0);
|
||||
@@ -866,6 +886,8 @@ mod tests {
|
||||
font: String::new(),
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
page: 1,
|
||||
@@ -902,6 +924,8 @@ mod tests {
|
||||
font: String::new(),
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
page: 1,
|
||||
|
||||
+1298
-2
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,520 @@
|
||||
//! Text-quality detection: deciding when an extracted text layer is too broken
|
||||
//! to serve and a page should fall back to OCR.
|
||||
//!
|
||||
//! Extraction can produce plausible-looking bytes that are actually garbage —
|
||||
//! failed CID→Unicode mappings, broken ToUnicode CMaps, mojibake. These
|
||||
//! detectors catch that and let callers set `needs_ocr`. They come in two
|
||||
//! layers, sharing the same primitives:
|
||||
//!
|
||||
//! - **Markdown-level** ([`detect_encoding_issues`], [`is_garbage_text`],
|
||||
//! [`is_cid_garbage`]) run on a page's final markdown string. Used as a
|
||||
//! backstop on the region-extraction and whole-document paths.
|
||||
//! - **Item/span-level** ([`analyze_text_quality`],
|
||||
//! [`region_items_have_decoding_issue`]) run on individual `TextItem`s and
|
||||
//! accumulate per-page evidence, so localized garbled spans on an otherwise
|
||||
//! clean page are caught without a single span having to condemn the page.
|
||||
//!
|
||||
//! Detection classes, roughly by signal:
|
||||
//! - **Replacement runs**: U+FFFD clusters ([`has_replacement_text_run`]).
|
||||
//! - **Private-use / C1-control runs**: CID passthrough landing in PUA or the
|
||||
//! C1 block ([`has_private_use_text_run`], [`has_cid_control_token`]).
|
||||
//! - **Dollar-as-space**: `Word$Word$Word` from broken CMaps
|
||||
//! ([`has_dollar_as_space_pattern`]).
|
||||
//! - **Non-alphanumeric dominance**: symbol soup ([`is_garbage_text`]).
|
||||
//! - **Substitution-cipher letter statistics**: pure-ASCII output whose letter
|
||||
//! distribution is a permutation of natural language ([`CipherGarbleStats`]).
|
||||
|
||||
use crate::types::TextItem;
|
||||
use crate::{add_ocr_reason, OCR_REASON_SUSPECTED_GARBLED_TEXT};
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
/// Detect broken font encodings in extracted markdown text.
|
||||
///
|
||||
/// Two heuristics:
|
||||
/// 1. **U+FFFD**: Any replacement character indicates decode failures.
|
||||
/// 2. **Dollar-as-space**: Pattern like `Word$Word$Word` where `$` is used as a
|
||||
/// word separator due to broken ToUnicode CMaps. Triggers when either:
|
||||
/// - More than 50% of `$` are between letters (clear substitution pattern), OR
|
||||
/// - More than 20 letter-dollar-letter occurrences (even if some `$` are also
|
||||
/// used as trailing/leading separators, 20+ is far beyond normal financial text).
|
||||
pub(crate) fn detect_encoding_issues(markdown: &str) -> bool {
|
||||
// Heuristic 1: U+FFFD replacement characters
|
||||
if markdown.contains('\u{FFFD}') {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Heuristic 2: dollar-as-space pattern
|
||||
if has_dollar_as_space_pattern(markdown) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Heuristic 3: substitution-cipher letter statistics (broken ToUnicode)
|
||||
let mut stats = CipherGarbleStats::default();
|
||||
stats.add_text(markdown);
|
||||
stats.looks_garbled()
|
||||
}
|
||||
|
||||
fn has_dollar_as_space_pattern(markdown: &str) -> bool {
|
||||
let total_dollars = markdown.matches('$').count();
|
||||
if total_dollars > 10 {
|
||||
let bytes = markdown.as_bytes();
|
||||
let mut letter_dollar_letter = 0usize;
|
||||
for i in 1..bytes.len().saturating_sub(1) {
|
||||
if bytes[i] == b'$'
|
||||
&& bytes[i - 1].is_ascii_alphabetic()
|
||||
&& bytes[i + 1].is_ascii_alphabetic()
|
||||
{
|
||||
letter_dollar_letter += 1;
|
||||
}
|
||||
}
|
||||
if letter_dollar_letter > 20 || letter_dollar_letter * 2 > total_dollars {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// English letter frequencies (percent, a–z). Used as a natural-language
|
||||
/// reference: every Latin-script language in the eval corpus (Swedish,
|
||||
/// Finnish, Turkish, German, romaji) scores ≥ 0.80 cosine similarity against
|
||||
/// it, while substitution-cipher text scores ~0.53.
|
||||
const ENGLISH_LETTER_FREQ: [f64; 26] = [
|
||||
8.2, 1.5, 2.8, 4.3, 12.7, 2.2, 2.0, 6.1, 7.0, 0.15, 0.8, 4.0, 2.4, 6.7, 7.5, 1.9, 0.1, 6.0,
|
||||
6.3, 9.1, 2.8, 1.0, 2.4, 0.15, 2.0, 0.07,
|
||||
];
|
||||
|
||||
/// Letter statistics for detecting substitution-cipher garbling: broken
|
||||
/// ToUnicode CMaps that shift every character by a per-range constant (e.g.
|
||||
/// `Certificate` extracted as `8VceZWZTReV`). Such text is 100% printable
|
||||
/// ASCII with word-like token lengths, so it defeats `is_garbage_text` and
|
||||
/// produces no replacement characters — it needs its own discriminator.
|
||||
#[derive(Debug, Default)]
|
||||
struct CipherGarbleStats {
|
||||
/// Case-folded ASCII letter histogram.
|
||||
letter_counts: [u32; 26],
|
||||
ascii_letters: usize,
|
||||
ascii_vowels: usize,
|
||||
/// Accented Latin letters (Latin-1 Supplement through Latin Extended-B,
|
||||
/// plus Latin Extended Additional). Count toward Latin dominance only.
|
||||
latin_ext_letters: usize,
|
||||
non_latin_letters: usize,
|
||||
/// Adjacent ASCII-letter pairs, and how many of them switch from
|
||||
/// lowercase straight to uppercase mid-word.
|
||||
letter_bigrams: usize,
|
||||
case_shift_bigrams: usize,
|
||||
}
|
||||
|
||||
impl CipherGarbleStats {
|
||||
fn add_text(&mut self, text: &str) {
|
||||
let mut prev: Option<char> = None;
|
||||
for ch in text.chars() {
|
||||
if ch.is_ascii_alphabetic() {
|
||||
let idx = (ch.to_ascii_lowercase() as u8 - b'a') as usize;
|
||||
self.letter_counts[idx] += 1;
|
||||
self.ascii_letters += 1;
|
||||
if matches!(ch.to_ascii_lowercase(), 'a' | 'e' | 'i' | 'o' | 'u') {
|
||||
self.ascii_vowels += 1;
|
||||
}
|
||||
if let Some(p) = prev {
|
||||
self.letter_bigrams += 1;
|
||||
if p.is_ascii_lowercase() && ch.is_ascii_uppercase() {
|
||||
self.case_shift_bigrams += 1;
|
||||
}
|
||||
}
|
||||
prev = Some(ch);
|
||||
} else {
|
||||
if ch.is_alphabetic() {
|
||||
if matches!(ch as u32, 0xC0..=0x24F | 0x1E00..=0x1EFF) {
|
||||
self.latin_ext_letters += 1;
|
||||
} else {
|
||||
self.non_latin_letters += 1;
|
||||
}
|
||||
}
|
||||
prev = None;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Cosine similarity between the observed letter histogram and English
|
||||
/// letter frequencies. A shifted alphabet permutes the histogram, which
|
||||
/// destroys the similarity regardless of the shift amount.
|
||||
fn english_cosine(&self) -> f64 {
|
||||
if self.ascii_letters == 0 {
|
||||
return 1.0;
|
||||
}
|
||||
let n = self.ascii_letters as f64;
|
||||
let mut dot = 0.0;
|
||||
let mut norm_obs = 0.0;
|
||||
for (count, freq) in self.letter_counts.iter().zip(ENGLISH_LETTER_FREQ) {
|
||||
let p = *count as f64 / n;
|
||||
dot += p * freq;
|
||||
norm_obs += p * p;
|
||||
}
|
||||
let norm_en = ENGLISH_LETTER_FREQ
|
||||
.iter()
|
||||
.map(|f| f * f)
|
||||
.sum::<f64>()
|
||||
.sqrt();
|
||||
dot / (norm_obs.sqrt() * norm_en)
|
||||
}
|
||||
|
||||
/// Cosine similarity between the observed histogram and English
|
||||
/// frequencies after sorting BOTH descending — i.e. comparing the *shape*
|
||||
/// of the frequency profile, ignoring which letter sits where. A
|
||||
/// substitution cipher is a bijection, so it preserves this shape exactly
|
||||
/// (att10k 0.97, arbitrary shifts 0.99) regardless of case or offset.
|
||||
/// Non-linguistic ASCII has a different profile: a small alphabet is far
|
||||
/// steeper (random DNA 0.74, hex dumps 0.81), so the shape diverges.
|
||||
fn english_shape_cosine(&self) -> f64 {
|
||||
if self.ascii_letters == 0 {
|
||||
return 1.0;
|
||||
}
|
||||
let n = self.ascii_letters as f64;
|
||||
let mut obs: [f64; 26] = std::array::from_fn(|i| self.letter_counts[i] as f64 / n);
|
||||
obs.sort_unstable_by(|a, b| b.total_cmp(a));
|
||||
let mut en = ENGLISH_LETTER_FREQ;
|
||||
en.sort_unstable_by(|a, b| b.total_cmp(a));
|
||||
|
||||
let dot: f64 = obs.iter().zip(en).map(|(o, e)| o * e).sum();
|
||||
let norm_obs = obs.iter().map(|o| o * o).sum::<f64>().sqrt();
|
||||
let norm_en = en.iter().map(|e| e * e).sum::<f64>().sqrt();
|
||||
dot / (norm_obs * norm_en)
|
||||
}
|
||||
|
||||
/// Thresholds validated against the 380-document pdf-evals snapshot
|
||||
/// corpus (0 false positives) and the garbled ParseBench `att10k` page
|
||||
/// (vowel ratio 0.245, case-shift rate 0.225, cosine 0.532). Closest
|
||||
/// legitimate document on each axis: vowel ratio 0.264 (circuit
|
||||
/// schematic), case-shift rate 0.021, cosine 0.801.
|
||||
fn looks_garbled(&self) -> bool {
|
||||
// Need a statistically meaningful, Latin-dominant sample.
|
||||
if self.ascii_letters < 200
|
||||
|| self.non_latin_letters > self.ascii_letters + self.latin_ext_letters
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// Real Latin-script text keeps vowels above ~30% of letters even in
|
||||
// acronym- and part-number-heavy documents; shifted text starves them.
|
||||
let vowel_ratio = self.ascii_vowels as f64 / self.ascii_letters as f64;
|
||||
if vowel_ratio > 0.30 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Signal 1: lowercase→uppercase transitions inside words. A shifted
|
||||
// lowercase alphabet straddles the ASCII uppercase block ('i'→'Z',
|
||||
// 't'→'e'), so garbled words flip case constantly. Real documents
|
||||
// stay ≤ 0.02 even with camelCase identifiers.
|
||||
let case_shifts = self.letter_bigrams >= 100
|
||||
&& self.case_shift_bigrams as f64 >= self.letter_bigrams as f64 * 0.10;
|
||||
|
||||
// Signal 2: the histogram is a permutation of natural language — an
|
||||
// English-like frequency SHAPE (sorted cosine high) but with letters
|
||||
// in the wrong POSITIONS (unsorted cosine low). This is the signature
|
||||
// of a substitution cipher and is case-independent, so it catches
|
||||
// all-lowercase and all-uppercase shifts as well as case-straddling
|
||||
// ones. Genuinely non-linguistic ASCII that is merely "unlike English"
|
||||
// fails one of the two halves: DNA/hex dumps have too steep a profile
|
||||
// (shape cosine < 0.90), while protein sequences, ticker symbols and
|
||||
// base64 are not sufficiently unlike English in position (unsorted
|
||||
// cosine ≥ 0.60) — so none of them are routed to OCR.
|
||||
let permuted_language = self.english_cosine() < 0.60 && self.english_shape_cosine() >= 0.90;
|
||||
|
||||
case_shifts || permuted_language
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct TextQualityReport {
|
||||
pub(crate) pages_needing_ocr: Vec<u32>,
|
||||
pub(crate) has_encoding_issues: bool,
|
||||
pub(crate) reasons_by_page: BTreeMap<u32, Vec<String>>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
struct PageTextQualityEvidence {
|
||||
chars: usize,
|
||||
replacement_chars: usize,
|
||||
replacement_spans: usize,
|
||||
longest_replacement_run: usize,
|
||||
cipher_garble: CipherGarbleStats,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum TextSpanIssueKind {
|
||||
Replacement,
|
||||
Strong,
|
||||
}
|
||||
|
||||
pub(crate) fn analyze_text_quality(items: &[TextItem]) -> TextQualityReport {
|
||||
let mut reasons_by_page = BTreeMap::new();
|
||||
let mut evidence_by_page = BTreeMap::<u32, PageTextQualityEvidence>::new();
|
||||
|
||||
for item in items {
|
||||
if !matches!(item.item_type, crate::types::ItemType::Text) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let evidence = evidence_by_page.entry(item.page).or_default();
|
||||
evidence.chars += item.text.chars().filter(|ch| !ch.is_whitespace()).count();
|
||||
evidence.cipher_garble.add_text(&item.text);
|
||||
|
||||
match text_span_decoding_issue_kind(&item.text) {
|
||||
Some(TextSpanIssueKind::Strong) => {
|
||||
add_ocr_reason(
|
||||
&mut reasons_by_page,
|
||||
item.page,
|
||||
OCR_REASON_SUSPECTED_GARBLED_TEXT,
|
||||
);
|
||||
}
|
||||
Some(TextSpanIssueKind::Replacement) => {
|
||||
let stats = replacement_text_stats(&item.text);
|
||||
evidence.replacement_chars += stats.0;
|
||||
evidence.replacement_spans += 1;
|
||||
evidence.longest_replacement_run = evidence.longest_replacement_run.max(stats.1);
|
||||
}
|
||||
None => {}
|
||||
}
|
||||
}
|
||||
|
||||
for (page, evidence) in evidence_by_page {
|
||||
if reasons_by_page.contains_key(&page) {
|
||||
continue;
|
||||
}
|
||||
if page_replacement_evidence_needs_ocr(&evidence) || evidence.cipher_garble.looks_garbled()
|
||||
{
|
||||
add_ocr_reason(
|
||||
&mut reasons_by_page,
|
||||
page,
|
||||
OCR_REASON_SUSPECTED_GARBLED_TEXT,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let pages_needing_ocr: Vec<u32> = reasons_by_page.keys().copied().collect();
|
||||
TextQualityReport {
|
||||
has_encoding_issues: !pages_needing_ocr.is_empty(),
|
||||
pages_needing_ocr,
|
||||
reasons_by_page,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn region_items_have_decoding_issue(items: &[TextItem]) -> bool {
|
||||
items.iter().any(|item| {
|
||||
matches!(item.item_type, crate::types::ItemType::Text)
|
||||
&& text_span_has_decoding_issue(&item.text)
|
||||
})
|
||||
}
|
||||
|
||||
fn text_span_has_decoding_issue(text: &str) -> bool {
|
||||
text_span_decoding_issue_kind(text).is_some()
|
||||
}
|
||||
|
||||
fn text_span_decoding_issue_kind(text: &str) -> Option<TextSpanIssueKind> {
|
||||
let text = text.trim();
|
||||
if text.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
if has_dollar_as_space_pattern(text)
|
||||
|| has_private_use_text_run(text)
|
||||
|| is_cid_garbage(text)
|
||||
|| has_cid_control_token(text)
|
||||
{
|
||||
return Some(TextSpanIssueKind::Strong);
|
||||
}
|
||||
|
||||
if has_replacement_text_run(text) {
|
||||
return Some(TextSpanIssueKind::Replacement);
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
fn replacement_text_stats(text: &str) -> (usize, usize) {
|
||||
let mut replacement = 0usize;
|
||||
let mut current_run = 0usize;
|
||||
let mut longest_run = 0usize;
|
||||
|
||||
for ch in text.chars() {
|
||||
if ch == '\u{FFFD}' {
|
||||
replacement += 1;
|
||||
current_run += 1;
|
||||
longest_run = longest_run.max(current_run);
|
||||
} else {
|
||||
current_run = 0;
|
||||
}
|
||||
}
|
||||
|
||||
(replacement, longest_run)
|
||||
}
|
||||
|
||||
fn page_replacement_evidence_needs_ocr(evidence: &PageTextQualityEvidence) -> bool {
|
||||
if evidence.replacement_chars == 0 || evidence.chars == 0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// If the entire page is only a short broken text layer, even a short
|
||||
// replacement run is enough evidence. On otherwise text-heavy pages,
|
||||
// require density so math formulas do not force full-page OCR.
|
||||
if evidence.chars <= 80 && evidence.longest_replacement_run >= 2 {
|
||||
return true;
|
||||
}
|
||||
|
||||
let replacement_density_bps = evidence.replacement_chars * 10_000 / evidence.chars;
|
||||
let enough_bad_text = evidence.replacement_chars >= 12 && replacement_density_bps >= 500;
|
||||
let repeated_bad_spans = evidence.replacement_spans >= 3 && replacement_density_bps >= 250;
|
||||
let long_bad_run = evidence.longest_replacement_run >= 8 && replacement_density_bps >= 250;
|
||||
|
||||
enough_bad_text || repeated_bad_spans || long_bad_run
|
||||
}
|
||||
|
||||
fn has_replacement_text_run(text: &str) -> bool {
|
||||
let (replacement, longest_run) = replacement_text_stats(text);
|
||||
longest_run >= 2 || replacement >= 3
|
||||
}
|
||||
|
||||
fn has_private_use_text_run(text: &str) -> bool {
|
||||
let mut total = 0usize;
|
||||
let mut private_use = 0usize;
|
||||
let mut current_run = 0usize;
|
||||
let mut longest_run = 0usize;
|
||||
|
||||
for ch in text.chars() {
|
||||
if ch.is_whitespace() {
|
||||
current_run = 0;
|
||||
continue;
|
||||
}
|
||||
total += 1;
|
||||
if is_private_use_char(ch) {
|
||||
private_use += 1;
|
||||
current_run += 1;
|
||||
longest_run = longest_run.max(current_run);
|
||||
} else {
|
||||
current_run = 0;
|
||||
}
|
||||
}
|
||||
|
||||
if private_use == 0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
longest_run >= 3 || (total >= 5 && private_use >= 2 && private_use * 2 >= total)
|
||||
}
|
||||
|
||||
fn has_cid_control_token(text: &str) -> bool {
|
||||
text.split_whitespace().any(token_has_cid_control)
|
||||
}
|
||||
|
||||
fn token_has_cid_control(token: &str) -> bool {
|
||||
let mut total = 0usize;
|
||||
let mut c1_control = 0usize;
|
||||
|
||||
for ch in token.chars() {
|
||||
total += 1;
|
||||
if ('\u{0080}'..='\u{009F}').contains(&ch) {
|
||||
c1_control += 1;
|
||||
}
|
||||
}
|
||||
|
||||
total >= 5 && c1_control >= 2 && c1_control * 20 >= total
|
||||
}
|
||||
|
||||
fn is_private_use_char(ch: char) -> bool {
|
||||
matches!(
|
||||
ch as u32,
|
||||
0xE000..=0xF8FF | 0xF0000..=0xFFFFD | 0x100000..=0x10FFFD
|
||||
)
|
||||
}
|
||||
|
||||
/// Check if extracted text is predominantly garbage (non-alphanumeric).
|
||||
///
|
||||
/// Broken font encodings produce text like "----1-.-.-.___ --.-. .._ I_---."
|
||||
/// where most characters are punctuation/symbols. Real text in any language
|
||||
/// has >50% alphanumeric characters.
|
||||
pub(crate) fn is_garbage_text(markdown: &str) -> bool {
|
||||
let mut alphanum = 0usize;
|
||||
let mut non_alphanum = 0usize;
|
||||
|
||||
let chars: Vec<char> = markdown.chars().collect();
|
||||
let mut i = 0usize;
|
||||
while i < chars.len() {
|
||||
let ch = chars[i];
|
||||
let mut run_end = i + 1;
|
||||
while run_end < chars.len() && chars[run_end] == ch {
|
||||
run_end += 1;
|
||||
}
|
||||
|
||||
let is_decorative_leader = matches!(ch, '.' | '_' | '·') && run_end - i >= 3;
|
||||
if !is_decorative_leader {
|
||||
for &run_ch in &chars[i..run_end] {
|
||||
if run_ch.is_whitespace() {
|
||||
continue;
|
||||
}
|
||||
// Skip markdown syntax chars that we add (not from the PDF)
|
||||
if matches!(run_ch, '#' | '*' | '|' | '-' | '\n') {
|
||||
continue;
|
||||
}
|
||||
if run_ch.is_alphanumeric() {
|
||||
alphanum += 1;
|
||||
} else {
|
||||
non_alphanum += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
i = run_end;
|
||||
}
|
||||
|
||||
let total = alphanum + non_alphanum;
|
||||
total >= 50 && alphanum * 2 < total
|
||||
}
|
||||
|
||||
/// Detect garbage from failed CID-to-Unicode mapping on Identity-H fonts.
|
||||
///
|
||||
/// When CID values don't correspond to Unicode codepoints, the raw bytes often
|
||||
/// produce characters in the C1 control range (U+0080–U+009F) or Private Use
|
||||
/// Area, mixed with random Latin Extended characters. Valid text in any
|
||||
/// language almost never contains C1 controls. We also fall back to the
|
||||
/// general `is_garbage_text` check for non-alphanumeric-heavy patterns.
|
||||
pub(crate) fn is_cid_garbage(text: &str) -> bool {
|
||||
if is_garbage_text(text) {
|
||||
return true;
|
||||
}
|
||||
let mut total = 0usize;
|
||||
let mut c1_control = 0usize;
|
||||
let mut high_latin = 0usize;
|
||||
for ch in text.chars() {
|
||||
if ch.is_whitespace() {
|
||||
continue;
|
||||
}
|
||||
total += 1;
|
||||
// C1 control characters (U+0080–U+009F) — almost never in real text
|
||||
if ch == '·' {
|
||||
continue;
|
||||
}
|
||||
if ('\u{0080}'..='\u{009F}').contains(&ch) {
|
||||
c1_control += 1;
|
||||
}
|
||||
// High Latin-1 (U+00A0–U+00FF) — legitimate in Western European text
|
||||
// but when combined with ASCII in CID passthrough, indicates mojibake
|
||||
// from CID values being misinterpreted as Latin-1 characters.
|
||||
if ('\u{00A0}'..='\u{00FF}').contains(&ch) {
|
||||
high_latin += 1;
|
||||
}
|
||||
}
|
||||
if total < 5 {
|
||||
return false;
|
||||
}
|
||||
// If ≥5% of non-whitespace chars are C1 controls, it's garbage
|
||||
if c1_control >= 2 && c1_control * 20 >= total {
|
||||
return true;
|
||||
}
|
||||
// If ≥40% of non-whitespace chars are high Latin-1 AND the text has few
|
||||
// ASCII letters, it's likely CID-as-Latin-1 mojibake (Japanese/CJK PDFs
|
||||
// where CID values 0x80-0xFF become accented Latin characters). Keep a
|
||||
// minimum length so short math tokens like "2×()×" do not route a clean
|
||||
// page to OCR.
|
||||
let ascii_letters = text.chars().filter(|c| c.is_ascii_alphabetic()).count();
|
||||
total >= 20 && high_latin * 5 >= total * 2 && ascii_letters * 3 < total
|
||||
}
|
||||
@@ -93,6 +93,9 @@ pub fn is_bold_font(font_name: &str) -> bool {
|
||||
|| lower.contains("extrabold")
|
||||
|| lower.contains("ultrabold")
|
||||
|| lower.contains("medium") && !lower.contains("mediumitalic") // Some fonts use Medium for semi-bold
|
||||
// URW Type 1 fonts abbreviate Medium as "Medi" (e.g. NimbusRomNo9L-Medi,
|
||||
// the Times-Bold substitute in LaTeX documents; -MediItal is bold italic).
|
||||
|| lower.contains("-medi") && !lower.contains("mediumital")
|
||||
}
|
||||
|
||||
/// Detect if a font name indicates italic/oblique style
|
||||
@@ -762,6 +765,17 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
#[test]
|
||||
fn bold_font_urw_medi_abbreviation() {
|
||||
// URW Type 1 fonts (LaTeX default Times) abbreviate Medium as "Medi"
|
||||
assert!(is_bold_font("NROFIU+NimbusRomNo9L-Medi"));
|
||||
assert!(is_bold_font("NimbusRomNo9L-MediItal"));
|
||||
assert!(!is_bold_font("DSSZWN+NimbusRomNo9L-Regu"));
|
||||
assert!(!is_bold_font("NimbusRomNo9L-ReguItal"));
|
||||
// Medium-Italic exclusion still holds
|
||||
assert!(!is_bold_font("Foo-MediumItalic"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_soft_hyphen() {
|
||||
assert_eq!(expand_ligatures("con\u{00AD}tent"), "content");
|
||||
@@ -883,6 +897,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1002,6 +1018,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -1078,6 +1096,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
+191
-33
@@ -4,11 +4,17 @@
|
||||
|
||||
use log::{debug, warn};
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
use std::borrow::Cow;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::glyph_names::glyph_to_char;
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
static BUILTIN_CMAPS: include_dir::Dir<'_> =
|
||||
include_dir::include_dir!("$CARGO_MANIFEST_DIR/external/bcmaps");
|
||||
|
||||
/// A parsed ToUnicode CMap mapping CIDs to Unicode strings
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct ToUnicodeCMap {
|
||||
@@ -329,7 +335,7 @@ impl ToUnicodeCMap {
|
||||
if let (Some(start), Some(end), Some(base)) = (
|
||||
parse_hex_u16(&start_hex),
|
||||
parse_hex_u16(&end_hex),
|
||||
parse_hex_u32(&base_hex),
|
||||
hex_to_unicode_scalar(&base_hex),
|
||||
) {
|
||||
self.ranges.push((start, end, base));
|
||||
}
|
||||
@@ -575,32 +581,86 @@ fn parse_hex_u16(hex: &str) -> Option<u16> {
|
||||
u16::from_str_radix(hex.trim(), 16).ok()
|
||||
}
|
||||
|
||||
/// Parse a hex string to u32
|
||||
fn parse_hex_u32(hex: &str) -> Option<u32> {
|
||||
u32::from_str_radix(hex.trim(), 16).ok()
|
||||
}
|
||||
|
||||
/// Convert a hex string to a Unicode string
|
||||
/// Handles both 2-byte (BMP) and 4-byte (supplementary) codepoints
|
||||
/// Convert a ToUnicode destination hex string to Unicode.
|
||||
///
|
||||
/// PDF ToUnicode destinations are UTF-16BE strings. Supplementary-plane
|
||||
/// characters are encoded as surrogate pairs, so treating each 4-hex chunk as
|
||||
/// a scalar drops emoji like D83CDF1F.
|
||||
fn hex_to_unicode_string(hex: &str) -> Option<String> {
|
||||
let hex = hex.trim();
|
||||
let mut result = String::new();
|
||||
|
||||
// Process 4 hex digits at a time
|
||||
let mut i = 0;
|
||||
while i + 4 <= hex.len() {
|
||||
if let Ok(cp) = u32::from_str_radix(&hex[i..i + 4], 16) {
|
||||
if let Some(c) = char::from_u32(cp) {
|
||||
result.push(c);
|
||||
}
|
||||
}
|
||||
i += 4;
|
||||
let hex: String = hex.chars().filter(|ch| !ch.is_ascii_whitespace()).collect();
|
||||
if hex.is_empty() || !hex.len().is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
|
||||
if result.is_empty() {
|
||||
None
|
||||
let bytes: Option<Vec<u8>> = (0..hex.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
|
||||
.collect();
|
||||
let bytes = bytes?;
|
||||
|
||||
if bytes.len().is_multiple_of(2) {
|
||||
let units: Vec<u16> = bytes
|
||||
.chunks_exact(2)
|
||||
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
|
||||
.collect();
|
||||
if let Ok(result) = String::from_utf16(&units) {
|
||||
if !result.is_empty() {
|
||||
return Some(normalize_tounicode_destination(result));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Be permissive for non-standard one-byte destinations.
|
||||
if bytes.len() == 1 {
|
||||
let ch = bytes[0] as char;
|
||||
if !ch.is_control() || ch == '\t' || ch == '\n' {
|
||||
return Some(ch.to_string());
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
fn normalize_tounicode_destination(text: String) -> String {
|
||||
let is_multi_char = text.chars().nth(1).is_some();
|
||||
|
||||
// Some malformed producer CMaps put a list of alternative whitespace or
|
||||
// hyphen codepoints into one destination. Keep ordinary multi-character
|
||||
// mappings intact unless that malformed signature is present.
|
||||
if is_multi_char
|
||||
&& text.chars().all(char::is_whitespace)
|
||||
&& text.chars().any(|ch| matches!(ch, '\t' | '\n' | '\r'))
|
||||
{
|
||||
return if text.contains('\t') {
|
||||
"\t".to_string()
|
||||
} else {
|
||||
" ".to_string()
|
||||
};
|
||||
}
|
||||
|
||||
if is_multi_char
|
||||
&& text.contains('\u{00ad}')
|
||||
&& text.chars().all(|ch| {
|
||||
matches!(
|
||||
ch,
|
||||
'-' | '\u{00ad}' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2212}'
|
||||
)
|
||||
})
|
||||
{
|
||||
return "-".to_string();
|
||||
}
|
||||
|
||||
text
|
||||
}
|
||||
|
||||
fn hex_to_unicode_scalar(hex: &str) -> Option<u32> {
|
||||
let text = hex_to_unicode_string(hex)?;
|
||||
let mut chars = text.chars();
|
||||
let ch = chars.next()?;
|
||||
if chars.next().is_none() {
|
||||
Some(ch as u32)
|
||||
} else {
|
||||
Some(result)
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1095,9 +1155,7 @@ fn build_gid_to_unicode(face: &ttf_parser::Face<'_>) -> Option<HashMap<u16, char
|
||||
/// Build a ToUnicodeCMap from pdf.js built-in binary CMaps (bcmaps).
|
||||
fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
let name = format!("Adobe-{}-UCS2.bcmap", ordering);
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(name);
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&name)?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -1105,13 +1163,14 @@ fn build_cmap_from_builtin_cmap(ordering: &str) -> Option<ToUnicodeCMap> {
|
||||
cmap.code_byte_length = 2;
|
||||
debug!(
|
||||
"Built-in CMap {}: char_map={} ranges={}",
|
||||
path.display(),
|
||||
name,
|
||||
cmap.char_map.len(),
|
||||
cmap.ranges.len()
|
||||
);
|
||||
Some(cmap)
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
if let Ok(dir) = std::env::var("PDF_INSPECTOR_BCMAPS_DIR") {
|
||||
let p = PathBuf::from(dir);
|
||||
@@ -1128,6 +1187,18 @@ fn find_bcmaps_dir() -> Option<PathBuf> {
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let path = find_bcmaps_dir()?.join(name);
|
||||
std::fs::read(path).ok().map(Cow::Owned)
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
fn read_builtin_cmap_file(name: &str) -> Option<Cow<'static, [u8]>> {
|
||||
let file = BUILTIN_CMAPS.get_file(name)?;
|
||||
Some(Cow::Borrowed(file.contents()))
|
||||
}
|
||||
|
||||
fn parse_binary_cmap(data: &[u8]) -> Result<ToUnicodeCMap, String> {
|
||||
let mut stream = BinaryCMapStream::new(data);
|
||||
let _header = stream.read_byte().ok_or("unexpected EOF in bcmap header")?;
|
||||
@@ -1438,9 +1509,7 @@ fn parse_encoding_cmap_object(obj: &Object, doc: &Document) -> Option<EncodingCM
|
||||
}
|
||||
|
||||
fn load_builtin_encoding_cmap(name: &str) -> Option<EncodingCMap> {
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
parse_binary_cmap_encoding(&data).ok()
|
||||
}
|
||||
|
||||
@@ -1725,9 +1794,7 @@ fn load_builtin_cmap_by_name(name: &str) -> Option<ToUnicodeCMap> {
|
||||
if !name.ends_with("UCS2") {
|
||||
return None;
|
||||
}
|
||||
let dir = find_bcmaps_dir()?;
|
||||
let path = dir.join(format!("{}.bcmap", name));
|
||||
let data = std::fs::read(&path).ok()?;
|
||||
let data = read_builtin_cmap_file(&format!("{}.bcmap", name))?;
|
||||
let mut cmap = parse_binary_cmap(&data).ok()?;
|
||||
if cmap.char_map.is_empty() && cmap.ranges.is_empty() {
|
||||
return None;
|
||||
@@ -2607,6 +2674,97 @@ endbfrange
|
||||
assert_eq!(cmap.lookup(0x0005), Some("C".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_surrogate_pair_emoji() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<16> <D83CDF1F>
|
||||
<9D> <D83CDFAD>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.code_byte_length, 1);
|
||||
assert_eq!(cmap.lookup(0x16), Some("🌟".to_string()));
|
||||
assert_eq!(cmap.lookup(0x9D), Some("🎭".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfrange_surrogate_pair_base() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<C8> <C9> <D83CDFD8>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.code_byte_length, 1);
|
||||
assert_eq!(cmap.lookup(0xC8), Some("🏘".to_string()));
|
||||
assert_eq!(cmap.lookup(0xC9), Some("🏙".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfrange_preserves_single_hyphen_like_base() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<21> <22> <2013>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("–".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("—".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_spaced_destination_hex_without_control_noise() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
3 beginbfchar
|
||||
<21> < 0009 000d 0020 00a0 >
|
||||
<22> < 002d 00ad 2010 >
|
||||
<23> <00a0>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("\t".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("-".to_string()));
|
||||
assert_eq!(cmap.lookup(0x23), Some("\u{00a0}".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_preserves_valid_multi_character_destinations() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
4 beginbfchar
|
||||
<21> <002d002d>
|
||||
<22> <20132013>
|
||||
<23> <002000a0>
|
||||
<24> <00660069>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("--".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("––".to_string()));
|
||||
assert_eq!(cmap.lookup(0x23), Some(" \u{00a0}".to_string()));
|
||||
assert_eq!(cmap.lookup(0x24), Some("fi".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remap_to_sequential() {
|
||||
// Simulate a broken CMap where GIDs are from pre-subsetting:
|
||||
|
||||
+36
-7
@@ -116,6 +116,14 @@ pub struct TextItem {
|
||||
pub is_bold: bool,
|
||||
/// Whether the font is italic
|
||||
pub is_italic: bool,
|
||||
/// Whether the text is underlined (drawn rule/thin rect under the
|
||||
/// baseline — PDFs have no underline font flag, so this is detected
|
||||
/// geometrically after extraction; see `extractor::underline`).
|
||||
pub is_underline: bool,
|
||||
/// Whether the text is struck out (drawn rule/thin rect crossing the
|
||||
/// glyphs at mid x-height). Same geometric detection as underline,
|
||||
/// different vertical window; see `extractor::underline`.
|
||||
pub is_strikeout: bool,
|
||||
/// Type of item (text, image, link)
|
||||
pub item_type: ItemType,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator.
|
||||
@@ -137,12 +145,17 @@ pub struct TextLine {
|
||||
|
||||
impl TextLine {
|
||||
pub fn text(&self) -> String {
|
||||
self.text_with_formatting(false, false)
|
||||
self.text_with_formatting(false, false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic markdown formatting
|
||||
pub fn text_with_formatting(&self, format_bold: bool, format_italic: bool) -> String {
|
||||
if !format_bold && !format_italic {
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_underline: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -151,6 +164,7 @@ impl TextLine {
|
||||
let mut result = String::new();
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -176,9 +190,13 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes
|
||||
let item_bold = format_bold && item.is_bold;
|
||||
let item_italic = format_italic && item.is_italic;
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -189,6 +207,10 @@ impl TextLine {
|
||||
result.push_str("**");
|
||||
current_bold = false;
|
||||
}
|
||||
if current_underline && !item_underline {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -196,6 +218,10 @@ impl TextLine {
|
||||
}
|
||||
|
||||
// Open new styles
|
||||
if item_underline && !current_underline {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -215,6 +241,9 @@ impl TextLine {
|
||||
if current_bold {
|
||||
result.push_str("**");
|
||||
}
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
BIN
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+291
-3
@@ -2,13 +2,14 @@
|
||||
|
||||
use pdf_inspector::detector::{estimate_page_count_from_bytes, DetectionConfig, ScanStrategy};
|
||||
use pdf_inspector::extractor::group_into_lines;
|
||||
use pdf_inspector::types::ItemType;
|
||||
use pdf_inspector::types::TextLine;
|
||||
use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, process_pdf_mem,
|
||||
process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions, PdfType,
|
||||
TextItem,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
@@ -103,6 +104,8 @@ fn make_text_item(text: &str, x: f32, y: f32, font_size: f32, page: u32) -> Text
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -128,6 +131,8 @@ fn make_text_item_with_font(
|
||||
page,
|
||||
is_bold: is_bold_font(font),
|
||||
is_italic: is_italic_font(font),
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -332,6 +337,7 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
#[test]
|
||||
fn test_markdown_options_default() {
|
||||
let opts = MarkdownOptions::default();
|
||||
assert_eq!(opts.profile, pdf_inspector::MarkdownProfile::Fidelity);
|
||||
assert!(opts.detect_headers);
|
||||
assert!(opts.detect_lists);
|
||||
assert!(opts.detect_code);
|
||||
@@ -341,6 +347,7 @@ fn test_markdown_options_default() {
|
||||
#[test]
|
||||
fn test_markdown_options_custom() {
|
||||
let opts = MarkdownOptions {
|
||||
profile: pdf_inspector::MarkdownProfile::Compact,
|
||||
detect_headers: false,
|
||||
detect_lists: true,
|
||||
detect_code: false,
|
||||
@@ -356,6 +363,7 @@ fn test_markdown_options_custom() {
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!opts.detect_headers);
|
||||
assert_eq!(opts.profile, pdf_inspector::MarkdownProfile::Compact);
|
||||
assert!(opts.detect_lists);
|
||||
assert!(!opts.detect_code);
|
||||
assert_eq!(opts.base_font_size, Some(14.0));
|
||||
@@ -1097,6 +1105,7 @@ fn test_pages_needing_ocr_field_accessible() {
|
||||
title: None,
|
||||
ocr_recommended: false,
|
||||
pages_needing_ocr: Vec::new(),
|
||||
ocr_reasons_by_page: std::collections::BTreeMap::new(),
|
||||
};
|
||||
assert!(detection_result.pages_needing_ocr.is_empty());
|
||||
|
||||
@@ -1106,6 +1115,7 @@ fn test_pages_needing_ocr_field_accessible() {
|
||||
page_count: 1,
|
||||
processing_time_ms: 0,
|
||||
pages_needing_ocr: vec![1, 3],
|
||||
ocr_reasons_by_page: Vec::new(),
|
||||
title: None,
|
||||
confidence: 1.0,
|
||||
layout: pdf_inspector::LayoutComplexity::default(),
|
||||
@@ -1352,6 +1362,32 @@ fn test_extract_regions_mem_identity_h_needs_ocr() {
|
||||
);
|
||||
}
|
||||
|
||||
/// ParseBench `text_simple__att10k.pdf` (issue #118): the producer authored a
|
||||
/// broken ToUnicode CMap that shifts every character by a per-range constant,
|
||||
/// and the embedded subset font has no `cmap` table to recover from. The
|
||||
/// resulting ciphertext is 100% printable ASCII, so it must be caught by the
|
||||
/// substitution-cipher statistics and routed to OCR instead of served silently.
|
||||
#[test]
|
||||
fn test_extract_pages_mem_shifted_cipher_tounicode_needs_ocr() {
|
||||
let buf = std::fs::read("tests/fixtures/shifted_cipher_tounicode.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert!(
|
||||
result.pages[0].needs_ocr,
|
||||
"shifted-cipher garbled page should be flagged needs_ocr"
|
||||
);
|
||||
assert!(
|
||||
result.pages[0].markdown.is_empty(),
|
||||
"garbled markdown should be suppressed"
|
||||
);
|
||||
assert_eq!(result.pages_needing_ocr, vec![1]);
|
||||
assert_eq!(
|
||||
result.pages[0].ocr_reason.as_deref(),
|
||||
Some("suspected_garbled_text")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_multiple_regions_per_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
@@ -1648,6 +1684,33 @@ fn test_bits_pilani_page8_table_detection() {
|
||||
assert!(!region.needs_ocr, "Page 8 table should still be detected");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_uses_line_grid() {
|
||||
// Stroked-grid table (m/l/S path operators forming a 2x2 grid).
|
||||
// The heuristic text-only detector handles the same cells already,
|
||||
// so this guards that the line-backed path doesn't regress: the
|
||||
// markdown still contains all four data cells.
|
||||
let buf = synthetic_vector_grid_pdf(false);
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(0, vec![[40.0, 50.0, 220.0, 760.0]])]).unwrap();
|
||||
let region = &results[0].regions[0];
|
||||
assert!(
|
||||
!region.needs_ocr,
|
||||
"stroked-grid table should be extracted, got needs_ocr=true"
|
||||
);
|
||||
for tok in ["A1", "B1", "A2", "B2"] {
|
||||
assert!(
|
||||
region.text.contains(tok),
|
||||
"expected '{tok}' in output, got: {}",
|
||||
region.text
|
||||
);
|
||||
}
|
||||
assert!(
|
||||
region.text.contains('|'),
|
||||
"expected pipe-delimited markdown"
|
||||
);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// extract_tables_with_structure_mem tests (TSR-aware path)
|
||||
// =========================================================================
|
||||
@@ -3364,3 +3427,228 @@ fn test_synthetic_type0_broken_tounicode_emits_fffd_not_latin1_mojibake() {
|
||||
result.pages_needing_ocr
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Image XObject emission
|
||||
// ============================================================================
|
||||
|
||||
/// Build a minimal PDF containing one Image XObject placed at a known CTM.
|
||||
/// `image_ctm` is the 6-element matrix applied to the unit square by the
|
||||
/// `Do` operator (per PDF spec section 8.9.5 "Image Coordinate System").
|
||||
/// For an axis-aligned image at `(x, y)` with size `w × h`, that's
|
||||
/// `[w, 0, 0, h, x, y]`.
|
||||
fn make_pdf_with_image(image_ctm: [f32; 6]) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
fn add_stream_object(
|
||||
pdf: &mut Vec<u8>,
|
||||
offsets: &mut Vec<usize>,
|
||||
id: usize,
|
||||
dict: &str,
|
||||
stream_bytes: &[u8],
|
||||
) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(
|
||||
format!("<< {} /Length {} >>\nstream\n", dict, stream_bytes.len()).as_bytes(),
|
||||
);
|
||||
pdf.extend_from_slice(stream_bytes);
|
||||
pdf.extend_from_slice(b"\nendstream\nendobj\n");
|
||||
}
|
||||
|
||||
// 1: catalog → 2: pages → 3: page with XObject /Im0 → 4: content stream
|
||||
// 5: font → 6: image XObject (1×1 grayscale)
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] \
|
||||
/Resources << /Font << /F1 5 0 R >> /XObject << /Im0 6 0 R >> >> \
|
||||
/Contents 4 0 R >>",
|
||||
);
|
||||
let [a, b, c, d, e, f] = image_ctm;
|
||||
// BT/ET around a small text item just so the page isn't classified as
|
||||
// image-only (which would route to a different code path). Then save
|
||||
// graphics state, apply the image CTM, invoke Im0, restore.
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 100 700 Td (Hi) Tj ET\nq {} {} {} {} {} {} cm /Im0 Do Q",
|
||||
a, b, c, d, e, f
|
||||
);
|
||||
add_stream_object(&mut pdf, &mut offsets, 4, "", content.as_bytes());
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
// 1×1 grayscale image; the single byte is mid-gray. Contents don't
|
||||
// matter to the extractor — it only cares about the XObject's
|
||||
// /Subtype and the CTM at the `Do` operator.
|
||||
let image_pixel = [128u8];
|
||||
add_stream_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"/Type /XObject /Subtype /Image /Width 1 /Height 1 \
|
||||
/ColorSpace /DeviceGray /BitsPerComponent 8",
|
||||
&image_pixel,
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_text_with_positions_emits_image_bboxes() {
|
||||
// Place a 200×100 image at (50, 600) in PDF user space (origin
|
||||
// bottom-left). The Do operator applies the CTM to a unit square,
|
||||
// so for an axis-aligned image, CTM = [w, 0, 0, h, x, y].
|
||||
let pdf = make_pdf_with_image([200.0, 0.0, 0.0, 100.0, 50.0, 600.0]);
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract");
|
||||
|
||||
let images: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|i| matches!(i.item_type, ItemType::Image))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
images.len(),
|
||||
1,
|
||||
"expected exactly one Image item, got items: {:?}",
|
||||
items
|
||||
.iter()
|
||||
.map(|i| (&i.text, &i.item_type))
|
||||
.collect::<Vec<_>>()
|
||||
);
|
||||
let img = images[0];
|
||||
assert!((img.x - 50.0).abs() < 0.01, "x={}", img.x);
|
||||
assert!((img.y - 600.0).abs() < 0.01, "y={}", img.y);
|
||||
assert!((img.width - 200.0).abs() < 0.01, "width={}", img.width);
|
||||
assert!((img.height - 100.0).abs() < 0.01, "height={}", img.height);
|
||||
assert_eq!(img.page, 1);
|
||||
// text field carries the legacy `[Image: <resource-name>]` form that
|
||||
// the markdown emitter already knows how to parse.
|
||||
assert_eq!(img.text, "[Image: Im0]");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_image_xobject_bbox_handles_rotated_ctm() {
|
||||
// 90° rotation CTM: a unit square at the origin maps to a square
|
||||
// rotated counter-clockwise about (0,0), then translated to (200, 300).
|
||||
// For a 100×100 image, that's CTM = [0, 100, -100, 0, 200, 300]
|
||||
// (apply the rotation: (1,0) → (0,100); (0,1) → (-100,0)).
|
||||
// The page-space corners are:
|
||||
// (0,0) → (200, 300)
|
||||
// (1,0) → (200, 400)
|
||||
// (1,1) → (100, 400)
|
||||
// (0,1) → (100, 300)
|
||||
// → AABB: x=100..200 (w=100), y=300..400 (h=100).
|
||||
let pdf = make_pdf_with_image([0.0, 100.0, -100.0, 0.0, 200.0, 300.0]);
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract");
|
||||
let img = items
|
||||
.iter()
|
||||
.find(|i| matches!(i.item_type, ItemType::Image))
|
||||
.expect("image item");
|
||||
assert!((img.x - 100.0).abs() < 0.01, "x={}", img.x);
|
||||
assert!((img.y - 300.0).abs() < 0.01, "y={}", img.y);
|
||||
assert!((img.width - 100.0).abs() < 0.01, "width={}", img.width);
|
||||
assert!((img.height - 100.0).abs() < 0.01, "height={}", img.height);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_image_emission_does_not_change_default_markdown() {
|
||||
// Default `MarkdownOptions::include_images = false` — adding image
|
||||
// emission MUST NOT make `extract_pages_markdown` start producing
|
||||
// `![Image: …]` placeholders for everyone. Existing callers that
|
||||
// upgrade should see no diff in their markdown.
|
||||
let pdf = make_pdf_with_image([200.0, 0.0, 0.0, 100.0, 50.0, 600.0]);
|
||||
let result = extract_pages_markdown_mem(&pdf, None).expect("extract");
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert!(
|
||||
!result.pages[0].markdown.contains("Image:"),
|
||||
"default markdown leaked an image placeholder: {:?}",
|
||||
result.pages[0].markdown
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_options_default_has_include_images_false() {
|
||||
// Explicit assertion so anyone flipping this back catches it in CI.
|
||||
// See `MarkdownOptions::default` in src/markdown/mod.rs for the
|
||||
// long-form rationale.
|
||||
let opts = MarkdownOptions::default();
|
||||
assert!(!opts.include_images);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encrypted_pdf_decrypts_with_correct_password() {
|
||||
let path = "tests/fixtures/encrypted-secret123.pdf";
|
||||
|
||||
// No password: the file is encrypted and can't be read.
|
||||
let no_pw = process_pdf_with_options(path, PdfOptions::new());
|
||||
assert!(
|
||||
matches!(no_pw, Err(PdfError::Encrypted)),
|
||||
"expected Encrypted without a password, got {no_pw:?}"
|
||||
);
|
||||
|
||||
// Wrong password: still rejected.
|
||||
let wrong = process_pdf_with_options(path, PdfOptions::new().password("wrong"));
|
||||
assert!(
|
||||
matches!(wrong, Err(PdfError::Encrypted)),
|
||||
"expected Encrypted with a wrong password, got {wrong:?}"
|
||||
);
|
||||
|
||||
// Correct password: decrypts and extracts real content.
|
||||
let ok = process_pdf_with_options(path, PdfOptions::new().password("secret123"))
|
||||
.expect("correct password should decrypt");
|
||||
let md = ok.markdown.unwrap_or_default();
|
||||
// Assert a stable fixture token so a garbled-but-long extraction (the
|
||||
// encrypted-stream regression this guards) still fails the test.
|
||||
assert!(
|
||||
md.contains("Procurement"),
|
||||
"decrypted markdown should contain the fixture's real text, got {} chars",
|
||||
md.len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pdf_options_debug_redacts_password() {
|
||||
let opts = PdfOptions::new().password("secret123");
|
||||
let dbg = format!("{opts:?}");
|
||||
assert!(
|
||||
!dbg.contains("secret123"),
|
||||
"password leaked in Debug: {dbg}"
|
||||
);
|
||||
assert!(dbg.contains("REDACTED"), "expected redaction marker: {dbg}");
|
||||
}
|
||||
|
||||
@@ -188,7 +188,7 @@
|
||||
|156|23/7|General renovation works to Block B at Belonie Secondary School|MOE|Belvedere Builders|SR869,505.75|
|
||||
|157|30/7|Procurement of Engine Block and Crankshaft for Engine A11|PUC|Ras Tek Pvt Ltd|Euro798,650.00|
|
||||
|158|30/7|procurement of Wartsila Engine spares|PUC|Wartsila Eastern Africa ltd|Euro158,424.00|
|
||||
|159|30/7|Proposed walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|
||||
|159|30/7|Proposed walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|
||||
|160|30/7|Procurement of transfer pump control panel|PUC|CA Engineering Consultancy Pte Ltd|SGD14,600.00|
|
||||
|161|30/7|Consultancy service for North to South Victoria Bye- Pass road and utilities organisation|MLUH|Sonnel Seychelles LTD|SR1,332,000.00|
|
||||
|162 AUG|30/7|Procurement of the supply of sodium cardonate|PUC|HPL Chemical LTD|USD42,600.00|
|
||||
@@ -237,7 +237,7 @@
|
||||
|201|24/9|Procurement of vehicle x 2|SLTA|Abhaye Valabhji Pty Ltd|SR1000.000.00|
|
||||
||OCT|||||
|
||||
|202|1/10|Proposed new traffic lane to 5th June Avenue|SLTA|Divy Constrution|SR2,864,589.00|
|
||||
|203|1/10||Proposed Walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|
||||
|203|1/10||Proposed Walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot(Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|
||||
|204|1/10|Proposed Reconstrcution of Burnt House-Au Cap|MLUH|Furui Construction|SR946,130.00|
|
||||
|205|1/10|Variation on the project associated with the procurement of seven 100m3/day containerised plant|PUC|Tornado Group (UAE)|USD172,500.00|
|
||||
|206|1/10|Works on the breaker system at Bel Omber desalination plant|PUC|United Concrete Products (Sey)Ltd|SR1,998,993.11|
|
||||
|
||||
@@ -10,7 +10,7 @@ Department of the Treasury **Internal Revenue Service**
|
||||
|
||||
### This publication contains:
|
||||
|
||||
**Form 4070A, Employee’s Daily Record of** Tips **Form 4070, Employee’s Report of Tips to** Employer
|
||||
**Form 4070A,** Employee’s Daily Record of Tips **Form 4070,** Employee’s Report of Tips to Employer
|
||||
|
||||
For the period
|
||||
|
||||
@@ -22,7 +22,9 @@ Name and address of employee
|
||||
|
||||
**Publication 1244 (Rev. 7-96)** Cat. No. 44472W
|
||||
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use Form 4070A, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—If you** receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use Form 4070, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
### Instructions
|
||||
|
||||
You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
|
||||
*(continued on inside of back cover)*
|
||||
|
||||
@@ -30,14 +32,14 @@ Form **4070A** Employee’s Daily Record of Tips (Rev. July 1996) **This is a vo
|
||||
|
||||
Establishment name (if different)
|
||||
|
||||
Date Date **a. Tips received**
|
||||
Date Date **a.** Tips received
|
||||
|
||||
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
|
||||
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
|
||||
tips of directly from customers received other employees paid tips rec’d. entry and other employees 1 2 3 4 5 **Subtotals** **For Paperwork Reduction Act Notice, see Instructions on the back of Form 4070. Page 1**
|
||||
|
||||
Date Date **a. Tips received**
|
||||
Date Date **a.** Tips received
|
||||
|
||||
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
|
||||
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
|
||||
tips of directly from customers received other employees paid tips rec’d. entry and other employees
|
||||
|
||||
7 8 9 10 11 12 13 14 15 **Subtotals**
|
||||
@@ -48,11 +50,11 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
|
||||
**Page 3**
|
||||
|
||||
27 28 29 30 31 **Subtotals** **from pages** **1, 2, and 3** **Totals**
|
||||
27 28 29 30 31 **Subtotals from pages** **1, 2, and 3** **Totals**
|
||||
|
||||
**1.** Report total cash tips (col. a) on Form 4070, line 1.
|
||||
**2.** Report total credit card tips (col. b) on Form 4070, line 2.
|
||||
**3.** Report total tips paid out (col. c) on Form 4070, line 3. **Page 4**
|
||||
**1.** Report total cash tips (col. **a**) on Form 4070, line **1.**
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
|
||||
@@ -66,15 +68,15 @@ Employer’s name and address (include establishment name, if different) **1** C
|
||||
|
||||
**3** Tips paid out
|
||||
|
||||
Month or shorter period in which tips were received **4** Net tips (lines 1 + 2 - 3) from, 19, to, 19 Signature Date
|
||||
Month or shorter period in which tips were received **4** Net tips (lines **1 + 2 - 3**) from, 19, to, 19 Signature Date
|
||||
|
||||
**Paperwork Reduction Act Notice.—We ask for the** information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: Recordkeeping—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; Learning **about the law—each form, 2 min.; Preparing Form 4070,** 13 min.; Form 4070A, 55 min.; and Copying and **providing Form 4070, 10 min.; Form 4070A, 14 min.** If you have comments concerning the accuracy of these time estimates or suggestions for making these
|
||||
**Paperwork Reduction Act Notice.—**We ask for the information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: **Recordkeeping**—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; **Learning** **about the law**—each form, 2 min.; **Preparing** Form 4070, 13 min.; Form 4070A, 55 min.; and **Copying and** **providing** Form 4070, 10 min.; Form 4070A, 14 min. If you have comments concerning the accuracy of these time estimates or suggestions for making these
|
||||
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—Use this form to report tips you receive to** your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See Pub. 531, Reporting Tip Income, for more information. You can get additional copies of Pub. 1244, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—**Use this form to report tips you receive to your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See **Pub. 531**, Reporting Tip Income, for more information. You can get additional copies of **Pub. 1244**, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
|
||||
**Instructions (continued)**
|
||||
<u>Instructions (continued)</u>
|
||||
|
||||
**Unreported Tips.—If you received tips of $20 or** more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you must use Form 1040 and Form 4137, Social Security and Medicare Tax on Unreported Tip Income, to report them. You may not use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act cannot use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—Get Pub. 531, Reporting** Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—If you do not keep a daily** record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
**Unreported Tips.—**If you received tips of $20 or more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you **must** use Form 1040 and **Form 4137,** Social Security and Medicare Tax on Unreported Tip Income, to report them. You may **not** use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act **cannot** use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—**Get **Pub. 531,** Reporting Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—**If you do not keep a daily record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
|
||||
### Instructions (continued)
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
8 4 Z E L L / L U R I E R E A L E S T A T E C E N T E R
|
||||
|
||||
**Table I: Cap rate correlations** **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
**Table I:** Cap rate correlations **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
|
||||
* Based on 25 years of data for the 10-yrT & S&P DivYld; and 14 years for BBB.
|
||||
**Figure 1:** NCREIF cap rates vs. 10-yearTreasury
|
||||
@@ -20,7 +20,9 @@ R E V I E W 8 5
|
||||
|
||||
**Figure 2:** Capratespreadsover10-yearTreasury
|
||||
|
||||
**Basis Points -200** -400
|
||||
**Basis Points** -200
|
||||
|
||||
-400
|
||||
|
||||
-600
|
||||
|
||||
@@ -32,7 +34,7 @@ R E V I E W 8 5
|
||||
|
||||
1982 1986 1990 1994 1998 2002 2006
|
||||
|
||||
**Table II: Correlationsofspreadsbypropertytype** **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
**Table II:** Correlationsofspreadsbypropertytype **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
|
||||
||Multifamily|Industrial|CBD Office|
|
||||
|---|---|---|---|
|
||||
|
||||
+19
-28
@@ -1,17 +1,8 @@
|
||||
(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows:
|
||||
§1.1563-3T Rules for determining stock ownership (temporary).
|
||||
|
||||
(a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)
|
||||
through (d)(2)(iii). (iv) Statement. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
|
||||
|
||||
(A) A description of each of the controlled groups in which the corporation
|
||||
could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and
|
||||
|
||||
(B) The following representation: [INSERT NAME AND EMPLOYER
|
||||
IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].
|
||||
|
||||
(v) Election-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of
|
||||
this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in
|
||||
||||(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows: §1.1563-3T Rules for determining stock ownership (temporary). (a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)|
|
||||
|---|---|---|---|
|
||||
||through (d)(2)(iii).|||
|
||||
||(iv)|Statement|. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include-- (A) A description of each of the controlled groups in which the corporation could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and (B) The following representation: [INSERT NAME AND EMPLOYER IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].|
|
||||
||(v)|Election|-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in|
|
||||
|
||||
|termination of membership in the controlled group in which such corporation has||
|
||||
|---|---|
|
||||
@@ -29,48 +20,48 @@ this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of
|
||||
Federal income tax return (including any amended return filed on or before the due date (including extensions) of such original return) timely filed on or after May 30,
|
||||
|
||||
2006.
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: §1.6012-2 Corporations required to make returns of income.
|
||||
(2) <u>Expiration date</u>. The applicability of this section will expire on May 26,
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: <u>§1.6012-2 Corporations required to make returns of income</u>.
|
||||
* * * * *
|
||||
(c) [Reserved]. For further guidance, see §1.6012-2T(c).
|
||||
* * * * *
|
||||
(k) [Reserved]. For further guidance, see §1.6012-2T(k)(1).
|
||||
|
||||
Par. 52. Section 1.6012-2T is added to read as follows: §1.6012-2T Corporations required to make returns of income (temporary).
|
||||
Par. 52. Section 1.6012-2T is added to read as follows: <u>§1.6012-2T Corporations required to make returns of income (temporary)</u>.
|
||||
|
||||
(a) through (b) [Reserved]. For further guidance, see §1.6012-2(a) through
|
||||
(b).
|
||||
(c) Insurance companies-- (1) Domestic life insurance companies-- (i) In
|
||||
general. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
|
||||
<u>general</u>. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
|
||||
|
||||
(A) A copy of its annual statement which shows the reserves used by the
|
||||
company in computing the taxable income reported on its return; and
|
||||
|
||||
(B) A copy of Schedule A (real estate) and of Schedule D (bonds and stocks),
|
||||
or any successor thereto, of such annual statement. (ii) Mutual savings banks. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
|
||||
or any successor thereto, of such annual statement. (ii) <u>Mutual savings banks</u>. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
|
||||
|
||||
(2) Domestic nonlife insurance companies. Every domestic insurance
|
||||
(2) <u>Domestic nonlife insurance companies</u>. Every domestic insurance
|
||||
company other than a life insurance company shall make a return on Form 1120PC. This includes organizations described in section 501(m)(1) that provide commercial- type insurance and organizations described in section 833. Except as provided in paragraph (c)(4) of this section, such company shall file with its return a copy of its
|
||||
|
||||
annual statement (or a pro forma annual statement), including the underwriting and investment exhibit for the year covered by such return.
|
||||
|
||||
(3) Foreign insurance companies. The provisions of paragraphs (c)(1) and
|
||||
(3) <u>Foreign insurance companies</u>. The provisions of paragraphs (c)(1) and
|
||||
(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company.
|
||||
(4) Exception for insurance companies filing their Federal income tax returns
|
||||
electronically. If an insurance company described in paragraph (c)(1), (c)(2), or
|
||||
(4) <u>Exception for insurance companies filing their Federal income tax returns</u>
|
||||
<u>electronically</u>. If an insurance company described in paragraph (c)(1), (c)(2), or
|
||||
|
||||
(c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e).
|
||||
(5) Definition. For purposes of this section, the term annual statement means
|
||||
(5) <u>Definition</u>. For purposes of this section, the term <u>annual statement</u> means
|
||||
the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of
|
||||
|
||||
Columbia. The term annual statement also includes a pro forma annual statement if the insurance company is not required to file the NAIC annual statement.
|
||||
|
||||
(d) through (j) [Reserved]. For further guidance, see §1.6012-2(d) through (j).
|
||||
(k) Effective date-- (1) Applicability date. This section applies to any original
|
||||
(k) <u>Effective date</u>-- (1) <u>Applicability date</u>. This section applies to any original
|
||||
Federal income tax return (including any amended return filed on or before the due date (including extensions) of such original return) timely filed on or after May 30,
|
||||
|
||||
2006.
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
(2) <u>Expiration date</u>. The applicability of this section will expire on May 26,
|
||||
2009.
|
||||
|
||||
|||Par. 53. For each entry in the “Location” column of the following table,|
|
||||
@@ -165,7 +156,7 @@ section and paragraph
|
||||
PART 602--OMB CONTROL NUMBERS UNDER THE PAPERWORK REDUCTION ACT Par. 54. The authority citation for part 602 continues to read as follows: Authority: 26 U.S.C. 7805. Par. 55. In §602.101, paragraph (b) is amended to read as follows:
|
||||
|
||||
1. The following entries to the table are removed:
|
||||
§602.101 OMB Control numbers.
|
||||
<u>§602.101 OMB Control numbers</u>.
|
||||
|
||||
* * * * *
|
||||
(b) * * *
|
||||
@@ -180,7 +171,7 @@ CFR part or section where Current OMB identified or described control No.
|
||||
1.1081-11………………………………………………………………. 1545-2019
|
||||
* * * * * **______________________________________________________________**
|
||||
2. The following entries are added in numerical order to the table:
|
||||
§602.101 OMB Control numbers.
|
||||
<u>§602.101 OMB Control numbers</u>.
|
||||
|
||||
* * * * *
|
||||
(b) * * *
|
||||
|
||||
@@ -6,27 +6,29 @@
|
||||
|
||||
#### Thermodynamic Properties
|
||||
|
||||
**of**
|
||||
|
||||
®
|
||||
**of** ®
|
||||
|
||||
# Freon 12
|
||||
|
||||
**(R-12)** **Technical Information** **Technical Information**
|
||||
##### (R-12)
|
||||
|
||||
##### Technical Information Technical Information
|
||||
|
||||
**®** **Thermodynamic Properties of Freon 12 Refrigerant** **(R-12)** **SI Units**
|
||||
|
||||
Tables of the thermodynamic **Units** properties of R-12 have been developed and are presented here. P = Pressure in kPa. Absolute This information is based on values calculated using the NIST REFPROP T = Temperature in Celcius Database (McLinden, M.O., Klein,
|
||||
|
||||
S.A., Lemmon, E.W., and Peskin, Vf = Fluid (liquid) specific volume
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998). kilograms per cubic meter
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998).
|
||||
kilograms per cubic meter
|
||||
|
||||
##### H = Enthalpy (kJ/kg)
|
||||
|
||||
##### S = Entropy (kJ/kg.K)
|
||||
|
||||
##### Physical Properties
|
||||
|
||||
|Chemical Formula|CCl2F2|
|
||||
|Chemical Formula|CCl₂F₂|
|
||||
|---|---|
|
||||
|Molecular mass|120.91|
|
||||
|Boiling Point At one atmosphere|-29.75°C|
|
||||
@@ -43,9 +45,9 @@ l
|
||||
|
||||
**Freon** **®** **12 Saturation Properties-Temperature Table**
|
||||
|
||||
|Temp|Pressure||Volume|||Density||Enthalpy|||Entropy|Temp|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|°C|[kPa]|[m3 Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m3] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|Temp|Pressure||Volume||Density||Enthalpy|||Entropy|Temp|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|°C|[kPa]|[m³ Liquid v f|/kg] Vapour v g|[kg/m³ Liquid d f|] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|
||||
|-100|1.2|0.0006|10.0000|1679.0|0.100|113.3|192.8|306.1|0.6077|1.7210|-100|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
Generated
+1304
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,37 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "0.1.2"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "README.md"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[dependencies]
|
||||
console_error_panic_hook = "0.1"
|
||||
js-sys = "0.3"
|
||||
pdf-inspector = { path = ".." }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde-wasm-bindgen = "0.6"
|
||||
wasm-bindgen = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
wasm-bindgen-test = "0.3"
|
||||
|
||||
[profile.release]
|
||||
codegen-units = 1
|
||||
lto = true
|
||||
opt-level = "s"
|
||||
strip = true
|
||||
|
||||
[package.metadata.wasm-pack.profile.release]
|
||||
# Rust 1.95 emits bulk-memory instructions that the binaryen bundled with
|
||||
# wasm-pack 0.15.0 does not yet validate. rustc still performs the release,
|
||||
# size, and LTO optimizations above.
|
||||
wasm-opt = false
|
||||
@@ -0,0 +1,58 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
|
||||
Third-party notices
|
||||
===================
|
||||
|
||||
Adobe CMaps
|
||||
-----------
|
||||
|
||||
The WebAssembly binary embeds binary CMaps derived from Adobe CMap resources.
|
||||
|
||||
Copyright 1990-2009 Adobe Systems Incorporated.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
|
||||
Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
|
||||
Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
|
||||
Neither the name of Adobe Systems Incorporated nor the names of its
|
||||
contributors may be used to endorse or promote products derived from this
|
||||
software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGE.
|
||||
@@ -0,0 +1,60 @@
|
||||
# @firecrawl/pdf-inspector-wasm
|
||||
|
||||
Browser WebAssembly bindings for [pdf-inspector](https://github.com/firecrawl/pdf-inspector). Classify PDFs and extract structured Markdown locally from a `Uint8Array`, using the same Rust core as the native Node.js, Python, and Rust packages.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-wasm
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```ts
|
||||
import init, { processPdf } from "@firecrawl/pdf-inspector-wasm";
|
||||
|
||||
await init();
|
||||
|
||||
const response = await fetch("/annual-report.pdf");
|
||||
const pdf = new Uint8Array(await response.arrayBuffer());
|
||||
const result = processPdf(pdf);
|
||||
|
||||
console.log(result.pdfType);
|
||||
console.log(result.markdown);
|
||||
```
|
||||
|
||||
Pass options when you need selected pages or compact Markdown:
|
||||
|
||||
```ts
|
||||
const result = processPdf(pdf, {
|
||||
pages: [1, 3, 5],
|
||||
profile: "compact",
|
||||
includePageMarkers: true,
|
||||
});
|
||||
```
|
||||
|
||||
The package also exports:
|
||||
|
||||
- `detectPdf(pdf, options?)` for detection without extraction.
|
||||
- `classifyPdf(pdf)` for the lightweight result shape shared with the native Node.js API.
|
||||
- `extractText(pdf)` for plain text.
|
||||
- `version()` for the WASM package version.
|
||||
|
||||
## Browser behavior
|
||||
|
||||
- Parsing runs locally. PDF bytes are not uploaded anywhere.
|
||||
- The build is single-threaded and does not require cross-origin isolation.
|
||||
- CMaps are embedded so CJK font decoding does not depend on a filesystem.
|
||||
- Extraction is synchronous after `init()`. For large documents, call it from a Web Worker to keep the UI responsive.
|
||||
- Image-only documents still require a separate OCR step.
|
||||
|
||||
## Build from source
|
||||
|
||||
```bash
|
||||
cargo install wasm-pack --version 0.15.0 --locked
|
||||
wasm-pack build wasm --target web --scope firecrawl --release
|
||||
```
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
+440
@@ -0,0 +1,440 @@
|
||||
use pdf_inspector::{
|
||||
LayoutComplexity, MarkdownProfile, PageOcrReasons, PdfOptions, PdfProcessResult, PdfType,
|
||||
ProcessMode,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wasm_bindgen::prelude::*;
|
||||
|
||||
#[wasm_bindgen(typescript_custom_section)]
|
||||
const TYPESCRIPT_TYPES: &str = r#"
|
||||
export type PdfType = "TextBased" | "Scanned" | "ImageBased" | "Mixed";
|
||||
export type MarkdownProfile = "fidelity" | "compact";
|
||||
|
||||
export interface ProcessOptions {
|
||||
/** Restrict extraction to these 1-indexed page numbers. */
|
||||
pages?: number[];
|
||||
/** Password for an encrypted PDF. */
|
||||
password?: string;
|
||||
/** Source-faithful output by default, or compact output for fewer tokens. */
|
||||
profile?: MarkdownProfile;
|
||||
/** Insert `<!-- Page N -->` markers between pages. */
|
||||
includePageMarkers?: boolean;
|
||||
/** Include image placeholders in Markdown output. */
|
||||
includeImages?: boolean;
|
||||
}
|
||||
|
||||
export interface PageOcrReasons {
|
||||
/** 1-indexed page number. */
|
||||
page: number;
|
||||
reasons: string[];
|
||||
}
|
||||
|
||||
export interface LayoutComplexity {
|
||||
isComplex: boolean;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithTables: number[];
|
||||
/** 1-indexed page numbers. */
|
||||
pagesWithColumns: number[];
|
||||
}
|
||||
|
||||
export interface PdfProcessResult {
|
||||
pdfType: PdfType;
|
||||
markdown?: string;
|
||||
pageCount: number;
|
||||
processingTimeMs: number;
|
||||
/** 1-indexed page numbers. */
|
||||
pagesNeedingOcr: number[];
|
||||
ocrReasonsByPage: PageOcrReasons[];
|
||||
title?: string;
|
||||
confidence: number;
|
||||
layout: LayoutComplexity;
|
||||
hasEncodingIssues: boolean;
|
||||
}
|
||||
|
||||
export interface PdfClassification {
|
||||
pdfType: PdfType;
|
||||
pageCount: number;
|
||||
/** 0-indexed page numbers, matching the native Node.js API. */
|
||||
pagesNeedingOcr: number[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export function processPdf(data: Uint8Array, options?: ProcessOptions): PdfProcessResult;
|
||||
export function detectPdf(data: Uint8Array, options?: Pick<ProcessOptions, "password">): PdfProcessResult;
|
||||
export function classifyPdf(data: Uint8Array): PdfClassification;
|
||||
export function extractText(data: Uint8Array): string;
|
||||
export function version(): string;
|
||||
"#;
|
||||
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
#[serde(default, rename_all = "camelCase", deny_unknown_fields)]
|
||||
struct WasmProcessOptions {
|
||||
pages: Option<Vec<u32>>,
|
||||
password: Option<String>,
|
||||
profile: Option<WasmMarkdownProfile>,
|
||||
include_page_markers: Option<bool>,
|
||||
include_images: Option<bool>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
enum WasmMarkdownProfile {
|
||||
Fidelity,
|
||||
Compact,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPageOcrReasons {
|
||||
page: u32,
|
||||
reasons: Vec<String>,
|
||||
}
|
||||
|
||||
impl From<PageOcrReasons> for WasmPageOcrReasons {
|
||||
fn from(value: PageOcrReasons) -> Self {
|
||||
Self {
|
||||
page: value.page,
|
||||
reasons: value.reasons,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmLayoutComplexity {
|
||||
is_complex: bool,
|
||||
pages_with_tables: Vec<u32>,
|
||||
pages_with_columns: Vec<u32>,
|
||||
}
|
||||
|
||||
impl From<LayoutComplexity> for WasmLayoutComplexity {
|
||||
fn from(value: LayoutComplexity) -> Self {
|
||||
Self {
|
||||
is_complex: value.is_complex,
|
||||
pages_with_tables: value.pages_with_tables,
|
||||
pages_with_columns: value.pages_with_columns,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfProcessResult {
|
||||
pdf_type: &'static str,
|
||||
markdown: Option<String>,
|
||||
page_count: u32,
|
||||
processing_time_ms: f64,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
ocr_reasons_by_page: Vec<WasmPageOcrReasons>,
|
||||
title: Option<String>,
|
||||
confidence: f64,
|
||||
layout: WasmLayoutComplexity,
|
||||
has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
impl From<PdfProcessResult> for WasmPdfProcessResult {
|
||||
fn from(value: PdfProcessResult) -> Self {
|
||||
Self {
|
||||
pdf_type: pdf_type_name(value.pdf_type),
|
||||
markdown: value.markdown,
|
||||
page_count: value.page_count,
|
||||
processing_time_ms: value.processing_time_ms as f64,
|
||||
pages_needing_ocr: value.pages_needing_ocr,
|
||||
ocr_reasons_by_page: value
|
||||
.ocr_reasons_by_page
|
||||
.into_iter()
|
||||
.map(Into::into)
|
||||
.collect(),
|
||||
title: value.title,
|
||||
confidence: value.confidence as f64,
|
||||
layout: value.layout.into(),
|
||||
has_encoding_issues: value.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
struct WasmPdfClassification {
|
||||
pdf_type: &'static str,
|
||||
page_count: u32,
|
||||
pages_needing_ocr: Vec<u32>,
|
||||
confidence: f64,
|
||||
}
|
||||
|
||||
fn pdf_type_name(pdf_type: PdfType) -> &'static str {
|
||||
match pdf_type {
|
||||
PdfType::TextBased => "TextBased",
|
||||
PdfType::Scanned => "Scanned",
|
||||
PdfType::ImageBased => "ImageBased",
|
||||
PdfType::Mixed => "Mixed",
|
||||
}
|
||||
}
|
||||
|
||||
fn js_error(context: &str, error: impl std::fmt::Display) -> JsValue {
|
||||
js_sys::Error::new(&format!("{context}: {error}")).into()
|
||||
}
|
||||
|
||||
fn deserialize_options(value: JsValue) -> Result<WasmProcessOptions, JsValue> {
|
||||
if value.is_undefined() || value.is_null() {
|
||||
return Ok(WasmProcessOptions::default());
|
||||
}
|
||||
|
||||
serde_wasm_bindgen::from_value(value).map_err(|error| js_error("invalid options", error))
|
||||
}
|
||||
|
||||
fn build_options(value: JsValue, mode: ProcessMode) -> Result<PdfOptions, JsValue> {
|
||||
let options = deserialize_options(value)?;
|
||||
if options
|
||||
.pages
|
||||
.as_ref()
|
||||
.is_some_and(|pages| pages.contains(&0))
|
||||
{
|
||||
return Err(js_error(
|
||||
"invalid options",
|
||||
"pages are 1-indexed; page 0 is invalid",
|
||||
));
|
||||
}
|
||||
|
||||
let mut result = PdfOptions::new().mode(mode);
|
||||
if let Some(pages) = options.pages {
|
||||
result = result.pages(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(profile) = options.profile {
|
||||
result.markdown.profile = match profile {
|
||||
WasmMarkdownProfile::Fidelity => MarkdownProfile::Fidelity,
|
||||
WasmMarkdownProfile::Compact => MarkdownProfile::Compact,
|
||||
};
|
||||
}
|
||||
if let Some(include_page_markers) = options.include_page_markers {
|
||||
result.markdown.include_page_numbers = include_page_markers;
|
||||
}
|
||||
if let Some(include_images) = options.include_images {
|
||||
result.markdown.include_images = include_images;
|
||||
}
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn serialize<T: Serialize>(value: &T) -> Result<JsValue, JsValue> {
|
||||
serde_wasm_bindgen::to_value(value).map_err(|error| js_error("serialize result", error))
|
||||
}
|
||||
|
||||
fn initialize() {
|
||||
console_error_panic_hook::set_once();
|
||||
}
|
||||
|
||||
/// Process PDF bytes entirely inside WebAssembly.
|
||||
#[wasm_bindgen(js_name = processPdf, skip_typescript)]
|
||||
pub fn process_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::Full)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("process PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Classify PDF bytes without extracting text or producing Markdown.
|
||||
#[wasm_bindgen(js_name = detectPdf, skip_typescript)]
|
||||
pub fn detect_pdf(data: &[u8], options: JsValue) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let options = build_options(options, ProcessMode::DetectOnly)?;
|
||||
let started = js_sys::Date::now();
|
||||
let mut result = pdf_inspector::process_pdf_mem_with_options(data, options)
|
||||
.map_err(|error| js_error("detect PDF", error))?;
|
||||
result.processing_time_ms = (js_sys::Date::now() - started).max(0.0) as u64;
|
||||
serialize(&WasmPdfProcessResult::from(result))
|
||||
}
|
||||
|
||||
/// Return the lightweight classification shape used by the native Node API.
|
||||
#[wasm_bindgen(js_name = classifyPdf, skip_typescript)]
|
||||
pub fn classify_pdf(data: &[u8]) -> Result<JsValue, JsValue> {
|
||||
initialize();
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(data).map_err(|error| js_error("classify PDF", error))?;
|
||||
serialize(&WasmPdfClassification {
|
||||
pdf_type: pdf_type_name(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes without Markdown conversion.
|
||||
#[wasm_bindgen(js_name = extractText, skip_typescript)]
|
||||
pub fn extract_text(data: &[u8]) -> Result<String, JsValue> {
|
||||
initialize();
|
||||
let items = pdf_inspector::extractor::extract_text_with_positions_mem(data)
|
||||
.map_err(|error| js_error("extract text", error))?;
|
||||
Ok(
|
||||
pdf_inspector::extractor::group_into_lines_preserving_all_text(items)
|
||||
.into_iter()
|
||||
.map(|line| line.text())
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n"),
|
||||
)
|
||||
}
|
||||
|
||||
/// Return the WebAssembly package version.
|
||||
#[wasm_bindgen(skip_typescript)]
|
||||
pub fn version() -> String {
|
||||
env!("CARGO_PKG_VERSION").to_string()
|
||||
}
|
||||
|
||||
#[cfg(all(test, target_arch = "wasm32"))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use js_sys::Reflect;
|
||||
use wasm_bindgen_test::*;
|
||||
|
||||
const TEXT_PDF: &[u8] = include_bytes!("../../tests/fixtures/thermo-freon12.pdf");
|
||||
const ENCRYPTED_PDF: &[u8] = include_bytes!("../../tests/fixtures/encrypted-secret123.pdf");
|
||||
|
||||
fn synthetic_korea1_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /Font << /F0 5 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
|
||||
// Adobe-Korea1 CID 1086 (0x043E) maps to U+AC00 (Korean syllable GA).
|
||||
// There is deliberately no ToUnicode stream: decoding must use the
|
||||
// embedded predefined CMap rather than lopdf's plain-text fallback.
|
||||
// Korea1 CIDs 21 and 19 map to ASCII "4" and "2". Place them near
|
||||
// the bottom edge so they look exactly like a numeric page footer.
|
||||
let content = "BT /F0 12 Tf 50 100 Td <043E> Tj 0 -60 Td <00150013> Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type0 /BaseFont /SyntheticKorea1 /Encoding /Identity-H /DescendantFonts [6 0 R] >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /SyntheticKorea1 /CIDSystemInfo << /Registry (Adobe) /Ordering (Korea1) /Supplement 2 >> /FontDescriptor 7 0 R /DW 1000 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /FontDescriptor /FontName /SyntheticKorea1 /Flags 4 /FontBBox [-100 -200 1000 900] /ItalicAngle 0 /Ascent 800 /Descent -200 /CapHeight 700 /StemV 80 >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn processes_pdf_to_markdown() {
|
||||
let result = process_pdf(TEXT_PDF, JsValue::UNDEFINED).expect("process PDF");
|
||||
let pdf_type = Reflect::get(&result, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn rejects_non_pdf_bytes() {
|
||||
assert!(process_pdf(b"not a PDF", JsValue::UNDEFINED).is_err());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn classifies_and_extracts_plain_text() {
|
||||
let classification = classify_pdf(TEXT_PDF).expect("classify PDF");
|
||||
let pdf_type = Reflect::get(&classification, &JsValue::from_str("pdfType"))
|
||||
.expect("pdfType")
|
||||
.as_string()
|
||||
.expect("pdfType string");
|
||||
let text = extract_text(TEXT_PDF).expect("extract text");
|
||||
|
||||
assert_eq!(pdf_type, "TextBased");
|
||||
assert!(!text.is_empty());
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn extracts_cjk_and_preserves_numeric_page_footer() {
|
||||
let text = extract_text(&synthetic_korea1_pdf()).expect("extract predefined CMap text");
|
||||
|
||||
assert_eq!(text, "가\n42");
|
||||
}
|
||||
|
||||
#[wasm_bindgen_test]
|
||||
fn opens_encrypted_pdf_with_password() {
|
||||
assert!(process_pdf(ENCRYPTED_PDF, JsValue::UNDEFINED).is_err());
|
||||
|
||||
let options = js_sys::Object::new();
|
||||
Reflect::set(
|
||||
&options,
|
||||
&JsValue::from_str("password"),
|
||||
&JsValue::from_str("secret123"),
|
||||
)
|
||||
.expect("set password");
|
||||
let result = process_pdf(ENCRYPTED_PDF, options.into()).expect("process encrypted PDF");
|
||||
let markdown = Reflect::get(&result, &JsValue::from_str("markdown"))
|
||||
.expect("markdown")
|
||||
.as_string()
|
||||
.expect("markdown string");
|
||||
|
||||
assert!(!markdown.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user