Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a813de5561 | ||
|
|
217e1745fb | ||
|
|
04abab951f | ||
|
|
a410d5aa08 | ||
|
|
7747b3a086 | ||
|
|
ae6246ba0c | ||
|
|
2a7ad5891d | ||
|
|
3fb545284b | ||
|
|
1c32e4bd69 | ||
|
|
8121ae97ce | ||
|
|
98990cc550 |
+21
-14
@@ -14,13 +14,15 @@ jobs:
|
||||
name: Test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
@@ -32,11 +34,12 @@ jobs:
|
||||
name: Format
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: rustfmt
|
||||
|
||||
- name: Check formatting
|
||||
@@ -49,15 +52,16 @@ jobs:
|
||||
name: Clippy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: clippy
|
||||
|
||||
@@ -71,13 +75,15 @@ jobs:
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
key: build
|
||||
|
||||
@@ -88,16 +94,17 @@ jobs:
|
||||
name: WebAssembly
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: Swatinem/rust-cache@v2
|
||||
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
|
||||
with:
|
||||
workspaces: |
|
||||
wasm -> target
|
||||
|
||||
@@ -24,13 +24,13 @@ jobs:
|
||||
name: github-pages
|
||||
url: ${{ steps.deploy.outputs.page_url }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Upload site artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0
|
||||
with:
|
||||
path: site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deploy
|
||||
uses: actions/deploy-pages@v4
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5.0.0
|
||||
|
||||
@@ -27,7 +27,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -85,17 +85,19 @@ jobs:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
uses: rust-lang/crates-io-auth-action@c6f97d42243bad5fab37ca0427f495c86d5b1a18 # v1.0.5
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -98,21 +98,21 @@ jobs:
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Build wheel
|
||||
uses: PyO3/maturin-action@v1
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
target: ${{ matrix.target }}
|
||||
args: --release --out dist
|
||||
manylinux: auto
|
||||
|
||||
- name: Upload wheel
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: wheels-${{ matrix.target }}
|
||||
path: dist/*.whl
|
||||
@@ -124,16 +124,16 @@ jobs:
|
||||
name: Build sdist
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Build sdist
|
||||
uses: PyO3/maturin-action@v1
|
||||
uses: PyO3/maturin-action@e83996d129638aa358a18fbd1dfb82f0b0fb5d3b # v1.51.0
|
||||
with:
|
||||
command: sdist
|
||||
args: --out dist
|
||||
|
||||
- name: Upload sdist
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: sdist
|
||||
path: dist/*.tar.gz
|
||||
@@ -149,7 +149,7 @@ jobs:
|
||||
id-token: write
|
||||
steps:
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: dist
|
||||
merge-multiple: true
|
||||
@@ -158,7 +158,7 @@ jobs:
|
||||
run: ls -la dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
with:
|
||||
packages-dir: dist
|
||||
# Tolerate already-uploaded files so a manual re-run can complete
|
||||
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -71,14 +71,15 @@ jobs:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: wasm32-unknown-unknown
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
@@ -61,31 +61,61 @@ jobs:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
# napi-cross builds gnu targets against an old glibc sysroot for
|
||||
# broad distro compatibility; musl targets cross-compile with
|
||||
# zig via cargo-zigbuild (napi's -x flag).
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
build-flags: --use-napi-cross
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
build-flags: -x
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
|
||||
with:
|
||||
toolchain: stable
|
||||
targets: ${{ matrix.target }}
|
||||
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
|
||||
with:
|
||||
bun-version: latest
|
||||
|
||||
- name: Install zig
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: mlugg/setup-zig@d1434d08867e3ee9daa34448df10607b98908d29 # v2.2.1
|
||||
with:
|
||||
version: 0.14.1
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.target, 'musl')
|
||||
uses: taiki-e/install-action@67729d5c413db75907f0ad1e39bb04b9c868ff60 # v2.85.7
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
tool: cargo-zigbuild
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
~/.napi-rs
|
||||
napi/target/
|
||||
key: ${{ runner.os }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
key: ${{ matrix.target }}-cargo-napi-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-napi-
|
||||
${{ matrix.target }}-cargo-napi-
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: napi
|
||||
@@ -93,10 +123,10 @@ jobs:
|
||||
|
||||
- name: Build native addon
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release
|
||||
run: bunx napi build --platform --release --target ${{ matrix.target }} ${{ matrix.build-flags }}
|
||||
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
@@ -104,7 +134,7 @@ jobs:
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
@@ -112,23 +142,68 @@ jobs:
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
smoke-test:
|
||||
name: Smoke test ${{ matrix.target }}
|
||||
needs: [check-version, build]
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-gnu
|
||||
- os: ubuntu-24.04-arm
|
||||
target: aarch64-unknown-linux-musl
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Download native binary
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi
|
||||
|
||||
- name: Download generated JS bindings
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: js-bindings
|
||||
path: napi
|
||||
|
||||
# musl binaries must load under a real musl libc, so run inside Alpine.
|
||||
- name: Run smoke test (Alpine)
|
||||
if: contains(matrix.target, 'musl')
|
||||
run: docker run --rm -v "$PWD:/repo" -w /repo/napi node:24-alpine node test.mjs
|
||||
|
||||
- name: Run smoke test
|
||||
if: ${{ !contains(matrix.target, 'musl') }}
|
||||
working-directory: napi
|
||||
run: node test.mjs
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: [check-version, build]
|
||||
needs: [check-version, build, smoke-test]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
with:
|
||||
node-version: '24'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Download all artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
@@ -154,9 +229,12 @@ jobs:
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"linux-x64-musl": { os: ["linux"], cpu: ["x64"], libc: ["musl"] },
|
||||
"linux-arm64-gnu": { os: ["linux"], cpu: ["arm64"], libc: ["glibc"] },
|
||||
"linux-arm64-musl": { os: ["linux"], cpu: ["arm64"], libc: ["musl"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
|
||||
+2
-2
@@ -46,14 +46,14 @@ ttf-parser = "0.25"
|
||||
# Native builds keep lopdf's parallel parser and CLI logging. Browser WASM is
|
||||
# deliberately single-threaded so it works without cross-origin isolation.
|
||||
[target.'cfg(not(target_arch = "wasm32"))'.dependencies]
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
lopdf = { version = "0.42.0", features = ["rayon"] }
|
||||
rayon = "1.10"
|
||||
env_logger = "0.11"
|
||||
|
||||
# Browser builds use JavaScript randomness for encrypted PDFs and embed the
|
||||
# bundled CMaps because there is no filesystem at runtime.
|
||||
[target.'cfg(target_arch = "wasm32")'.dependencies]
|
||||
lopdf = { version = "0.41.0", default-features = false, features = ["wasm_js"] }
|
||||
lopdf = { version = "0.42.0", default-features = false, features = ["wasm_js"] }
|
||||
include_dir = "0.7"
|
||||
|
||||
[dev-dependencies]
|
||||
|
||||
@@ -238,7 +238,7 @@ The converter handles:
|
||||
|---|---|
|
||||
| Headings (H1-H4) | Font size tiers relative to body text, with 0.5pt clustering |
|
||||
| Bold/italic | Font name patterns (Bold, Italic, Oblique) |
|
||||
| Bullet lists | `*`, `-`, `*`, `○`, `●`, `◦` prefixes |
|
||||
| Bullet lists | `•`, `-`, `*`, `○`, `●`, `◦` prefixes |
|
||||
| Numbered lists | `1.`, `1)`, `(1)` patterns |
|
||||
| Letter lists | `a.`, `a)`, `(a)` patterns |
|
||||
| Code blocks | Monospace fonts (Courier, Consolas, Monaco, Menlo, Fira Code, JetBrains Mono) and keyword detection |
|
||||
|
||||
+14
-3
@@ -111,7 +111,8 @@ class PdfResult: # process_pdf / detect_pdf
|
||||
markdown: str | None # extracted Markdown (None for detect_pdf)
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
title: str | None
|
||||
confidence: float # 0.0 - 1.0
|
||||
is_complex_layout: bool
|
||||
@@ -119,6 +120,10 @@ class PdfResult: # process_pdf / detect_pdf
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool # broken font encodings — consider OCR fallback
|
||||
|
||||
class PageOcrReasons: # per-page OCR diagnostics
|
||||
page: int # 1-indexed
|
||||
reasons: list[str] # machine-readable reason identifiers
|
||||
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
@@ -140,14 +145,20 @@ class TextItem: # extract_text_with_positions
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText: # extract_text_in_regions
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
ocr_reason: str | None # machine-readable OCR reason
|
||||
|
||||
class PageRegionTexts: # extract_text_in_regions
|
||||
page: int # 0-indexed
|
||||
regions: list[RegionText] # RegionText: text: str, needs_ocr: bool
|
||||
regions: list[RegionText]
|
||||
|
||||
class PagesExtractionResult: # extract_pages_markdown
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr, ocr_reason
|
||||
pages_with_tables: list[int] # 1-indexed
|
||||
pages_with_columns: list[int] # 1-indexed
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
is_complex: bool # any page has tables or multi-column layout
|
||||
```
|
||||
|
||||
Generated
+24
-2
@@ -499,11 +499,13 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"js-sys",
|
||||
"libc",
|
||||
"r-efi",
|
||||
"rand_core",
|
||||
"wasip2",
|
||||
"wasip3",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -557,6 +559,25 @@ version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
|
||||
|
||||
[[package]]
|
||||
name = "include_dir"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||
dependencies = [
|
||||
"include_dir_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir_macros"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.0"
|
||||
@@ -672,9 +693,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.41.0"
|
||||
version = "0.42.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -833,6 +854,7 @@ name = "pdf-inspector"
|
||||
version = "0.1.7"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
"log",
|
||||
"lopdf",
|
||||
"once_cell",
|
||||
|
||||
+7
-4
@@ -34,7 +34,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries for **Linux x64**, **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -116,9 +116,12 @@ Prebuilt binaries ship as platform-specific packages installed automatically via
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| Linux | x64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-x64-musl` |
|
||||
| Linux | ARM64 (glibc) | `@firecrawl/pdf-inspector-linux-arm64-gnu` |
|
||||
| Linux | ARM64 (musl/Alpine) | `@firecrawl/pdf-inspector-linux-arm64-musl` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
+6
-3
@@ -8,9 +8,12 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+10
-4
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.11.2",
|
||||
"version": "1.12.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -37,6 +37,9 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"x86_64-unknown-linux-musl",
|
||||
"aarch64-unknown-linux-gnu",
|
||||
"aarch64-unknown-linux-musl",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
]
|
||||
@@ -49,8 +52,11 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.2",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.2",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.2"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.12.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.12.0"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,9 @@ class PdfResult:
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed page numbers that need OCR."""
|
||||
ocr_reasons_by_page: list["PageOcrReasons"]
|
||||
"""Machine-readable OCR reasons by 1-indexed page."""
|
||||
title: Optional[str]
|
||||
confidence: float
|
||||
is_complex_layout: bool
|
||||
@@ -17,6 +20,13 @@ class PdfResult:
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool
|
||||
|
||||
class PageOcrReasons:
|
||||
"""OCR reasons for a single 1-indexed page."""
|
||||
page: int
|
||||
"""1-indexed page number."""
|
||||
reasons: list[str]
|
||||
"""Machine-readable OCR reason identifiers."""
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
@@ -47,6 +57,8 @@ class RegionText:
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
"""True when the text should not be trusted."""
|
||||
ocr_reason: Optional[str]
|
||||
"""Machine-readable OCR reason when the cause is known."""
|
||||
|
||||
class PageRegionTexts:
|
||||
"""Extracted text for one page's regions."""
|
||||
@@ -62,6 +74,8 @@ class PageMarkdown:
|
||||
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
|
||||
needs_ocr: bool
|
||||
"""True when text on this page is unreliable and OCR should be used instead."""
|
||||
ocr_reason: Optional[str]
|
||||
"""Machine-readable OCR reason when the cause is known."""
|
||||
|
||||
class PagesExtractionResult:
|
||||
"""Per-page markdown output with document-wide layout classification."""
|
||||
@@ -73,6 +87,8 @@ class PagesExtractionResult:
|
||||
"""1-indexed pages where multi-column layout was detected."""
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed pages that need OCR."""
|
||||
ocr_reasons_by_page: list[PageOcrReasons]
|
||||
"""Machine-readable OCR reasons by 1-indexed page."""
|
||||
is_complex: bool
|
||||
"""True if any page has tables or multi-column layout."""
|
||||
|
||||
|
||||
+32
-5
@@ -2,8 +2,8 @@
|
||||
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
|
||||
PdfType, ProcessMode, TextItem,
|
||||
extract_text_with_positions_pages_with_password, process_pdf_with_options, LayoutComplexity,
|
||||
PdfOptions, PdfType, ProcessMode, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
use std::env;
|
||||
@@ -103,9 +103,18 @@ fn format_items_json(items: &[TextItem]) -> String {
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_items_json(
|
||||
pdf_path: &str,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<String, pdf_inspector::PdfError> {
|
||||
extract_text_with_positions_pages_with_password(pdf_path, page_filter, password)
|
||||
.map(|items| format_items_json(&items))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::format_items_json;
|
||||
use super::{extract_items_json, format_items_json};
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
@@ -137,6 +146,24 @@ mod tests {
|
||||
assert!(json.contains(r#""item_type":"text""#));
|
||||
assert!(json.contains(r#""mcid":7"#));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn items_json_uses_supplied_pdf_password() {
|
||||
let path = "tests/fixtures/encrypted-secret123.pdf";
|
||||
|
||||
let without_password = extract_items_json(path, None, None);
|
||||
assert!(
|
||||
without_password.is_err(),
|
||||
"encrypted fixture unexpectedly extracted without a password"
|
||||
);
|
||||
|
||||
let json = extract_items_json(path, None, Some("secret123"))
|
||||
.expect("correct password should decrypt positioned text");
|
||||
assert!(
|
||||
json.contains("Procurement"),
|
||||
"decrypted item JSON should contain fixture text, got {json}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
@@ -257,8 +284,8 @@ fn main() {
|
||||
});
|
||||
|
||||
if items_json_output {
|
||||
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
|
||||
Ok(items) => println!("{}", format_items_json(&items)),
|
||||
match extract_items_json(pdf_path, page_filter.as_ref(), password.as_deref()) {
|
||||
Ok(json) => println!("{}", json),
|
||||
Err(e) => {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
process::exit(1);
|
||||
|
||||
+31
-219
@@ -150,134 +150,11 @@ static COURIER: &[(char, u16)] = &[
|
||||
('\u{FB02}', 600),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static COURIER_BOLD: &[(char, u16)] = &[
|
||||
(' ', 600), ('!', 600), ('"', 600), ('#', 600), ('$', 600), ('%', 600),
|
||||
('&', 600), ('\'', 600), ('(', 600), (')', 600), ('*', 600), ('+', 600),
|
||||
(',', 600), ('-', 600), ('.', 600), ('/', 600), ('0', 600), ('1', 600),
|
||||
('2', 600), ('3', 600), ('4', 600), ('5', 600), ('6', 600), ('7', 600),
|
||||
('8', 600), ('9', 600), (':', 600), (';', 600), ('<', 600), ('=', 600),
|
||||
('>', 600), ('?', 600), ('@', 600), ('A', 600), ('B', 600), ('C', 600),
|
||||
('D', 600), ('E', 600), ('F', 600), ('G', 600), ('H', 600), ('I', 600),
|
||||
('J', 600), ('K', 600), ('L', 600), ('M', 600), ('N', 600), ('O', 600),
|
||||
('P', 600), ('Q', 600), ('R', 600), ('S', 600), ('T', 600), ('U', 600),
|
||||
('V', 600), ('W', 600), ('X', 600), ('Y', 600), ('Z', 600), ('[', 600),
|
||||
('\\', 600), (']', 600), ('^', 600), ('_', 600), ('`', 600), ('a', 600),
|
||||
('b', 600), ('c', 600), ('d', 600), ('e', 600), ('f', 600), ('g', 600),
|
||||
('h', 600), ('i', 600), ('j', 600), ('k', 600), ('l', 600), ('m', 600),
|
||||
('n', 600), ('o', 600), ('p', 600), ('q', 600), ('r', 600), ('s', 600),
|
||||
('t', 600), ('u', 600), ('v', 600), ('w', 600), ('x', 600), ('y', 600),
|
||||
('z', 600), ('{', 600), ('|', 600), ('}', 600), ('~', 600), ('\u{00A1}', 600),
|
||||
('\u{00A2}', 600), ('\u{00A3}', 600), ('\u{00A4}', 600), ('\u{00A5}', 600), ('\u{00A6}', 600), ('\u{00A7}', 600),
|
||||
('\u{00A8}', 600), ('\u{00A9}', 600), ('\u{00AA}', 600), ('\u{00AB}', 600), ('\u{00AC}', 600), ('\u{00AE}', 600),
|
||||
('\u{00AF}', 600), ('\u{00B0}', 600), ('\u{00B1}', 600), ('\u{00B2}', 600), ('\u{00B3}', 600), ('\u{00B4}', 600),
|
||||
('\u{00B5}', 600), ('\u{00B6}', 600), ('\u{00B7}', 600), ('\u{00B8}', 600), ('\u{00B9}', 600), ('\u{00BA}', 600),
|
||||
('\u{00BB}', 600), ('\u{00BC}', 600), ('\u{00BD}', 600), ('\u{00BE}', 600), ('\u{00BF}', 600), ('\u{00C0}', 600),
|
||||
('\u{00C1}', 600), ('\u{00C2}', 600), ('\u{00C3}', 600), ('\u{00C4}', 600), ('\u{00C5}', 600), ('\u{00C6}', 600),
|
||||
('\u{00C7}', 600), ('\u{00C8}', 600), ('\u{00C9}', 600), ('\u{00CA}', 600), ('\u{00CB}', 600), ('\u{00CC}', 600),
|
||||
('\u{00CD}', 600), ('\u{00CE}', 600), ('\u{00CF}', 600), ('\u{00D0}', 600), ('\u{00D1}', 600), ('\u{00D2}', 600),
|
||||
('\u{00D3}', 600), ('\u{00D4}', 600), ('\u{00D5}', 600), ('\u{00D6}', 600), ('\u{00D7}', 600), ('\u{00D8}', 600),
|
||||
('\u{00D9}', 600), ('\u{00DA}', 600), ('\u{00DB}', 600), ('\u{00DC}', 600), ('\u{00DD}', 600), ('\u{00DE}', 600),
|
||||
('\u{00DF}', 600), ('\u{00E0}', 600), ('\u{00E1}', 600), ('\u{00E2}', 600), ('\u{00E3}', 600), ('\u{00E4}', 600),
|
||||
('\u{00E5}', 600), ('\u{00E6}', 600), ('\u{00E7}', 600), ('\u{00E8}', 600), ('\u{00E9}', 600), ('\u{00EA}', 600),
|
||||
('\u{00EB}', 600), ('\u{00EC}', 600), ('\u{00ED}', 600), ('\u{00EE}', 600), ('\u{00EF}', 600), ('\u{00F0}', 600),
|
||||
('\u{00F1}', 600), ('\u{00F2}', 600), ('\u{00F3}', 600), ('\u{00F4}', 600), ('\u{00F5}', 600), ('\u{00F6}', 600),
|
||||
('\u{00F7}', 600), ('\u{00F8}', 600), ('\u{00F9}', 600), ('\u{00FA}', 600), ('\u{00FB}', 600), ('\u{00FC}', 600),
|
||||
('\u{00FD}', 600), ('\u{00FE}', 600), ('\u{00FF}', 600), ('\u{0131}', 600), ('\u{0141}', 600), ('\u{0142}', 600),
|
||||
('\u{0152}', 600), ('\u{0153}', 600), ('\u{0160}', 600), ('\u{0161}', 600), ('\u{0178}', 600), ('\u{017D}', 600),
|
||||
('\u{017E}', 600), ('\u{0192}', 600), ('\u{02C6}', 600), ('\u{02C7}', 600), ('\u{02D8}', 600), ('\u{02D9}', 600),
|
||||
('\u{02DA}', 600), ('\u{02DB}', 600), ('\u{02DC}', 600), ('\u{02DD}', 600), ('\u{2013}', 600), ('\u{2014}', 600),
|
||||
('\u{2018}', 600), ('\u{2019}', 600), ('\u{201A}', 600), ('\u{201C}', 600), ('\u{201D}', 600), ('\u{201E}', 600),
|
||||
('\u{2020}', 600), ('\u{2021}', 600), ('\u{2022}', 600), ('\u{2026}', 600), ('\u{2030}', 600), ('\u{2039}', 600),
|
||||
('\u{203A}', 600), ('\u{2044}', 600), ('\u{20AC}', 600), ('\u{2122}', 600), ('\u{2212}', 600), ('\u{FB01}', 600),
|
||||
('\u{FB02}', 600),
|
||||
];
|
||||
static COURIER_BOLD: &[(char, u16)] = COURIER;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static COURIER_OBLIQUE: &[(char, u16)] = &[
|
||||
(' ', 600), ('!', 600), ('"', 600), ('#', 600), ('$', 600), ('%', 600),
|
||||
('&', 600), ('\'', 600), ('(', 600), (')', 600), ('*', 600), ('+', 600),
|
||||
(',', 600), ('-', 600), ('.', 600), ('/', 600), ('0', 600), ('1', 600),
|
||||
('2', 600), ('3', 600), ('4', 600), ('5', 600), ('6', 600), ('7', 600),
|
||||
('8', 600), ('9', 600), (':', 600), (';', 600), ('<', 600), ('=', 600),
|
||||
('>', 600), ('?', 600), ('@', 600), ('A', 600), ('B', 600), ('C', 600),
|
||||
('D', 600), ('E', 600), ('F', 600), ('G', 600), ('H', 600), ('I', 600),
|
||||
('J', 600), ('K', 600), ('L', 600), ('M', 600), ('N', 600), ('O', 600),
|
||||
('P', 600), ('Q', 600), ('R', 600), ('S', 600), ('T', 600), ('U', 600),
|
||||
('V', 600), ('W', 600), ('X', 600), ('Y', 600), ('Z', 600), ('[', 600),
|
||||
('\\', 600), (']', 600), ('^', 600), ('_', 600), ('`', 600), ('a', 600),
|
||||
('b', 600), ('c', 600), ('d', 600), ('e', 600), ('f', 600), ('g', 600),
|
||||
('h', 600), ('i', 600), ('j', 600), ('k', 600), ('l', 600), ('m', 600),
|
||||
('n', 600), ('o', 600), ('p', 600), ('q', 600), ('r', 600), ('s', 600),
|
||||
('t', 600), ('u', 600), ('v', 600), ('w', 600), ('x', 600), ('y', 600),
|
||||
('z', 600), ('{', 600), ('|', 600), ('}', 600), ('~', 600), ('\u{00A1}', 600),
|
||||
('\u{00A2}', 600), ('\u{00A3}', 600), ('\u{00A4}', 600), ('\u{00A5}', 600), ('\u{00A6}', 600), ('\u{00A7}', 600),
|
||||
('\u{00A8}', 600), ('\u{00A9}', 600), ('\u{00AA}', 600), ('\u{00AB}', 600), ('\u{00AC}', 600), ('\u{00AE}', 600),
|
||||
('\u{00AF}', 600), ('\u{00B0}', 600), ('\u{00B1}', 600), ('\u{00B2}', 600), ('\u{00B3}', 600), ('\u{00B4}', 600),
|
||||
('\u{00B5}', 600), ('\u{00B6}', 600), ('\u{00B7}', 600), ('\u{00B8}', 600), ('\u{00B9}', 600), ('\u{00BA}', 600),
|
||||
('\u{00BB}', 600), ('\u{00BC}', 600), ('\u{00BD}', 600), ('\u{00BE}', 600), ('\u{00BF}', 600), ('\u{00C0}', 600),
|
||||
('\u{00C1}', 600), ('\u{00C2}', 600), ('\u{00C3}', 600), ('\u{00C4}', 600), ('\u{00C5}', 600), ('\u{00C6}', 600),
|
||||
('\u{00C7}', 600), ('\u{00C8}', 600), ('\u{00C9}', 600), ('\u{00CA}', 600), ('\u{00CB}', 600), ('\u{00CC}', 600),
|
||||
('\u{00CD}', 600), ('\u{00CE}', 600), ('\u{00CF}', 600), ('\u{00D0}', 600), ('\u{00D1}', 600), ('\u{00D2}', 600),
|
||||
('\u{00D3}', 600), ('\u{00D4}', 600), ('\u{00D5}', 600), ('\u{00D6}', 600), ('\u{00D7}', 600), ('\u{00D8}', 600),
|
||||
('\u{00D9}', 600), ('\u{00DA}', 600), ('\u{00DB}', 600), ('\u{00DC}', 600), ('\u{00DD}', 600), ('\u{00DE}', 600),
|
||||
('\u{00DF}', 600), ('\u{00E0}', 600), ('\u{00E1}', 600), ('\u{00E2}', 600), ('\u{00E3}', 600), ('\u{00E4}', 600),
|
||||
('\u{00E5}', 600), ('\u{00E6}', 600), ('\u{00E7}', 600), ('\u{00E8}', 600), ('\u{00E9}', 600), ('\u{00EA}', 600),
|
||||
('\u{00EB}', 600), ('\u{00EC}', 600), ('\u{00ED}', 600), ('\u{00EE}', 600), ('\u{00EF}', 600), ('\u{00F0}', 600),
|
||||
('\u{00F1}', 600), ('\u{00F2}', 600), ('\u{00F3}', 600), ('\u{00F4}', 600), ('\u{00F5}', 600), ('\u{00F6}', 600),
|
||||
('\u{00F7}', 600), ('\u{00F8}', 600), ('\u{00F9}', 600), ('\u{00FA}', 600), ('\u{00FB}', 600), ('\u{00FC}', 600),
|
||||
('\u{00FD}', 600), ('\u{00FE}', 600), ('\u{00FF}', 600), ('\u{0131}', 600), ('\u{0141}', 600), ('\u{0142}', 600),
|
||||
('\u{0152}', 600), ('\u{0153}', 600), ('\u{0160}', 600), ('\u{0161}', 600), ('\u{0178}', 600), ('\u{017D}', 600),
|
||||
('\u{017E}', 600), ('\u{0192}', 600), ('\u{02C6}', 600), ('\u{02C7}', 600), ('\u{02D8}', 600), ('\u{02D9}', 600),
|
||||
('\u{02DA}', 600), ('\u{02DB}', 600), ('\u{02DC}', 600), ('\u{02DD}', 600), ('\u{2013}', 600), ('\u{2014}', 600),
|
||||
('\u{2018}', 600), ('\u{2019}', 600), ('\u{201A}', 600), ('\u{201C}', 600), ('\u{201D}', 600), ('\u{201E}', 600),
|
||||
('\u{2020}', 600), ('\u{2021}', 600), ('\u{2022}', 600), ('\u{2026}', 600), ('\u{2030}', 600), ('\u{2039}', 600),
|
||||
('\u{203A}', 600), ('\u{2044}', 600), ('\u{20AC}', 600), ('\u{2122}', 600), ('\u{2212}', 600), ('\u{FB01}', 600),
|
||||
('\u{FB02}', 600),
|
||||
];
|
||||
static COURIER_OBLIQUE: &[(char, u16)] = COURIER;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static COURIER_BOLDOBLIQUE: &[(char, u16)] = &[
|
||||
(' ', 600), ('!', 600), ('"', 600), ('#', 600), ('$', 600), ('%', 600),
|
||||
('&', 600), ('\'', 600), ('(', 600), (')', 600), ('*', 600), ('+', 600),
|
||||
(',', 600), ('-', 600), ('.', 600), ('/', 600), ('0', 600), ('1', 600),
|
||||
('2', 600), ('3', 600), ('4', 600), ('5', 600), ('6', 600), ('7', 600),
|
||||
('8', 600), ('9', 600), (':', 600), (';', 600), ('<', 600), ('=', 600),
|
||||
('>', 600), ('?', 600), ('@', 600), ('A', 600), ('B', 600), ('C', 600),
|
||||
('D', 600), ('E', 600), ('F', 600), ('G', 600), ('H', 600), ('I', 600),
|
||||
('J', 600), ('K', 600), ('L', 600), ('M', 600), ('N', 600), ('O', 600),
|
||||
('P', 600), ('Q', 600), ('R', 600), ('S', 600), ('T', 600), ('U', 600),
|
||||
('V', 600), ('W', 600), ('X', 600), ('Y', 600), ('Z', 600), ('[', 600),
|
||||
('\\', 600), (']', 600), ('^', 600), ('_', 600), ('`', 600), ('a', 600),
|
||||
('b', 600), ('c', 600), ('d', 600), ('e', 600), ('f', 600), ('g', 600),
|
||||
('h', 600), ('i', 600), ('j', 600), ('k', 600), ('l', 600), ('m', 600),
|
||||
('n', 600), ('o', 600), ('p', 600), ('q', 600), ('r', 600), ('s', 600),
|
||||
('t', 600), ('u', 600), ('v', 600), ('w', 600), ('x', 600), ('y', 600),
|
||||
('z', 600), ('{', 600), ('|', 600), ('}', 600), ('~', 600), ('\u{00A1}', 600),
|
||||
('\u{00A2}', 600), ('\u{00A3}', 600), ('\u{00A4}', 600), ('\u{00A5}', 600), ('\u{00A6}', 600), ('\u{00A7}', 600),
|
||||
('\u{00A8}', 600), ('\u{00A9}', 600), ('\u{00AA}', 600), ('\u{00AB}', 600), ('\u{00AC}', 600), ('\u{00AE}', 600),
|
||||
('\u{00AF}', 600), ('\u{00B0}', 600), ('\u{00B1}', 600), ('\u{00B2}', 600), ('\u{00B3}', 600), ('\u{00B4}', 600),
|
||||
('\u{00B5}', 600), ('\u{00B6}', 600), ('\u{00B7}', 600), ('\u{00B8}', 600), ('\u{00B9}', 600), ('\u{00BA}', 600),
|
||||
('\u{00BB}', 600), ('\u{00BC}', 600), ('\u{00BD}', 600), ('\u{00BE}', 600), ('\u{00BF}', 600), ('\u{00C0}', 600),
|
||||
('\u{00C1}', 600), ('\u{00C2}', 600), ('\u{00C3}', 600), ('\u{00C4}', 600), ('\u{00C5}', 600), ('\u{00C6}', 600),
|
||||
('\u{00C7}', 600), ('\u{00C8}', 600), ('\u{00C9}', 600), ('\u{00CA}', 600), ('\u{00CB}', 600), ('\u{00CC}', 600),
|
||||
('\u{00CD}', 600), ('\u{00CE}', 600), ('\u{00CF}', 600), ('\u{00D0}', 600), ('\u{00D1}', 600), ('\u{00D2}', 600),
|
||||
('\u{00D3}', 600), ('\u{00D4}', 600), ('\u{00D5}', 600), ('\u{00D6}', 600), ('\u{00D7}', 600), ('\u{00D8}', 600),
|
||||
('\u{00D9}', 600), ('\u{00DA}', 600), ('\u{00DB}', 600), ('\u{00DC}', 600), ('\u{00DD}', 600), ('\u{00DE}', 600),
|
||||
('\u{00DF}', 600), ('\u{00E0}', 600), ('\u{00E1}', 600), ('\u{00E2}', 600), ('\u{00E3}', 600), ('\u{00E4}', 600),
|
||||
('\u{00E5}', 600), ('\u{00E6}', 600), ('\u{00E7}', 600), ('\u{00E8}', 600), ('\u{00E9}', 600), ('\u{00EA}', 600),
|
||||
('\u{00EB}', 600), ('\u{00EC}', 600), ('\u{00ED}', 600), ('\u{00EE}', 600), ('\u{00EF}', 600), ('\u{00F0}', 600),
|
||||
('\u{00F1}', 600), ('\u{00F2}', 600), ('\u{00F3}', 600), ('\u{00F4}', 600), ('\u{00F5}', 600), ('\u{00F6}', 600),
|
||||
('\u{00F7}', 600), ('\u{00F8}', 600), ('\u{00F9}', 600), ('\u{00FA}', 600), ('\u{00FB}', 600), ('\u{00FC}', 600),
|
||||
('\u{00FD}', 600), ('\u{00FE}', 600), ('\u{00FF}', 600), ('\u{0131}', 600), ('\u{0141}', 600), ('\u{0142}', 600),
|
||||
('\u{0152}', 600), ('\u{0153}', 600), ('\u{0160}', 600), ('\u{0161}', 600), ('\u{0178}', 600), ('\u{017D}', 600),
|
||||
('\u{017E}', 600), ('\u{0192}', 600), ('\u{02C6}', 600), ('\u{02C7}', 600), ('\u{02D8}', 600), ('\u{02D9}', 600),
|
||||
('\u{02DA}', 600), ('\u{02DB}', 600), ('\u{02DC}', 600), ('\u{02DD}', 600), ('\u{2013}', 600), ('\u{2014}', 600),
|
||||
('\u{2018}', 600), ('\u{2019}', 600), ('\u{201A}', 600), ('\u{201C}', 600), ('\u{201D}', 600), ('\u{201E}', 600),
|
||||
('\u{2020}', 600), ('\u{2021}', 600), ('\u{2022}', 600), ('\u{2026}', 600), ('\u{2030}', 600), ('\u{2039}', 600),
|
||||
('\u{203A}', 600), ('\u{2044}', 600), ('\u{20AC}', 600), ('\u{2122}', 600), ('\u{2212}', 600), ('\u{FB01}', 600),
|
||||
('\u{FB02}', 600),
|
||||
];
|
||||
static COURIER_BOLDOBLIQUE: &[(char, u16)] = COURIER;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA: &[(char, u16)] = &[
|
||||
@@ -365,91 +242,9 @@ static HELVETICA_BOLD: &[(char, u16)] = &[
|
||||
('\u{FB02}', 611),
|
||||
];
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA_OBLIQUE: &[(char, u16)] = &[
|
||||
(' ', 278), ('!', 278), ('"', 355), ('#', 556), ('$', 556), ('%', 889),
|
||||
('&', 667), ('\'', 191), ('(', 333), (')', 333), ('*', 389), ('+', 584),
|
||||
(',', 278), ('-', 333), ('.', 278), ('/', 278), ('0', 556), ('1', 556),
|
||||
('2', 556), ('3', 556), ('4', 556), ('5', 556), ('6', 556), ('7', 556),
|
||||
('8', 556), ('9', 556), (':', 278), (';', 278), ('<', 584), ('=', 584),
|
||||
('>', 584), ('?', 556), ('@', 1015), ('A', 667), ('B', 667), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 722), ('I', 278),
|
||||
('J', 500), ('K', 667), ('L', 556), ('M', 833), ('N', 722), ('O', 778),
|
||||
('P', 667), ('Q', 778), ('R', 722), ('S', 667), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 944), ('X', 667), ('Y', 667), ('Z', 611), ('[', 278),
|
||||
('\\', 278), (']', 278), ('^', 469), ('_', 556), ('`', 333), ('a', 556),
|
||||
('b', 556), ('c', 500), ('d', 556), ('e', 556), ('f', 278), ('g', 556),
|
||||
('h', 556), ('i', 222), ('j', 222), ('k', 500), ('l', 222), ('m', 833),
|
||||
('n', 556), ('o', 556), ('p', 556), ('q', 556), ('r', 333), ('s', 500),
|
||||
('t', 278), ('u', 556), ('v', 500), ('w', 722), ('x', 500), ('y', 500),
|
||||
('z', 500), ('{', 334), ('|', 260), ('}', 334), ('~', 584), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 556), ('\u{00A3}', 556), ('\u{00A4}', 556), ('\u{00A5}', 556), ('\u{00A6}', 260), ('\u{00A7}', 556),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 737), ('\u{00AA}', 370), ('\u{00AB}', 556), ('\u{00AC}', 584), ('\u{00AE}', 737),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 584), ('\u{00B2}', 333), ('\u{00B3}', 333), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 556), ('\u{00B6}', 537), ('\u{00B7}', 278), ('\u{00B8}', 333), ('\u{00B9}', 333), ('\u{00BA}', 365),
|
||||
('\u{00BB}', 556), ('\u{00BC}', 834), ('\u{00BD}', 834), ('\u{00BE}', 834), ('\u{00BF}', 611), ('\u{00C0}', 667),
|
||||
('\u{00C1}', 667), ('\u{00C2}', 667), ('\u{00C3}', 667), ('\u{00C4}', 667), ('\u{00C5}', 667), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 278),
|
||||
('\u{00CD}', 278), ('\u{00CE}', 278), ('\u{00CF}', 278), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 584), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 667), ('\u{00DE}', 667),
|
||||
('\u{00DF}', 611), ('\u{00E0}', 556), ('\u{00E1}', 556), ('\u{00E2}', 556), ('\u{00E3}', 556), ('\u{00E4}', 556),
|
||||
('\u{00E5}', 556), ('\u{00E6}', 889), ('\u{00E7}', 500), ('\u{00E8}', 556), ('\u{00E9}', 556), ('\u{00EA}', 556),
|
||||
('\u{00EB}', 556), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 556),
|
||||
('\u{00F1}', 556), ('\u{00F2}', 556), ('\u{00F3}', 556), ('\u{00F4}', 556), ('\u{00F5}', 556), ('\u{00F6}', 556),
|
||||
('\u{00F7}', 584), ('\u{00F8}', 611), ('\u{00F9}', 556), ('\u{00FA}', 556), ('\u{00FB}', 556), ('\u{00FC}', 556),
|
||||
('\u{00FD}', 500), ('\u{00FE}', 556), ('\u{00FF}', 500), ('\u{0131}', 278), ('\u{0141}', 556), ('\u{0142}', 222),
|
||||
('\u{0152}', 1000), ('\u{0153}', 944), ('\u{0160}', 667), ('\u{0161}', 500), ('\u{0178}', 667), ('\u{017D}', 611),
|
||||
('\u{017E}', 500), ('\u{0192}', 556), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 556), ('\u{2014}', 1000),
|
||||
('\u{2018}', 222), ('\u{2019}', 222), ('\u{201A}', 222), ('\u{201C}', 333), ('\u{201D}', 333), ('\u{201E}', 333),
|
||||
('\u{2020}', 556), ('\u{2021}', 556), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 556), ('\u{2122}', 1000), ('\u{2212}', 584), ('\u{FB01}', 500),
|
||||
('\u{FB02}', 500),
|
||||
];
|
||||
static HELVETICA_OBLIQUE: &[(char, u16)] = HELVETICA;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static HELVETICA_BOLDOBLIQUE: &[(char, u16)] = &[
|
||||
(' ', 278), ('!', 333), ('"', 474), ('#', 556), ('$', 556), ('%', 889),
|
||||
('&', 722), ('\'', 238), ('(', 333), (')', 333), ('*', 389), ('+', 584),
|
||||
(',', 278), ('-', 333), ('.', 278), ('/', 278), ('0', 556), ('1', 556),
|
||||
('2', 556), ('3', 556), ('4', 556), ('5', 556), ('6', 556), ('7', 556),
|
||||
('8', 556), ('9', 556), (':', 333), (';', 333), ('<', 584), ('=', 584),
|
||||
('>', 584), ('?', 611), ('@', 975), ('A', 722), ('B', 722), ('C', 722),
|
||||
('D', 722), ('E', 667), ('F', 611), ('G', 778), ('H', 722), ('I', 278),
|
||||
('J', 556), ('K', 722), ('L', 611), ('M', 833), ('N', 722), ('O', 778),
|
||||
('P', 667), ('Q', 778), ('R', 722), ('S', 667), ('T', 611), ('U', 722),
|
||||
('V', 667), ('W', 944), ('X', 667), ('Y', 667), ('Z', 611), ('[', 333),
|
||||
('\\', 278), (']', 333), ('^', 584), ('_', 556), ('`', 333), ('a', 556),
|
||||
('b', 611), ('c', 556), ('d', 611), ('e', 556), ('f', 333), ('g', 611),
|
||||
('h', 611), ('i', 278), ('j', 278), ('k', 556), ('l', 278), ('m', 889),
|
||||
('n', 611), ('o', 611), ('p', 611), ('q', 611), ('r', 389), ('s', 556),
|
||||
('t', 333), ('u', 611), ('v', 556), ('w', 778), ('x', 556), ('y', 556),
|
||||
('z', 500), ('{', 389), ('|', 280), ('}', 389), ('~', 584), ('\u{00A1}', 333),
|
||||
('\u{00A2}', 556), ('\u{00A3}', 556), ('\u{00A4}', 556), ('\u{00A5}', 556), ('\u{00A6}', 280), ('\u{00A7}', 556),
|
||||
('\u{00A8}', 333), ('\u{00A9}', 737), ('\u{00AA}', 370), ('\u{00AB}', 556), ('\u{00AC}', 584), ('\u{00AE}', 737),
|
||||
('\u{00AF}', 333), ('\u{00B0}', 400), ('\u{00B1}', 584), ('\u{00B2}', 333), ('\u{00B3}', 333), ('\u{00B4}', 333),
|
||||
('\u{00B5}', 611), ('\u{00B6}', 556), ('\u{00B7}', 278), ('\u{00B8}', 333), ('\u{00B9}', 333), ('\u{00BA}', 365),
|
||||
('\u{00BB}', 556), ('\u{00BC}', 834), ('\u{00BD}', 834), ('\u{00BE}', 834), ('\u{00BF}', 611), ('\u{00C0}', 722),
|
||||
('\u{00C1}', 722), ('\u{00C2}', 722), ('\u{00C3}', 722), ('\u{00C4}', 722), ('\u{00C5}', 722), ('\u{00C6}', 1000),
|
||||
('\u{00C7}', 722), ('\u{00C8}', 667), ('\u{00C9}', 667), ('\u{00CA}', 667), ('\u{00CB}', 667), ('\u{00CC}', 278),
|
||||
('\u{00CD}', 278), ('\u{00CE}', 278), ('\u{00CF}', 278), ('\u{00D0}', 722), ('\u{00D1}', 722), ('\u{00D2}', 778),
|
||||
('\u{00D3}', 778), ('\u{00D4}', 778), ('\u{00D5}', 778), ('\u{00D6}', 778), ('\u{00D7}', 584), ('\u{00D8}', 778),
|
||||
('\u{00D9}', 722), ('\u{00DA}', 722), ('\u{00DB}', 722), ('\u{00DC}', 722), ('\u{00DD}', 667), ('\u{00DE}', 667),
|
||||
('\u{00DF}', 611), ('\u{00E0}', 556), ('\u{00E1}', 556), ('\u{00E2}', 556), ('\u{00E3}', 556), ('\u{00E4}', 556),
|
||||
('\u{00E5}', 556), ('\u{00E6}', 889), ('\u{00E7}', 556), ('\u{00E8}', 556), ('\u{00E9}', 556), ('\u{00EA}', 556),
|
||||
('\u{00EB}', 556), ('\u{00EC}', 278), ('\u{00ED}', 278), ('\u{00EE}', 278), ('\u{00EF}', 278), ('\u{00F0}', 611),
|
||||
('\u{00F1}', 611), ('\u{00F2}', 611), ('\u{00F3}', 611), ('\u{00F4}', 611), ('\u{00F5}', 611), ('\u{00F6}', 611),
|
||||
('\u{00F7}', 584), ('\u{00F8}', 611), ('\u{00F9}', 611), ('\u{00FA}', 611), ('\u{00FB}', 611), ('\u{00FC}', 611),
|
||||
('\u{00FD}', 556), ('\u{00FE}', 611), ('\u{00FF}', 556), ('\u{0131}', 278), ('\u{0141}', 611), ('\u{0142}', 278),
|
||||
('\u{0152}', 1000), ('\u{0153}', 944), ('\u{0160}', 667), ('\u{0161}', 556), ('\u{0178}', 667), ('\u{017D}', 611),
|
||||
('\u{017E}', 500), ('\u{0192}', 556), ('\u{02C6}', 333), ('\u{02C7}', 333), ('\u{02D8}', 333), ('\u{02D9}', 333),
|
||||
('\u{02DA}', 333), ('\u{02DB}', 333), ('\u{02DC}', 333), ('\u{02DD}', 333), ('\u{2013}', 556), ('\u{2014}', 1000),
|
||||
('\u{2018}', 278), ('\u{2019}', 278), ('\u{201A}', 278), ('\u{201C}', 500), ('\u{201D}', 500), ('\u{201E}', 500),
|
||||
('\u{2020}', 556), ('\u{2021}', 556), ('\u{2022}', 350), ('\u{2026}', 1000), ('\u{2030}', 1000), ('\u{2039}', 333),
|
||||
('\u{203A}', 333), ('\u{2044}', 167), ('\u{20AC}', 556), ('\u{2122}', 1000), ('\u{2212}', 584), ('\u{FB01}', 611),
|
||||
('\u{FB02}', 611),
|
||||
];
|
||||
static HELVETICA_BOLDOBLIQUE: &[(char, u16)] = HELVETICA_BOLD;
|
||||
|
||||
#[rustfmt::skip]
|
||||
static TIMES_ROMAN: &[(char, u16)] = &[
|
||||
@@ -771,6 +566,25 @@ static ZAPFDINGBATS_ENCODING: &[(u8, char)] = &[
|
||||
(0xFB, '\u{27BB}'), (0xFC, '\u{27BC}'), (0xFD, '\u{27BD}'), (0xFE, '\u{27BE}'),
|
||||
];
|
||||
|
||||
/// Every width table, for exhaustive testing.
|
||||
#[cfg(test)]
|
||||
static ALL_TABLES: &[(&str, &[(char, u16)])] = &[
|
||||
("COURIER", COURIER),
|
||||
("COURIER_BOLD", COURIER_BOLD),
|
||||
("COURIER_OBLIQUE", COURIER_OBLIQUE),
|
||||
("COURIER_BOLDOBLIQUE", COURIER_BOLDOBLIQUE),
|
||||
("HELVETICA", HELVETICA),
|
||||
("HELVETICA_BOLD", HELVETICA_BOLD),
|
||||
("HELVETICA_OBLIQUE", HELVETICA_OBLIQUE),
|
||||
("HELVETICA_BOLDOBLIQUE", HELVETICA_BOLDOBLIQUE),
|
||||
("TIMES_ROMAN", TIMES_ROMAN),
|
||||
("TIMES_BOLD", TIMES_BOLD),
|
||||
("TIMES_ITALIC", TIMES_ITALIC),
|
||||
("TIMES_BOLDITALIC", TIMES_BOLDITALIC),
|
||||
("SYMBOL", SYMBOL),
|
||||
("ZAPFDINGBATS", ZAPFDINGBATS),
|
||||
];
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -823,15 +637,13 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn tables_are_sorted_for_binary_search() {
|
||||
for table in [
|
||||
TIMES_ROMAN,
|
||||
TIMES_ITALIC,
|
||||
HELVETICA,
|
||||
COURIER,
|
||||
SYMBOL,
|
||||
ZAPFDINGBATS,
|
||||
] {
|
||||
assert!(table.windows(2).all(|w| w[0].0 < w[1].0));
|
||||
// Every table is queried by binary search, so all of them must be
|
||||
// sorted — not just a sample.
|
||||
for (name, table) in ALL_TABLES {
|
||||
assert!(
|
||||
table.windows(2).all(|w| w[0].0 < w[1].0),
|
||||
"{name} is not sorted"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+70
-3
@@ -198,9 +198,27 @@ pub(crate) fn build_type3_scales(
|
||||
let scale_y = (num(&matrix[2]).powi(2) + num(&matrix[3]).powi(2)).sqrt();
|
||||
let bbox_h = (num(&bbox[3]) - num(&bbox[1])).abs();
|
||||
let scale = bbox_h * scale_y;
|
||||
// Only store meaningful scales; degenerate bboxes ([0 0 0 0] is legal)
|
||||
// and near-1.0 factors keep the nominal size.
|
||||
if scale > 0.01 && (scale - 1.0).abs() > 0.05 {
|
||||
|
||||
// `scale` is the glyph box measured in text-space units. For a
|
||||
// self-consistent font it lands near 1.0 — the FontMatrix is the
|
||||
// reciprocal of the glyph-space em by construction — so the Tf
|
||||
// operand is already the rendered size and must be left alone.
|
||||
// A modest deviation is normal and must NOT trigger rescaling:
|
||||
// FontBBox is the glyph bounding box, not the em box, so it is
|
||||
// routinely somewhat smaller (descender..ascender ≈ 0.7) or larger
|
||||
// (tall accents > 1.0).
|
||||
//
|
||||
// Only a wildly inconsistent font gets renormalized. dvips/PK
|
||||
// bitmap fonts declare [1 0 0 -1 0 0] with glyphs spanning
|
||||
// hundreds of units, giving scale ≈ 159 against a nominal size of
|
||||
// 0.12pt — there the declared size carries no information. The
|
||||
// band is deliberately wide so that only that class qualifies,
|
||||
// while any matrix scale (including non-standard ones like 0.005
|
||||
// with a full-em bbox, scale = 5.0) is judged on the product
|
||||
// rather than on the matrix alone.
|
||||
const CONSISTENT_LO: f32 = 0.25;
|
||||
const CONSISTENT_HI: f32 = 4.0;
|
||||
if scale.is_finite() && scale > 0.0 && !(CONSISTENT_LO..=CONSISTENT_HI).contains(&scale) {
|
||||
scales.insert(String::from_utf8_lossy(font_name).to_string(), scale);
|
||||
}
|
||||
}
|
||||
@@ -1636,6 +1654,55 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// Build a one-font Type3 document and return its computed scale, if any.
|
||||
#[cfg(test)]
|
||||
fn type3_scale_for(matrix_y: f32, bbox_lo: i64, bbox_hi: i64) -> Option<f32> {
|
||||
use lopdf::{dictionary, Document, Object};
|
||||
let doc = Document::with_version("1.4");
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type3",
|
||||
"FontMatrix" => vec![
|
||||
Object::Real(matrix_y), Object::Integer(0), Object::Integer(0),
|
||||
Object::Real(matrix_y), Object::Integer(0), Object::Integer(0),
|
||||
],
|
||||
"FontBBox" => vec![
|
||||
Object::Integer(0), Object::Integer(bbox_lo),
|
||||
Object::Integer(600), Object::Integer(bbox_hi),
|
||||
],
|
||||
};
|
||||
let mut fonts = std::collections::BTreeMap::new();
|
||||
fonts.insert(b"T9".to_vec(), &font_dict);
|
||||
super::build_type3_scales(&doc, &fonts).get("T9").copied()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_skips_self_consistent_fonts() {
|
||||
// Conventional 1/1000 matrix with a descender..ascender bbox of 700
|
||||
// units: scale 0.7. The Tf operand is already the rendered size, so
|
||||
// renormalizing would report every size at 0.7x.
|
||||
assert_eq!(type3_scale_for(0.001, -200, 500), None);
|
||||
// Tall-accent bbox slightly over the em (1100 units, scale 1.1).
|
||||
assert_eq!(type3_scale_for(0.001, -100, 1000), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_applies_to_inconsistent_fonts_at_any_matrix_scale() {
|
||||
// Non-standard but valid matrix (0.005) with a full-em bbox:
|
||||
// scale 5.0, so the declared size is off by 5x and must be fixed.
|
||||
let s = type3_scale_for(0.005, 0, 1000).expect("0.005 matrix should rescale");
|
||||
assert!((s - 5.0).abs() < 0.01, "got {s}");
|
||||
// dvips/PK bitmap pattern: unit matrix, glyphs spanning ~159 units.
|
||||
let s = type3_scale_for(1.0, -156, 3).expect("PK pattern should rescale");
|
||||
assert!((s - 159.0).abs() < 0.5, "got {s}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type3_scale_ignores_degenerate_bbox() {
|
||||
// [0 0 0 0] is legal and carries no size information.
|
||||
assert_eq!(type3_scale_for(0.001, 0, 0), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn texcm_math_symbols_remap() {
|
||||
assert_eq!(
|
||||
|
||||
+769
-15
@@ -960,22 +960,713 @@ fn spans_multiple_columns(item: &TextItem, columns: &[ColumnRegion]) -> bool {
|
||||
overlap_count >= 2
|
||||
}
|
||||
|
||||
/// Check if a text item is likely a page number
|
||||
fn is_page_number(item: &TextItem) -> bool {
|
||||
const PAGE_NUMBER_Y_TOLERANCE: f32 = 3.0;
|
||||
const PAGE_NUMBER_CONTEXT_GAP_EM: f32 = 1.5;
|
||||
const PAGE_NUMBER_BOTTOM_Y: f32 = 100.0;
|
||||
const PAGE_NUMBER_TOP_Y: f32 = 720.0;
|
||||
const SPREAD_MIN_CONTENT_WIDTH_EM: f32 = 40.0;
|
||||
const SPREAD_EDGE_FRACTION: f32 = 0.25;
|
||||
const ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM: f32 = 26.0;
|
||||
|
||||
type ContextualCandidateOccurrence = (u32, f32, Vec<(usize, u32)>);
|
||||
|
||||
fn page_number_value(item: &TextItem) -> Option<u32> {
|
||||
if !matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let text = item.text.trim();
|
||||
|
||||
// Must be 1-4 digits only
|
||||
if text.is_empty() || text.len() > 4 {
|
||||
return false;
|
||||
}
|
||||
if !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return false;
|
||||
if text.is_empty() || text.len() > 4 || !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Must be at top or bottom of page.
|
||||
// US Letter = 792pt, A4 = 841pt. Page numbers are typically in the
|
||||
// top ~5% or bottom ~12% of the page.
|
||||
item.y > 720.0 || item.y < 100.0
|
||||
if item.y <= PAGE_NUMBER_TOP_Y && item.y >= PAGE_NUMBER_BOTTOM_Y {
|
||||
return None;
|
||||
}
|
||||
|
||||
text.parse().ok()
|
||||
}
|
||||
|
||||
/// Mark numeric slots that advance inside a repeated deep-margin line.
|
||||
///
|
||||
/// A folio can be emitted as part of a footer text run (for example,
|
||||
/// `42 Company report`) and therefore look contextual on a single page. Across
|
||||
/// the document, however, the surrounding text and Y position repeat while the
|
||||
/// numeric slot advances. Require that full signal before treating the slot as
|
||||
/// a folio so constant metadata and substantive rows near the page edge remain
|
||||
/// untouched.
|
||||
fn mark_repeated_folio_candidates(
|
||||
occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>>,
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
// Both thresholds are evidence floors: short documents still need four
|
||||
// occurrences, while long documents also need meaningful coverage. Using
|
||||
// `min` here would make one occurrence sufficient in a one-page document.
|
||||
let min_pages = 4usize.max((document_page_count * 30).div_ceil(100));
|
||||
|
||||
for occurrences in occurrences_by_signature.into_values() {
|
||||
if occurrences.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let distinct_pages: HashSet<u32> = occurrences.iter().map(|(page, _, _)| *page).collect();
|
||||
// Repeated table rows or duplicated drawing labels can share a
|
||||
// signature multiple times on one page. They are not running folios.
|
||||
if distinct_pages.len() != occurrences.len() || distinct_pages.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let min_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if max_y - min_y >= PAGE_NUMBER_Y_TOLERANCE {
|
||||
continue;
|
||||
}
|
||||
|
||||
let slot_count = occurrences[0].2.len();
|
||||
if slot_count == 0
|
||||
|| occurrences
|
||||
.iter()
|
||||
.any(|(_, _, candidates)| candidates.len() != slot_count)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
for slot in 0..slot_count {
|
||||
let mut values: Vec<(u32, u32, usize)> = occurrences
|
||||
.iter()
|
||||
.map(|(page, _, candidates)| {
|
||||
let (index, value) = candidates[slot];
|
||||
(*page, value, index)
|
||||
})
|
||||
.collect();
|
||||
values.sort_by_key(|(page, _, _)| *page);
|
||||
|
||||
let unique_values: HashSet<u32> = values.iter().map(|(_, value, _)| *value).collect();
|
||||
let mostly_unique = unique_values.len() * 5 >= values.len() * 4;
|
||||
let page_tracking_pairs = values
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let page_delta = pair[1].0 - pair[0].0;
|
||||
let value_delta = pair[1].1.saturating_sub(pair[0].1);
|
||||
value_delta == page_delta || value_delta == page_delta.saturating_mul(2)
|
||||
})
|
||||
.count();
|
||||
let mostly_tracks_page_order = page_tracking_pairs * 5 >= (values.len() - 1) * 4;
|
||||
// A running folio can be offset by front matter or advance twice per
|
||||
// PDF page in a two-page spread, but its magnitude should still be
|
||||
// plausible for the document. This keeps changing metadata such as
|
||||
// a sequence of years from becoming a deletion signal.
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
let plausible_magnitude = values
|
||||
.iter()
|
||||
.all(|(_, value, _)| *value <= max_plausible_folio);
|
||||
|
||||
if mostly_unique && mostly_tracks_page_order && plausible_magnitude {
|
||||
for (_, _, index) in values {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark the contextual half of a facing-page folio pair.
|
||||
///
|
||||
/// A landscape PDF can contain two printed pages per PDF page. One folio may be
|
||||
/// isolated while the other touches footer text; they remain a pair because
|
||||
/// they are consecutive, share a deep-margin baseline, and sit on opposite
|
||||
/// sides of the spread. The isolated half is strong evidence that the touching
|
||||
/// half is also a folio.
|
||||
fn mark_spread_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let mut candidates_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
if value.is_some() {
|
||||
candidates_by_page
|
||||
.entry(items[index].page)
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
}
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
for (page, page_candidates) in candidates_by_page {
|
||||
let Some(&(page_left, page_right)) = page_bounds.get(&page) else {
|
||||
continue;
|
||||
};
|
||||
let page_width = page_right - page_left;
|
||||
if page_width <= 0.0 {
|
||||
continue;
|
||||
}
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
let max_pair_font_size = page_width / SPREAD_MIN_CONTENT_WIDTH_EM;
|
||||
let edge_side = |index: usize| {
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Index strong folio evidence by value and spread edge. Sorted
|
||||
// baselines let each contextual candidate query only the two adjacent
|
||||
// values on the opposite edge in O(log n), rather than comparing every
|
||||
// candidate pair on numeric-heavy pages.
|
||||
let mut known_baselines: HashMap<(u32, bool), Vec<f32>> = HashMap::new();
|
||||
for &index in &page_candidates {
|
||||
if (contextual[index] && !explicit_folio[index])
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let value = candidate_values[index].unwrap();
|
||||
known_baselines
|
||||
.entry((value, side))
|
||||
.or_default()
|
||||
.push(items[index].y);
|
||||
}
|
||||
for baselines in known_baselines.values_mut() {
|
||||
baselines.sort_by(f32::total_cmp);
|
||||
}
|
||||
|
||||
for index in page_candidates {
|
||||
if !contextual[index]
|
||||
|| explicit_folio[index]
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(value) = candidate_values[index] else {
|
||||
continue;
|
||||
};
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let y = items[index].y;
|
||||
let paired = [value.checked_sub(1), value.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|other_value| known_baselines.get(&(other_value, !side)))
|
||||
.any(|baselines| {
|
||||
let first = baselines
|
||||
.partition_point(|baseline| *baseline <= y - PAGE_NUMBER_Y_TOLERANCE);
|
||||
baselines
|
||||
.get(first)
|
||||
.is_some_and(|baseline| *baseline < y + PAGE_NUMBER_Y_TOLERANCE)
|
||||
});
|
||||
if paired {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark a contextual folio that alternates with an isolated folio on the
|
||||
/// neighboring PDF page.
|
||||
///
|
||||
/// Facing pages commonly put folios on opposite outer edges. A running header
|
||||
/// can touch the right-hand folio while the next left-hand folio is isolated.
|
||||
/// A single isolated candidate is not enough to remove nearby contextual text.
|
||||
/// Require a second pre-existing anchor in the same advancing sequence, along
|
||||
/// with a genuinely wide content span, stable baselines/font sizes, and
|
||||
/// alternating outer edges.
|
||||
fn mark_adjacent_page_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Do not let newly inferred candidates recursively become evidence for
|
||||
// later candidates; every match must be anchored by evidence established
|
||||
// before this cross-page pass.
|
||||
let strong_folio_evidence = explicit_folio.to_vec();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
let edge_side = |index: usize| {
|
||||
let &(page_left, page_right) = page_bounds.get(&items[index].page)?;
|
||||
let page_width = page_right - page_left;
|
||||
if page_width < items[index].font_size * ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM {
|
||||
return None;
|
||||
}
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Anchor sequences by outer edge and the value/page offset. This enforces
|
||||
// forward page tracking and lets candidates query adjacent pages directly,
|
||||
// while a baseline-sorted index finds a second independent anchor without
|
||||
// a document-wide quadratic scan.
|
||||
let mut anchors_by_page: HashMap<(u32, bool, i64), Vec<usize>> = HashMap::new();
|
||||
let mut anchors_by_sequence: HashMap<(bool, i64), Vec<usize>> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if *value > max_plausible_folio || (contextual[index] && !strong_folio_evidence[index]) {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
anchors_by_page
|
||||
.entry((page, side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
anchors_by_sequence
|
||||
.entry((side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
for anchors in anchors_by_sequence.values_mut() {
|
||||
anchors.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
}
|
||||
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if !contextual[index] || explicit_folio[index] || *value > max_plausible_folio {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
let Some(sequence_anchors) = anchors_by_sequence.get(&(!side, offset)) else {
|
||||
continue;
|
||||
};
|
||||
let neighbor = [page.checked_sub(1), page.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|neighbor_page| anchors_by_page.get(&(neighbor_page, !side, offset)))
|
||||
.flatten()
|
||||
.copied()
|
||||
.find(|&neighbor_index| {
|
||||
(items[index].y - items[neighbor_index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
&& (items[index].font_size - items[neighbor_index].font_size).abs() < 1.0
|
||||
});
|
||||
let Some(neighbor_index) = neighbor else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let first = sequence_anchors.partition_point(|&anchor_index| {
|
||||
items[anchor_index].y <= items[index].y - PAGE_NUMBER_Y_TOLERANCE
|
||||
});
|
||||
let has_second_anchor = sequence_anchors[first..]
|
||||
.iter()
|
||||
.take_while(|&&anchor_index| {
|
||||
items[anchor_index].y < items[index].y + PAGE_NUMBER_Y_TOLERANCE
|
||||
})
|
||||
.any(|&anchor_index| {
|
||||
items[anchor_index].page != page
|
||||
&& items[anchor_index].page != items[neighbor_index].page
|
||||
&& (items[index].font_size - items[anchor_index].font_size).abs() < 1.0
|
||||
});
|
||||
if has_second_anchor {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Identify page-edge numeric items that belong to a nearby content run.
|
||||
///
|
||||
/// Numeric candidates on their own do not establish context for one another.
|
||||
/// A connected same-baseline run is contextual only when it also contains a
|
||||
/// non-candidate item, preserving lines such as `Chapter 1 2026` while still
|
||||
/// removing isolated numeric footer clusters.
|
||||
fn page_number_context_masks(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
document_page_count: usize,
|
||||
) -> (Vec<bool>, Vec<bool>) {
|
||||
let mut contextual = vec![false; items.len()];
|
||||
let mut explicit_folio = vec![false; items.len()];
|
||||
let mut occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>> =
|
||||
HashMap::new();
|
||||
let mut indices_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
for (index, item) in items.iter().enumerate() {
|
||||
if matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) && !item.text.trim().is_empty()
|
||||
{
|
||||
indices_by_page.entry(item.page).or_default().push(index);
|
||||
}
|
||||
}
|
||||
for mut page_indices in indices_by_page.into_values() {
|
||||
page_indices.sort_by(|&left, &right| {
|
||||
items[right]
|
||||
.y
|
||||
.total_cmp(&items[left].y)
|
||||
.then(items[left].x.total_cmp(&items[right].x))
|
||||
});
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in page_indices {
|
||||
if rows.last().is_some_and(|row| {
|
||||
(items[row[0]].y - items[index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
}) {
|
||||
rows.last_mut().unwrap().push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
let mut start = 0;
|
||||
while start < row.len() {
|
||||
let mut end = start + 1;
|
||||
let first = &items[row[start]];
|
||||
let mut group_right = first.x + effective_width(first);
|
||||
let mut group_font_size = first.font_size;
|
||||
|
||||
while end < row.len() {
|
||||
let item = &items[row[end]];
|
||||
let gap = item.x - group_right;
|
||||
if gap > group_font_size.max(item.font_size) * PAGE_NUMBER_CONTEXT_GAP_EM {
|
||||
break;
|
||||
}
|
||||
group_right = group_right.max(item.x + effective_width(item));
|
||||
group_font_size = group_font_size.max(item.font_size);
|
||||
end += 1;
|
||||
}
|
||||
|
||||
let group = &row[start..end];
|
||||
let has_lexical_context = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.chars()
|
||||
.any(|character| character.is_alphabetic())
|
||||
});
|
||||
// Numeric data near a page edge also needs protection, but a
|
||||
// lone long integer beside a short candidate is not enough to
|
||||
// establish context. Preserve explicit numeric structures
|
||||
// (list markers, ranges, comma-formatted values, dotted index
|
||||
// entries) and dense runs with at least one long integer.
|
||||
let numeric_like = |text: &str| {
|
||||
text.chars().any(|character| character.is_numeric())
|
||||
&& !text.chars().any(|character| character.is_alphabetic())
|
||||
};
|
||||
let is_structured_numeric_context = |index: usize| {
|
||||
if candidate_values[index].is_some() {
|
||||
return false;
|
||||
}
|
||||
let text = items[index].text.trim();
|
||||
numeric_like(text)
|
||||
&& text
|
||||
.chars()
|
||||
.any(|character| !character.is_numeric() && !character.is_whitespace())
|
||||
};
|
||||
let has_structured_numeric_context = group
|
||||
.iter()
|
||||
.any(|&index| is_structured_numeric_context(index));
|
||||
let numeric_item_count = group
|
||||
.iter()
|
||||
.filter(|&&index| numeric_like(items[index].text.trim()))
|
||||
.count();
|
||||
let has_long_integer = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_numeric())
|
||||
});
|
||||
let has_dense_numeric_context = numeric_item_count >= 3 && has_long_integer;
|
||||
let has_context = has_lexical_context
|
||||
|| has_structured_numeric_context
|
||||
|| has_dense_numeric_context;
|
||||
let has_candidate = row[start..end]
|
||||
.iter()
|
||||
.any(|&index| candidate_values[index].is_some());
|
||||
// Decorative centered folios have no lexical context, so
|
||||
// recognize the complete delimiter-number-delimiter triplet
|
||||
// before the contextual-content gate. This prevents `- 42 -`
|
||||
// from leaving a malformed `- -` line.
|
||||
if group.len() == 3
|
||||
&& items[group[0]].text.trim() == "-"
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim() == "-"
|
||||
{
|
||||
for &index in group {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
if has_context && has_candidate {
|
||||
let group = &row[start..end];
|
||||
let group_text = group
|
||||
.iter()
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let group_is_folio =
|
||||
crate::text_utils::is_explicit_page_number_expression(&group_text);
|
||||
let context_text = group
|
||||
.iter()
|
||||
.filter(|&&index| candidate_values[index].is_none())
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let candidates: Vec<(usize, u32)> = group
|
||||
.iter()
|
||||
.filter_map(|&index| candidate_values[index].map(|value| (index, value)))
|
||||
.collect();
|
||||
// Recurrence is only evidence for numeric slots at the
|
||||
// outer boundary of a contextual run. An embedded number
|
||||
// in repeated prose such as `Page 42 explains the result`
|
||||
// is substantive content, not a running folio.
|
||||
let recurrence_candidates: Vec<(usize, u32)> = candidates
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(index, _)| {
|
||||
group.first() == Some(index) || group.last() == Some(index)
|
||||
})
|
||||
.collect();
|
||||
let in_deep_margin = candidates.iter().all(|(index, _)| {
|
||||
items[*index].y < PAGE_NUMBER_BOTTOM_Y
|
||||
|| items[*index].y > PAGE_NUMBER_TOP_Y
|
||||
});
|
||||
if in_deep_margin
|
||||
&& !recurrence_candidates.is_empty()
|
||||
&& context_text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphanumeric())
|
||||
.count()
|
||||
>= 8
|
||||
{
|
||||
let signature = group
|
||||
.iter()
|
||||
.map(|&index| {
|
||||
if candidate_values[index].is_some() {
|
||||
"{number}".to_string()
|
||||
} else {
|
||||
items[index]
|
||||
.text
|
||||
.split_whitespace()
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ")
|
||||
.to_lowercase()
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
occurrences_by_signature
|
||||
.entry(signature)
|
||||
.or_default()
|
||||
.push((
|
||||
items[group[0]].page,
|
||||
items[group[0]].y,
|
||||
recurrence_candidates,
|
||||
));
|
||||
}
|
||||
for (position, &index) in group.iter().enumerate() {
|
||||
if let Some(value) = candidate_values[index] {
|
||||
let adjacent_context = [position.checked_sub(1), Some(position + 1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|position| group.get(position).copied())
|
||||
.any(|adjacent| {
|
||||
candidate_values[adjacent].is_none()
|
||||
&& items[adjacent].text.chars().any(|character| {
|
||||
!character.is_numeric() && !character.is_whitespace()
|
||||
})
|
||||
});
|
||||
let max_plausible_folio =
|
||||
(document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Large year/identifier-like values stay attached
|
||||
// to their lexical run even when a smaller numeric
|
||||
// candidate sits between them and the text.
|
||||
let implausible_folio_with_lexical_context =
|
||||
value > max_plausible_folio && has_lexical_context;
|
||||
contextual[index] = adjacent_context
|
||||
|| has_dense_numeric_context
|
||||
|| implausible_folio_with_lexical_context;
|
||||
let previous = position
|
||||
.checked_sub(1)
|
||||
.map(|position| items[group[position]].text.trim());
|
||||
let next = group
|
||||
.get(position + 1)
|
||||
.map(|&index| items[index].text.trim());
|
||||
let follows_page_label =
|
||||
previous.is_some_and(|text| text.eq_ignore_ascii_case("page"));
|
||||
let starts_of_expression =
|
||||
next.is_some_and(|text| text.eq_ignore_ascii_case("of"));
|
||||
let is_centered_folio = previous == Some("-") && next == Some("-");
|
||||
explicit_folio[index] |= group_is_folio
|
||||
&& (follows_page_label
|
||||
|| starts_of_expression
|
||||
|| is_centered_folio);
|
||||
}
|
||||
}
|
||||
// Remove the complete labeled expression rather than
|
||||
// leaving fragments such as `Page of 15`. A trailing
|
||||
// running-header suffix remains untouched.
|
||||
if group_is_folio
|
||||
&& group.len() >= 4
|
||||
&& items[group[0]].text.trim().eq_ignore_ascii_case("page")
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim().eq_ignore_ascii_case("of")
|
||||
&& items[group[3]]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_ascii_digit())
|
||||
{
|
||||
for &index in &group[..4] {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mark_repeated_folio_candidates(
|
||||
occurrences_by_signature,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
mark_spread_folio_pairs(items, candidate_values, &contextual, &mut explicit_folio);
|
||||
mark_adjacent_page_folio_pairs(
|
||||
items,
|
||||
candidate_values,
|
||||
&contextual,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
|
||||
(contextual, explicit_folio)
|
||||
}
|
||||
|
||||
/// Decide which digit-only page-edge items can be removed before layout.
|
||||
///
|
||||
/// PDF producers commonly emit one text-showing operation per word. A numeric
|
||||
/// item attached to neighboring content on the same baseline is therefore kept.
|
||||
/// Complete page-number expressions such as `Page 42` remain removable even
|
||||
/// though their numeric item has lexical context.
|
||||
fn page_number_removal_mask(items: &[TextItem], document_page_count: usize) -> Vec<bool> {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, value)| explicit_folio[index] || (value.is_some() && !contextual[index]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return whether selected-page extraction contains a page-edge number whose
|
||||
/// folio status depends on evidence from other pages. Isolated and explicitly
|
||||
/// labeled folios can be decided locally; only contextual candidates require
|
||||
/// a document-wide extraction pass.
|
||||
pub(super) fn needs_document_page_number_context(
|
||||
items: &[TextItem],
|
||||
document_page_count: usize,
|
||||
) -> bool {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.any(|(index, value)| value.is_some() && contextual[index] && !explicit_folio[index])
|
||||
}
|
||||
|
||||
/// Remove numeric folios using complete document context before downstream
|
||||
/// non-table layout partitions could separate the evidence needed to recognize
|
||||
/// them.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn filter_markdown_page_numbers(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> Vec<TextItem> {
|
||||
filter_markdown_page_numbers_with_removed_pages(items, document_page_count).0
|
||||
}
|
||||
|
||||
/// Filter Markdown folios while retaining the pages where items were removed.
|
||||
///
|
||||
/// The page set lets downstream table-continuation classification preserve its
|
||||
/// pre-filter semantics even though structural layout consumes the cleaned
|
||||
/// item collection.
|
||||
pub(crate) fn filter_markdown_page_numbers_with_removed_pages(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> (Vec<TextItem>, HashSet<u32>, Vec<bool>) {
|
||||
let remove = page_number_removal_mask(&items, document_page_count as usize);
|
||||
let mut removed_pages = HashSet::new();
|
||||
let items = items
|
||||
.into_iter()
|
||||
.zip(remove.iter().copied())
|
||||
.filter_map(|(item, remove)| {
|
||||
if remove {
|
||||
removed_pages.insert(item.page);
|
||||
None
|
||||
} else {
|
||||
Some(item)
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(items, removed_pages, remove)
|
||||
}
|
||||
|
||||
/// Group text items into lines, with multi-column support
|
||||
@@ -1205,6 +1896,27 @@ pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
)
|
||||
}
|
||||
|
||||
/// Group items after document-level page-number filtering has already run.
|
||||
///
|
||||
/// Partitioned Markdown layout uses this path so a contextual candidate that
|
||||
/// was preserved with its complete baseline context is not reconsidered after
|
||||
/// its neighboring text lands in another band or chart/prose zone.
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
@@ -1222,6 +1934,23 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
@@ -1234,12 +1963,25 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Markdown output omits standalone numeric headers/footers. Plain-text
|
||||
// callers opt out because dropping extracted text violates that API.
|
||||
// Markdown output omits standalone numeric headers/footers. Determine
|
||||
// standalone status from rough baseline context before layout analysis so
|
||||
// removed page numbers cannot affect column detection. Plain-text callers
|
||||
// opt out because dropping extracted text violates that API.
|
||||
let items = if filter_page_numbers {
|
||||
// Item-only grouping has no document metadata, so use the highest
|
||||
// observed 1-based page as its best available coverage denominator.
|
||||
// The Markdown document path passes the authoritative PDF page count
|
||||
// through `filter_markdown_page_numbers` before reaching this helper.
|
||||
let observed_page_count = items
|
||||
.iter()
|
||||
.map(|item| item.page as usize)
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
let remove = page_number_removal_mask(&items, observed_page_count);
|
||||
items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.zip(remove)
|
||||
.filter_map(|(item, remove)| (!remove).then_some(item))
|
||||
.collect()
|
||||
} else {
|
||||
items
|
||||
@@ -1254,6 +1996,15 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
|
||||
for page in pages {
|
||||
let page_items: Vec<TextItem> = items.iter().filter(|i| i.page == page).cloned().collect();
|
||||
// Page-edge numeric runs are weak evidence for column geometry. Keep
|
||||
// contextual values for line assembly, but prevent their preservation
|
||||
// from changing the page's inferred layout.
|
||||
let column_detection_items: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|item| page_number_value(item).is_none())
|
||||
.cloned()
|
||||
.collect();
|
||||
let column_detection_items = column_detection_items.as_slice();
|
||||
|
||||
// Use pre-computed threshold from fix_letterspaced_items if available
|
||||
// (computed before embedded-space removal, with full signal).
|
||||
@@ -1265,9 +2016,9 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
// their own positioned-region ordering and therefore stay on that path.
|
||||
if !chart_regions.contains_key(&page) {
|
||||
let preliminary_columns =
|
||||
detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
detect_columns(column_detection_items, page, table_pages.contains(&page));
|
||||
let detected_split =
|
||||
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
|
||||
(preliminary_columns.len() == 2).then(|| preliminary_columns[0].x_max);
|
||||
if let Some(band) = image_regions.get(&page).and_then(|regions| {
|
||||
super::reading_order::infer_image_anchored_flow(
|
||||
&page_items,
|
||||
@@ -1307,6 +2058,9 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
let col_input: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
if page_number_value(it).is_some() {
|
||||
return false;
|
||||
}
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Tight bounds: this only blinds the histogram to
|
||||
// chart-internal text; rows adjacent to the chart
|
||||
@@ -1319,7 +2073,7 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
.collect();
|
||||
detect_columns(&col_input, page, table_pages.contains(&page))
|
||||
}
|
||||
None => detect_columns(&page_items, page, table_pages.contains(&page)),
|
||||
None => detect_columns(column_detection_items, page, table_pages.contains(&page)),
|
||||
};
|
||||
|
||||
if columns.len() <= 1 {
|
||||
|
||||
+90
-2
@@ -153,16 +153,54 @@ pub(crate) fn extract_form_fields(
|
||||
},
|
||||
Err(_) => return items,
|
||||
};
|
||||
if fields.is_empty() {
|
||||
return items;
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(doc, field_ref, None, "", page_map, &mut items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
field_ref,
|
||||
None,
|
||||
"",
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
items
|
||||
}
|
||||
|
||||
/// Map widget annotation objects back to the page whose `/Annots` array owns
|
||||
/// them. Some valid widgets omit `/P`, so the page tree is the only reliable
|
||||
/// ownership signal available for page-filtered extraction.
|
||||
fn annotation_page_map(
|
||||
doc: &Document,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
) -> HashMap<ObjectId, u32> {
|
||||
let mut annotation_pages = HashMap::new();
|
||||
for (&page_id, &page_num) in page_map {
|
||||
let Some(annotations) = doc
|
||||
.get_dictionary(page_id)
|
||||
.ok()
|
||||
.and_then(|page| page.get(b"Annots").ok())
|
||||
.and_then(|annotations| resolve_array(doc, annotations))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
for annotation in annotations {
|
||||
if let Ok(annotation_id) = annotation.as_reference() {
|
||||
annotation_pages.insert(annotation_id, page_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
annotation_pages
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
@@ -170,6 +208,7 @@ pub(crate) fn walk_form_fields(
|
||||
parent_ft: Option<&[u8]>,
|
||||
parent_name: &str,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
) {
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
@@ -206,7 +245,15 @@ pub(crate) fn walk_form_fields(
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(doc, kid_ref, ft, &full_name, page_map, items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
kid_ref,
|
||||
ft,
|
||||
&full_name,
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -299,6 +346,7 @@ pub(crate) fn walk_form_fields(
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.and_then(|p| page_map.get(&p).copied())
|
||||
.or_else(|| annotation_pages.get(&field_id).copied())
|
||||
.unwrap_or(1);
|
||||
|
||||
let text = if full_name.is_empty() {
|
||||
@@ -324,3 +372,43 @@ pub(crate) fn walk_form_fields(
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::{dictionary, Object};
|
||||
|
||||
#[test]
|
||||
fn widget_without_page_reference_uses_owning_page_annotation() {
|
||||
let mut doc = Document::new();
|
||||
let widget_id = doc.add_object(dictionary! {
|
||||
"Type" => "Annot",
|
||||
"Subtype" => "Widget",
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("customer"),
|
||||
"V" => Object::string_literal("Alice"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
});
|
||||
let page_one_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
});
|
||||
let page_two_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Annots" => vec![Object::Reference(widget_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(widget_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::from([(page_one_id, 1), (page_two_id, 2)]);
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
}
|
||||
|
||||
+846
-21
@@ -28,9 +28,12 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
#[cfg(test)]
|
||||
use layout::filter_markdown_page_numbers;
|
||||
pub(crate) use layout::filter_markdown_page_numbers_with_removed_pages;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
@@ -82,17 +85,33 @@ pub fn extract_text_with_positions_pages<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<Vec<TextItem>, PdfError> {
|
||||
let (items, _rects, _lines) = extract_text_with_positions_and_rects(path, page_filter)?;
|
||||
let (items, _rects, _lines) =
|
||||
extract_text_with_positions_and_rects_with_password(path, page_filter, None)?;
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
/// Extract text with positions and rectangles from a file.
|
||||
pub(crate) fn extract_text_with_positions_and_rects<P: AsRef<Path>>(
|
||||
/// Extract text with positions from a file, limited to specific pages and
|
||||
/// decrypting with `password` when the PDF is encrypted.
|
||||
///
|
||||
/// `page_filter` is an optional set of 1-indexed page numbers to process.
|
||||
/// When `None`, all pages are processed.
|
||||
pub fn extract_text_with_positions_pages_with_password<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<Vec<TextItem>, PdfError> {
|
||||
let (items, _rects, _lines) =
|
||||
extract_text_with_positions_and_rects_with_password(path, page_filter, password)?;
|
||||
Ok(items)
|
||||
}
|
||||
|
||||
pub(crate) fn extract_text_with_positions_and_rects_with_password<P: AsRef<Path>>(
|
||||
path: P,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
password: Option<&str>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
let (doc, _) = crate::load_document_from_path_with_password(&path, password)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
@@ -141,17 +160,92 @@ pub(crate) fn extract_positioned_text_from_doc(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false)
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false, None)
|
||||
}
|
||||
|
||||
/// Extract with option to include invisible (Tr=3) text.
|
||||
/// Used for Mixed/template PDFs where the OCR text layer is invisible.
|
||||
pub(crate) fn extract_positioned_text_include_invisible(
|
||||
/// Extract selected pages and gather document-wide folio evidence only when a
|
||||
/// selected page contains an ambiguous contextual page-edge number. Errors on
|
||||
/// selected pages remain fatal; errors on context-only pages are skipped.
|
||||
pub(crate) fn extract_positioned_text_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, true)
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, false)
|
||||
}
|
||||
|
||||
/// Invisible-text variant of [`extract_positioned_text_with_folio_context`].
|
||||
pub(crate) fn extract_positioned_text_include_invisible_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, true)
|
||||
}
|
||||
|
||||
fn extract_positioned_text_with_folio_context_impl(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let Some(required_pages) = page_filter else {
|
||||
return extract_positioned_text_impl(doc, font_cmaps, None, include_invisible, None);
|
||||
};
|
||||
|
||||
let (
|
||||
(mut selected_items, mut selected_rects, mut selected_lines),
|
||||
mut page_thresholds,
|
||||
mut gid_encoded_pages,
|
||||
) = extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(required_pages),
|
||||
include_invisible,
|
||||
None,
|
||||
)?;
|
||||
if !layout::needs_document_page_number_context(&selected_items, doc.get_pages().len()) {
|
||||
return Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
));
|
||||
}
|
||||
|
||||
let context_pages: HashSet<u32> = doc
|
||||
.get_pages()
|
||||
.keys()
|
||||
.copied()
|
||||
.filter(|page| !required_pages.contains(page))
|
||||
.collect();
|
||||
let ((context_items, context_rects, context_lines), context_thresholds, context_gid_pages) =
|
||||
extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(&context_pages),
|
||||
include_invisible,
|
||||
Some(required_pages),
|
||||
)?;
|
||||
selected_items.extend(context_items);
|
||||
selected_rects.extend(context_rects);
|
||||
selected_lines.extend(context_lines);
|
||||
page_thresholds.extend(context_thresholds);
|
||||
gid_encoded_pages.extend(context_gid_pages);
|
||||
Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
))
|
||||
}
|
||||
|
||||
/// Extract all pages for document-wide analysis while allowing malformed
|
||||
/// unselected pages to be skipped. Any requested page still fails normally.
|
||||
pub(crate) fn extract_positioned_text_for_document_analysis(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
required_pages: &HashSet<u32>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, None, false, Some(required_pages))
|
||||
}
|
||||
|
||||
fn extract_positioned_text_impl(
|
||||
@@ -159,6 +253,7 @@ fn extract_positioned_text_impl(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
required_pages: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let pages = doc.get_pages();
|
||||
let mut all_items = Vec::new();
|
||||
@@ -180,15 +275,25 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) =
|
||||
extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
let page_result = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
);
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) = match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error) if required_pages.is_some_and(|required| !required.contains(page_num)) => {
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
@@ -318,7 +423,9 @@ fn extract_positioned_text_impl(
|
||||
}
|
||||
|
||||
// Extract AcroForm field values
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num);
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num)
|
||||
.into_iter()
|
||||
.filter(|item| page_filter.is_none_or(|filter| filter.contains(&item.page)));
|
||||
all_items.extend(form_items);
|
||||
|
||||
Ok((
|
||||
@@ -1532,6 +1639,724 @@ mod tests {
|
||||
assert_eq!(lines[0].text(), "42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_numeric_run_near_page_edge_is_not_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Total 730 seats");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_page_footer_separated_from_label_is_removed() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut footer_label = make_merge_item("DOCUMENT FOOTER", 60.0, 100.0);
|
||||
footer_label.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, footer_label]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "DOCUMENT FOOTER");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decorative_marker_does_not_contextualize_numeric_page_footer() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut page_number = make_merge_item("42", 37.0, 10.0);
|
||||
page_number.y = 30.0;
|
||||
let mut footer_label = make_merge_item("Company report footer", 68.0, 120.0);
|
||||
footer_label.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, page_number, footer_label]);
|
||||
|
||||
assert!(lines.iter().all(|line| !line.text().contains("42")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_is_removed_in_a_short_document() {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item("42", 57.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![label, page_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_with_running_header_suffix_is_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("of", 73.0, 12.0),
|
||||
make_merge_item("100", 89.0, 18.0),
|
||||
make_merge_item("Report header", 111.0, 78.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report header");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_of_total_expression_is_removed_without_leaving_fragments() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 482.0, 27.0),
|
||||
make_merge_item("1", 513.0, 6.0),
|
||||
make_merge_item("of", 523.0, 10.0),
|
||||
make_merge_item("15", 537.0, 12.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 46.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn document_folio_filter_survives_per_page_layout_splitting() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
items.extend([label, page_number]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 3);
|
||||
assert!(filtered
|
||||
.iter()
|
||||
.all(|item| !matches!(item.text.as_str(), "42" | "43" | "44")));
|
||||
let mut lines = Vec::new();
|
||||
for page in 1..=3 {
|
||||
let page_items = filtered
|
||||
.iter()
|
||||
.filter(|item| item.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
lines.extend(
|
||||
group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
page_items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert!(lines.iter().all(|line| line.text() == "Page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_page_edge_runs_do_not_contextualize_folios() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut long_number = make_merge_item("12345", 43.0, 30.0);
|
||||
long_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, long_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12345");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structured_and_dense_numeric_page_edge_runs_are_preserved() {
|
||||
let mut list_marker = make_merge_item("11)", 25.0, 18.0);
|
||||
list_marker.y = 50.0;
|
||||
let mut chapter = make_merge_item("13", 47.0, 12.0);
|
||||
chapter.y = 50.0;
|
||||
|
||||
let mut isbn_prefix = make_merge_item("9", 25.0, 6.0);
|
||||
isbn_prefix.page = 2;
|
||||
isbn_prefix.y = 50.0;
|
||||
let mut isbn_mid = make_merge_item("780113", 35.0, 36.0);
|
||||
isbn_mid.page = 2;
|
||||
isbn_mid.y = 50.0;
|
||||
let mut isbn_end = make_merge_item("227426", 75.0, 36.0);
|
||||
isbn_end.page = 2;
|
||||
isbn_end.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![list_marker, chapter, isbn_prefix, isbn_mid, isbn_end]);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "11) 13");
|
||||
assert_eq!(lines[1].text(), "9 780113 227426");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn incrementing_numeric_body_column_is_not_treated_as_a_folio() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "13"), (2, "14"), (3, "15")] {
|
||||
let mut row_number = make_merge_item(value, 72.0, 12.0);
|
||||
row_number.page = page;
|
||||
row_number.y = 730.0;
|
||||
let mut name = make_merge_item("Person", 90.0, 42.0);
|
||||
name.page = page;
|
||||
name.y = 730.0;
|
||||
items.extend([row_number, name]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert_eq!(lines[0].text(), "13 Person");
|
||||
assert_eq!(lines[1].text(), "14 Person");
|
||||
assert_eq!(lines[2].text(), "15 Person");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn advancing_number_in_repeated_deep_margin_footer_is_removed() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_substantive_page_number_prose_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44"), (4, "45")] {
|
||||
let mut page_label = make_merge_item("Page", 25.0, 28.0);
|
||||
page_label.page = page;
|
||||
page_label.y = 30.0;
|
||||
let mut number = make_merge_item(value, 57.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut explanation = make_merge_item("explains the result", 73.0, 108.0);
|
||||
explanation.page = page;
|
||||
explanation.y = 30.0;
|
||||
items.extend([page_label, number, explanation]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
for (line, value) in lines.iter().zip(["42", "43", "44", "45"]) {
|
||||
assert_eq!(line.text(), format!("Page {value} explains the result"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_candidates_do_not_bridge_lexical_context() {
|
||||
let mut report = make_merge_item("Report", 25.0, 40.0);
|
||||
report.y = 30.0;
|
||||
let mut year = make_merge_item("2026", 69.0, 24.0);
|
||||
year.y = 30.0;
|
||||
let mut folio = make_merge_item("42", 97.0, 12.0);
|
||||
folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![report, year, folio]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_folio_delimiters_are_removed_with_the_number() {
|
||||
let mut left = make_merge_item("-", 270.0, 6.0);
|
||||
left.y = 30.0;
|
||||
let mut number = make_merge_item("42", 280.0, 12.0);
|
||||
number.y = 30.0;
|
||||
let mut right = make_merge_item("-", 296.0, 6.0);
|
||||
right.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![left, number, right]);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_delimiters_inside_substantive_text_are_preserved() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Result", 240.0, 36.0),
|
||||
make_merge_item("-", 280.0, 6.0),
|
||||
make_merge_item("42", 290.0, 12.0),
|
||||
make_merge_item("-", 306.0, 6.0),
|
||||
make_merge_item("approved", 316.0, 48.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 30.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Result-42-approved");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn changing_year_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, year) in [(1, "2020"), (2, "2021"), (3, "2022"), (4, "2023")] {
|
||||
let mut year = make_merge_item(year, 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert_eq!(lines[0].text(), "2020 Annual report");
|
||||
assert_eq!(lines[3].text(), "2023 Annual report");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_repeated_margin_numbers_do_not_meet_the_folio_evidence_floor() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
if page <= 2 {
|
||||
let value = if page == 1 { "2" } else { "4" };
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
} else {
|
||||
let mut body = make_merge_item("Body text", 72.0, 54.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
}
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "2 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "4 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_document_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (10, "10"), (19, "19"), (28, "28")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "1 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "28 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_blank_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 20);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().any(|item| item.text == "4"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefiltered_contextual_number_survives_layout_partitioning() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 1);
|
||||
let partitioned_number: Vec<TextItem> = filtered
|
||||
.into_iter()
|
||||
.filter(|item| item.text == "730")
|
||||
.collect();
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
partitioned_number,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "730");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_partition_does_not_define_columns() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..20 {
|
||||
let y = 90.0 - row as f32 * 4.0;
|
||||
let mut left = make_merge_item(&(row + 1).to_string(), 50.0, 20.0);
|
||||
left.y = y;
|
||||
let mut right = make_merge_item(&(row + 101).to_string(), 350.0, 20.0);
|
||||
right.y = y;
|
||||
items.extend([left, right]);
|
||||
}
|
||||
assert_eq!(detect_columns(&items, 1, false).len(), 2);
|
||||
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 20);
|
||||
assert!(lines.iter().all(|line| line.items.len() == 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separated_content_is_not_treated_as_a_spread_folio_pair() {
|
||||
let mut value = make_merge_item("12", 100.0, 12.0);
|
||||
value.y = 30.0;
|
||||
let mut label = make_merge_item("Total", 116.0, 30.0);
|
||||
label.y = 30.0;
|
||||
let mut unrelated_number = make_merge_item("13", 300.0, 12.0);
|
||||
unrelated_number.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![value, label, unrelated_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12 Total");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_folio_uses_the_full_page_edge_band() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 80.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 80.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folio_on_facing_page_spread_is_removed() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut left_folio = make_merge_item("326", 35.0, 17.0);
|
||||
left_folio.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 61.0, 120.0);
|
||||
footer.y = 30.0;
|
||||
let mut right_folio = make_merge_item("327", 1148.0, 17.0);
|
||||
right_folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, left_folio, footer, right_folio]);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| !line.text().contains("326") && !line.text().contains("327")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folios_alternating_across_pages_are_removed() {
|
||||
let headers = [
|
||||
"Letter to shareholders",
|
||||
"Corporate governance report",
|
||||
"Business environment overview",
|
||||
"Consolidated financial statements",
|
||||
];
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=8 {
|
||||
let mut body = make_merge_item("Body text", 50.0, 500.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
|
||||
let mut folio = make_merge_item(&(page + 22).to_string(), 0.0, 14.0);
|
||||
folio.page = page;
|
||||
folio.y = 780.0;
|
||||
if page % 2 == 0 {
|
||||
folio.x = 50.0;
|
||||
items.push(folio);
|
||||
} else {
|
||||
let mut header = make_merge_item(headers[(page / 2) as usize], 350.0, 180.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
folio.x = 536.0;
|
||||
items.extend([header, folio]);
|
||||
}
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 8);
|
||||
|
||||
assert!(filtered.iter().all(|item| {
|
||||
!matches!(
|
||||
item.text.as_str(),
|
||||
"23" | "24" | "25" | "26" | "27" | "28" | "29" | "30"
|
||||
)
|
||||
}));
|
||||
assert!(headers
|
||||
.iter()
|
||||
.all(|header| filtered.iter().any(|item| item.text == *header)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_isolated_neighbor_does_not_remove_contextual_number() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 450.0, 70.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 526.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated = make_merge_item("2", 50.0, 7.0);
|
||||
isolated.page = 2;
|
||||
isolated.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, label, contextual, body_two, isolated], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().all(|item| item.text != "2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_content_span_does_not_establish_adjacent_page_edges() {
|
||||
let mut body_one = make_merge_item("Body text", 100.0, 120.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 170.0, 60.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 235.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated_two = make_merge_item("2", 100.0, 7.0);
|
||||
isolated_two.page = 2;
|
||||
isolated_two.y = 780.0;
|
||||
|
||||
let mut body_four = body_one.clone();
|
||||
body_four.page = 4;
|
||||
let mut isolated_four = make_merge_item("4", 100.0, 7.0);
|
||||
isolated_four.page = 4;
|
||||
isolated_four.y = 780.0;
|
||||
|
||||
let filtered = filter_markdown_page_numbers(
|
||||
vec![
|
||||
body_one,
|
||||
label,
|
||||
contextual,
|
||||
body_two,
|
||||
isolated_two,
|
||||
body_four,
|
||||
isolated_four,
|
||||
],
|
||||
4,
|
||||
);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_edge_number_on_an_adjacent_page_is_not_folio_evidence() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut isolated = make_merge_item("42", 50.0, 14.0);
|
||||
isolated.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut contextual = make_merge_item("43", 50.0, 14.0);
|
||||
contextual.page = 2;
|
||||
contextual.y = 780.0;
|
||||
let mut label = make_merge_item("cases reviewed", 70.0, 90.0);
|
||||
label.page = 2;
|
||||
label.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, isolated, body_two, contextual, label], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "43"));
|
||||
assert!(filtered.iter().any(|item| item.text == "cases reviewed"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_number_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
let mut year = make_merge_item("2026", 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines.iter().all(|line| line.text() == "2026 Annual report"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_prefix_does_not_remove_substantive_text_during_layout() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("explains", 73.0, 44.0),
|
||||
make_merge_item("the result", 121.0, 55.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42 explains the result");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_page_number_prefix_with_substantive_text_is_preserved_during_layout() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value, chapter) in [(1, "42", "Chapter 1"), (2, "43", "Chapter 2")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
let mut suffix = make_merge_item(chapter, 73.0, 58.0);
|
||||
suffix.page = page;
|
||||
suffix.y = 50.0;
|
||||
items.extend([label, page_number, suffix]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "Page 42 Chapter 1");
|
||||
assert_eq!(lines[1].text(), "Page 43 Chapter 2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_phrase_in_the_page_body_is_preserved() {
|
||||
let items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
];
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_numeric_context_near_page_edge_is_preserved() {
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
|
||||
let chapter_lines = group_into_lines(vec![chapter, chapter_number]);
|
||||
assert_eq!(chapter_lines.len(), 1);
|
||||
assert_eq!(chapter_lines[0].text(), "Chapter 1");
|
||||
|
||||
let mut year = make_merge_item("2026", 100.0, 24.0);
|
||||
year.y = 760.0;
|
||||
let mut report = make_merge_item("Report", 130.0, 36.0);
|
||||
report.y = 760.0;
|
||||
|
||||
let report_lines = group_into_lines(vec![year, report]);
|
||||
assert_eq!(report_lines.len(), 1);
|
||||
assert_eq!(report_lines[0].text(), "2026 Report");
|
||||
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
let mut edition_year = make_merge_item("2026", 163.0, 24.0);
|
||||
edition_year.y = 760.0;
|
||||
|
||||
let chained_lines = group_into_lines(vec![chapter, chapter_number, edition_year]);
|
||||
assert_eq!(chained_lines.len(), 1);
|
||||
assert_eq!(chained_lines[0].text(), "Chapter 1 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bold_italic_detection() {
|
||||
// Test bold detection
|
||||
|
||||
+385
-17
@@ -44,6 +44,16 @@ const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
/// Strikeout decorations are text-sized. Diagram connectors, signature
|
||||
/// lines, and chart rules often cross glyphs too, but extend well beyond the
|
||||
/// text they happen to intersect.
|
||||
const STRIKE_OWNER_PAD_EM: f32 = 0.75;
|
||||
const STRIKE_OWNER_MIN_PAD: f32 = 4.0;
|
||||
const STRIKE_ROW_Y_TOLERANCE_EM: f32 = 0.15;
|
||||
const STRIKE_ROW_Y_TOLERANCE_MIN: f32 = 5.0;
|
||||
const GRAPHIC_CONNECTION_EPS: f32 = 2.0;
|
||||
const GRAPHIC_CONNECTOR_MAX_THICKNESS: f32 = 4.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
@@ -387,6 +397,206 @@ fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
fn is_bare_list_marker(text: &str) -> bool {
|
||||
matches!(
|
||||
text.trim(),
|
||||
"•" | "◦" | "▪" | "▫" | "‣" | "⁃" | "●" | "○" | "■" | "□" | "-" | "*"
|
||||
)
|
||||
}
|
||||
|
||||
fn same_strike_row(left: &TextItem, right: &TextItem) -> bool {
|
||||
let font_size = left.font_size.max(right.font_size);
|
||||
let tolerance = (font_size * STRIKE_ROW_Y_TOLERANCE_EM).max(STRIKE_ROW_Y_TOLERANCE_MIN);
|
||||
(left.y - right.y).abs() <= tolerance
|
||||
}
|
||||
|
||||
fn is_inline_script(rule: &Rule, candidate: &TextItem, parent: &TextItem) -> bool {
|
||||
if !is_underline_candidate(candidate)
|
||||
|| is_bare_list_marker(&candidate.text)
|
||||
|| candidate.font_size <= 0.0
|
||||
|| candidate.font_size >= parent.font_size * 0.75
|
||||
|| candidate.text.len() > 4
|
||||
|| !candidate.text.chars().all(|c| c.is_ascii_digit())
|
||||
|| (candidate.y - parent.y).abs() > 5.0
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_ends_with_letter = parent.text.chars().last().is_some_and(char::is_alphabetic);
|
||||
if !parent_ends_with_letter {
|
||||
return false;
|
||||
}
|
||||
|
||||
let parent_right = parent.x + parent.width;
|
||||
let gap = candidate.x - parent_right;
|
||||
if gap >= parent.font_size * 0.2 || gap <= -parent.font_size * 0.3 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = rule.x2.min(candidate.x + candidate.width) - rule.x1.max(candidate.x);
|
||||
overlap >= candidate.width * MIN_X_OVERLAP
|
||||
}
|
||||
|
||||
/// Return the items owned by a snug mid-glyph rule.
|
||||
///
|
||||
/// Real strikeout decorations track the width of the deleted text, including
|
||||
/// runs split by font/style changes and adjacent numeric super/subscripts.
|
||||
/// Non-text graphics can cross the same vertical window, but arrow shafts,
|
||||
/// signature lines, fraction bars, and chart rules extend materially beyond
|
||||
/// the intersected glyphs. Requiring the rule to stay within a small em-sized
|
||||
/// pad of a contiguous matched row separates those cases without relying on
|
||||
/// document-specific fonts or coordinates.
|
||||
///
|
||||
/// Ownership is computed once per rule. This keeps the strikeout pass at the
|
||||
/// same rule-by-item scale as underline detection instead of rescanning the
|
||||
/// whole page for every matching item.
|
||||
fn snug_strike_owner_indices(rule: &Rule, items: &[TextItem]) -> Vec<usize> {
|
||||
let mut struck_indices: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
(is_underline_candidate(candidate)
|
||||
&& !is_bare_list_marker(&candidate.text)
|
||||
&& rule_strikes_item(rule, candidate))
|
||||
.then_some(index)
|
||||
})
|
||||
.collect();
|
||||
if struck_indices.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
struck_indices.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in struck_indices {
|
||||
if let Some(row) = rows
|
||||
.last_mut()
|
||||
.filter(|row| same_strike_row(&items[row[0]], &items[index]))
|
||||
{
|
||||
row.push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
let mut owned_indices = Vec::new();
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
|
||||
// Underline detection runs before the extractor's script-merging
|
||||
// pass. Include the same tightly adjacent numeric script shape here
|
||||
// when the rule spans it, so the owner width and semantic mark both
|
||||
// survive that later merge.
|
||||
let scripts: Vec<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(index, candidate)| {
|
||||
let parent_pos =
|
||||
row.partition_point(|&row_index| items[row_index].x <= candidate.x);
|
||||
let parent_index = parent_pos.checked_sub(1).map(|pos| row[pos])?;
|
||||
is_inline_script(rule, candidate, &items[parent_index]).then_some(index)
|
||||
})
|
||||
.collect();
|
||||
row.extend(scripts);
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
row.dedup();
|
||||
|
||||
let x1 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x2 = row
|
||||
.iter()
|
||||
.map(|&index| items[index].x + items[index].width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let max_font_size = row
|
||||
.iter()
|
||||
.map(|&index| items[index].font_size)
|
||||
.fold(0.0, f32::max);
|
||||
let pad = (max_font_size * STRIKE_OWNER_PAD_EM).max(STRIKE_OWNER_MIN_PAD);
|
||||
|
||||
if rule.x1 < x1 - pad || rule.x2 > x2 + pad {
|
||||
continue;
|
||||
}
|
||||
|
||||
let contiguous = row.windows(2).all(|pair| {
|
||||
let gap = items[pair[1]].x - (items[pair[0]].x + items[pair[0]].width);
|
||||
gap <= (max_font_size * 2.0).max(12.0)
|
||||
});
|
||||
if contiguous {
|
||||
owned_indices.extend(row);
|
||||
}
|
||||
}
|
||||
|
||||
owned_indices.sort_unstable();
|
||||
owned_indices.dedup();
|
||||
owned_indices
|
||||
}
|
||||
|
||||
/// Diagram and table rules participate in larger path geometry. A vertical
|
||||
/// or diagonal segment meeting the candidate rule is strong evidence that
|
||||
/// the horizontal segment is a connector, border, arrow, or symbol rather
|
||||
/// than an isolated text decoration.
|
||||
fn has_connected_nonhorizontal_segment(rule: &Rule, lines: &[UnderlineLine], page: u32) -> bool {
|
||||
lines.iter().any(|line| {
|
||||
if line.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let dx = line.x2 - line.x1;
|
||||
let dy = line.y2 - line.y1;
|
||||
if dy.abs() <= MAX_RULE_THICKNESS {
|
||||
return false;
|
||||
}
|
||||
|
||||
let y_min = line.y1.min(line.y2) - GRAPHIC_CONNECTION_EPS;
|
||||
let y_max = line.y1.max(line.y2) + GRAPHIC_CONNECTION_EPS;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let t = (rule.y - line.y1) / dy;
|
||||
if !(-0.05..=1.05).contains(&t) {
|
||||
return false;
|
||||
}
|
||||
let intersection_x = line.x1 + t * dx;
|
||||
intersection_x >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& intersection_x <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Filled diagrams often build connectors from intersecting thin rectangles
|
||||
/// instead of stroked path segments. Treat only narrow, vertically elongated
|
||||
/// rectangles as connector geometry; broad fills can legitimately sit behind
|
||||
/// struck text and must not veto its decoration.
|
||||
fn has_connected_nonhorizontal_rect(rule: &Rule, rects: &[PdfRect], page: u32) -> bool {
|
||||
rects.iter().any(|rect| {
|
||||
if rect.page != page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let (x1, x2) = if rect.width >= 0.0 {
|
||||
(rect.x, rect.x + rect.width)
|
||||
} else {
|
||||
(rect.x + rect.width, rect.x)
|
||||
};
|
||||
let (y1, y2) = if rect.height >= 0.0 {
|
||||
(rect.y, rect.y + rect.height)
|
||||
} else {
|
||||
(rect.y + rect.height, rect.y)
|
||||
};
|
||||
let width = x2 - x1;
|
||||
let height = y2 - y1;
|
||||
|
||||
width > 0.0
|
||||
&& width <= GRAPHIC_CONNECTOR_MAX_THICKNESS
|
||||
&& height > width * 2.0
|
||||
&& rule.y >= y1 - GRAPHIC_CONNECTION_EPS
|
||||
&& rule.y <= y2 + GRAPHIC_CONNECTION_EPS
|
||||
&& x2 >= rule.x1 - GRAPHIC_CONNECTION_EPS
|
||||
&& x1 <= rule.x2 + GRAPHIC_CONNECTION_EPS
|
||||
})
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
@@ -440,27 +650,43 @@ pub(crate) fn mark_underlined_items(
|
||||
.map(|(i, _)| i)
|
||||
.collect();
|
||||
|
||||
for item in items.iter_mut() {
|
||||
let mut strikeout_items = vec![false; items.len()];
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx)
|
||||
|| has_connected_nonhorizontal_segment(rule, lines, page)
|
||||
|| has_connected_nonhorizontal_rect(rule, rects, page)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
for item_idx in snug_strike_owner_indices(rule, items) {
|
||||
strikeout_items[item_idx] = true;
|
||||
}
|
||||
}
|
||||
|
||||
let underlined_items: HashSet<usize> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
is_underline_candidate(item)
|
||||
&& rules.iter().enumerate().any(|(rule_idx, rule)| {
|
||||
!tabular_rules.contains(&rule_idx)
|
||||
&& !fraction_rules.contains(&rule_idx)
|
||||
&& rule_matches_item(rule, item)
|
||||
})
|
||||
})
|
||||
.map(|(item_idx, _)| item_idx)
|
||||
.collect();
|
||||
|
||||
for (item_idx, item) in items.iter_mut().enumerate() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
// The fraction guard only gates UNDERLINE marking — a rule that
|
||||
// reads as a fraction bar from below can still legitimately
|
||||
// strike through a line above it.
|
||||
if !fraction_rules.contains(&rule_idx) && rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
if strikeout_items[item_idx] {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if underlined_items.contains(&item_idx) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -580,6 +806,148 @@ mod tests {
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_connector_crossing_text_is_not_a_strikeout() {
|
||||
let mut items = vec![item("diagram label", 160.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 280.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_rule_ending_inside_short_label_is_not_a_strikeout() {
|
||||
let mut items = vec![item("T 18", 300.0, 500.0, 20.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 315.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_diagram_segment_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let lines = vec![
|
||||
hline(99.0, 113.0, 503.0),
|
||||
UnderlineLine {
|
||||
x1: 106.0,
|
||||
y1: 496.0,
|
||||
x2: 109.0,
|
||||
y2: 510.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn connected_filled_rect_is_not_a_strikeout() {
|
||||
let mut items = vec![item("V8", 100.0, 500.0, 12.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 14.0),
|
||||
PdfRect {
|
||||
x: 106.0,
|
||||
y: 496.0,
|
||||
width: 2.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn broad_fill_behind_text_does_not_block_strikeout() {
|
||||
let mut items = vec![item("deleted", 100.0, 500.0, 40.0, 10.0)];
|
||||
let rects = vec![
|
||||
thin_rect(99.0, 502.6, 42.0),
|
||||
PdfRect {
|
||||
x: 90.0,
|
||||
y: 490.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
},
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
|
||||
assert!(items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_bullet_is_not_a_strikeout() {
|
||||
for marker in ["•", "-", "*"] {
|
||||
let mut items = vec![item(marker, 100.0, 500.0, 6.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 107.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(!items[0].is_strikeout, "marker {marker:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_marks_adjacent_split_runs_as_strikeout() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 500.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_groups_split_runs_with_baseline_drift() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("text", 142.0, 498.0, 25.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 168.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_superscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 503.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snug_rule_keeps_inline_subscript_in_strike_owner() {
|
||||
let mut items = vec![
|
||||
item("deleted", 100.0, 500.0, 40.0, 10.0),
|
||||
item("2", 140.5, 497.0, 4.0, 6.0),
|
||||
];
|
||||
let lines = vec![hline(99.0, 145.0, 503.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
|
||||
+224
-27
@@ -50,11 +50,11 @@ pub use detector::{
|
||||
};
|
||||
pub use extractor::{
|
||||
extract_text, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
extract_text_with_positions_pages,
|
||||
extract_text_with_positions_pages, extract_text_with_positions_pages_with_password,
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
MarkdownProfile,
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -462,16 +462,39 @@ pub fn extract_pages_markdown_mem(
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats.
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
// unselected page cannot make a valid requested page fail, but errors on a
|
||||
// requested page retain the normal extraction semantics.
|
||||
let required_pages: Option<HashSet<u32>> = pages.map(|pages| {
|
||||
pages
|
||||
.iter()
|
||||
.filter_map(|page| page.checked_add(1))
|
||||
.collect()
|
||||
});
|
||||
let ((all_items, all_rects, all_lines), page_thresholds, gid_pages) =
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||
if let Some(required_pages) = required_pages.as_ref() {
|
||||
extractor::extract_positioned_text_for_document_analysis(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
required_pages,
|
||||
)?
|
||||
} else {
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?
|
||||
};
|
||||
let text_quality = analyze_text_quality(&all_items);
|
||||
|
||||
// Compute layout complexity from full document (near-zero cost).
|
||||
let complexity = compute_layout_complexity(&all_items, &all_rects, &all_lines);
|
||||
// Resolve page numbers with full-document context before partitioning.
|
||||
// Per-page Markdown receives the original items plus these decisions so
|
||||
// table detection can retain legitimate numeric cells.
|
||||
let (filtered_items, removed_page_number_pages, page_number_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
|
||||
// Tables need the original numeric cells; columns use folio-cleaned
|
||||
// evidence so removed page numbers cannot create false layout metadata.
|
||||
let complexity = compute_layout_complexity(&all_items, &filtered_items, &all_rects, &all_lines);
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -502,12 +525,13 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
let page_1idx = page_0idx + 1;
|
||||
|
||||
// Filter items/rects for this page only
|
||||
let page_items: Vec<TextItem> = all_items
|
||||
// Partition items, removal decisions, and rects for this page only.
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page_1idx)
|
||||
.cloned()
|
||||
.collect();
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
let page_rects: Vec<PdfRect> = all_rects
|
||||
.iter()
|
||||
@@ -534,9 +558,14 @@ pub fn extract_pages_markdown_mem(
|
||||
options,
|
||||
&page_rects,
|
||||
&[],
|
||||
&page_thresholds,
|
||||
None,
|
||||
&[],
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_page_number_pages),
|
||||
prefiltered_page_number_mask: Some(&page_number_removal_mask),
|
||||
},
|
||||
)
|
||||
};
|
||||
|
||||
@@ -1061,7 +1090,12 @@ pub fn extract_tables_in_regions_mem(
|
||||
candidates.push(candidate);
|
||||
}
|
||||
}
|
||||
let detected = tables::detect_tables(&matched, base_font_size, false);
|
||||
let detected = tables::detect_tables_with_page_width(
|
||||
&matched,
|
||||
base_font_size,
|
||||
false,
|
||||
items.map_or(1.0, |items| tables::content_width(items)),
|
||||
);
|
||||
if let Some(candidate) = detected
|
||||
.iter()
|
||||
.find_map(|t| evaluate(TableCandidateSource::Heuristic, t))
|
||||
@@ -3605,7 +3639,10 @@ fn process_document(
|
||||
// Step 2 — Extraction (reuses the already-loaded document)
|
||||
let extracted = {
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extractor::extract_positioned_text_from_doc(
|
||||
// Most page-filtered requests extract only the selected pages. Gather
|
||||
// other pages only when a selected contextual folio needs cross-page
|
||||
// evidence; failures on those context-only pages are non-fatal.
|
||||
let result = extractor::extract_positioned_text_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3616,9 +3653,19 @@ fn process_document(
|
||||
// This unlocks OCR text layers behind scanned images.
|
||||
if pdf_type == PdfType::Mixed {
|
||||
if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) {
|
||||
let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect();
|
||||
let sample: String = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&item.page))
|
||||
})
|
||||
.take(200)
|
||||
.map(|item| item.text.as_str())
|
||||
.collect();
|
||||
if is_garbage_text(&sample) || sample.trim().is_empty() {
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3628,7 +3675,7 @@ fn process_document(
|
||||
}
|
||||
} else {
|
||||
// Normal extraction failed — try invisible as fallback
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3692,6 +3739,13 @@ fn process_document(
|
||||
let mut garbage_pages: std::collections::HashSet<u32> =
|
||||
std::collections::HashSet::new();
|
||||
for &pg in &ocr_set {
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|filter| !filter.contains(&pg))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let page_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.page == pg)
|
||||
@@ -3731,9 +3785,38 @@ fn process_document(
|
||||
}
|
||||
};
|
||||
|
||||
let selected_page = |page: u32| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&page))
|
||||
};
|
||||
let rects: Vec<_> = rects
|
||||
.into_iter()
|
||||
.filter(|rect| selected_page(rect.page))
|
||||
.collect();
|
||||
let lines: Vec<_> = lines
|
||||
.into_iter()
|
||||
.filter(|line| selected_page(line.page))
|
||||
.collect();
|
||||
let gid_encoded_pages: HashSet<_> = gid_encoded_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
let FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
} = select_items_with_document_folio_context(
|
||||
items,
|
||||
page_count,
|
||||
options.page_filter.as_ref(),
|
||||
);
|
||||
|
||||
let text_quality = analyze_text_quality(&items);
|
||||
merge_ocr_reasons(&mut ocr_reasons_by_page, text_quality.reasons_by_page);
|
||||
let layout = compute_layout_complexity(&items, &rects, &lines);
|
||||
let layout = compute_layout_complexity(&items, &layout_items, &rects, &lines);
|
||||
|
||||
let md = if options.mode == ProcessMode::Analyze {
|
||||
None
|
||||
@@ -3743,9 +3826,14 @@ fn process_document(
|
||||
options.markdown,
|
||||
&rects,
|
||||
&lines,
|
||||
&page_thresholds,
|
||||
struct_roles.as_ref(),
|
||||
&struct_tables,
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: struct_roles.as_ref(),
|
||||
struct_tables: &struct_tables,
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(removal_mask.as_slice()),
|
||||
},
|
||||
))
|
||||
};
|
||||
|
||||
@@ -5504,9 +5592,50 @@ mod looks_like_partial_table_tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct FolioFilteredItems {
|
||||
items: Vec<types::TextItem>,
|
||||
layout_items: Vec<types::TextItem>,
|
||||
removal_mask: Vec<bool>,
|
||||
removed_pages: HashSet<u32>,
|
||||
}
|
||||
|
||||
/// Resolve folios with complete document context, then select the caller's
|
||||
/// requested pages without losing those decisions.
|
||||
fn select_items_with_document_folio_context(
|
||||
all_items: Vec<types::TextItem>,
|
||||
page_count: u32,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> FolioFilteredItems {
|
||||
let (all_layout_items, all_removed_pages, all_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
let selected_page = |page: u32| page_filter.is_none_or(|filter| filter.contains(&page));
|
||||
|
||||
let (items, removal_mask) = all_items
|
||||
.into_iter()
|
||||
.zip(all_removal_mask)
|
||||
.filter(|(item, _)| selected_page(item.page))
|
||||
.unzip();
|
||||
let layout_items = all_layout_items
|
||||
.into_iter()
|
||||
.filter(|item| selected_page(item.page))
|
||||
.collect();
|
||||
let removed_pages = all_removed_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
|
||||
FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
}
|
||||
}
|
||||
|
||||
/// Analyse extracted items and rects for layout complexity.
|
||||
fn compute_layout_complexity(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
) -> LayoutComplexity {
|
||||
@@ -5528,6 +5657,7 @@ fn compute_layout_complexity(
|
||||
|
||||
// Check for side-by-side layout
|
||||
let owned_items: Vec<types::TextItem> = page_items.iter().map(|i| (*i).clone()).collect();
|
||||
let page_content_width = tables::content_width(&owned_items);
|
||||
let bands = markdown::split_side_by_side(&owned_items);
|
||||
|
||||
let band_ranges: Vec<(f32, f32)> = if bands.is_empty() {
|
||||
@@ -5578,7 +5708,12 @@ fn compute_layout_complexity(
|
||||
break;
|
||||
}
|
||||
// Heuristic fallback for borderless tables
|
||||
let heuristic_tables = tables::detect_tables(&band_items, base_size, false);
|
||||
let heuristic_tables = tables::detect_tables_with_page_width(
|
||||
&band_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
if has_data_table(&heuristic_tables) {
|
||||
found_table = true;
|
||||
break;
|
||||
@@ -5591,7 +5726,7 @@ fn compute_layout_complexity(
|
||||
|
||||
let mut pages_with_columns: Vec<u32> = Vec::new();
|
||||
for page in seen_pages {
|
||||
let cols = extractor::detect_columns(items, page, pages_with_tables.contains(&page));
|
||||
let cols = extractor::detect_columns(column_items, page, pages_with_tables.contains(&page));
|
||||
if cols.len() >= 2 {
|
||||
pages_with_columns.push(page);
|
||||
}
|
||||
@@ -5803,6 +5938,68 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sparse_folios_leave_no_layout_evidence() {
|
||||
let items = vec![
|
||||
test_item("1", 25.0, 20.0, 12.0, 10.0),
|
||||
test_item("2", 520.0, 60.0, 12.0, 10.0),
|
||||
];
|
||||
let (filtered, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(items.clone(), 1);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let filtered = compute_layout_complexity(&items, &filtered, &[], &[]);
|
||||
|
||||
assert!(!filtered.is_complex);
|
||||
assert!(filtered.pages_with_tables.is_empty());
|
||||
assert!(filtered.pages_with_columns.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_selection_keeps_document_wide_folio_layout_decisions() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
for row in 0..8 {
|
||||
let y = 20.0 + row as f32 * 8.0;
|
||||
let mut folio = test_item(&(row * 10 + page).to_string(), 25.0, y, 12.0, 10.0);
|
||||
folio.page = page;
|
||||
let mut footer =
|
||||
test_item(&format!("Footer row {row} summary"), 43.0, y, 470.0, 10.0);
|
||||
footer.page = page;
|
||||
let mut body = test_item(&format!("Body{row}"), 530.0, y, 55.0, 10.0);
|
||||
body.page = page;
|
||||
items.extend([folio, footer, body]);
|
||||
}
|
||||
}
|
||||
|
||||
let page_one_items: Vec<_> = items
|
||||
.iter()
|
||||
.filter(|item| item.page == 1)
|
||||
.cloned()
|
||||
.collect();
|
||||
let (page_local_layout, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(page_one_items.clone(), 4);
|
||||
let page_local = compute_layout_complexity(&page_one_items, &page_local_layout, &[], &[]);
|
||||
assert!(
|
||||
page_local.pages_with_columns.contains(&1),
|
||||
"fixture must reproduce page-local folio column evidence"
|
||||
);
|
||||
|
||||
let selected =
|
||||
select_items_with_document_folio_context(items, 4, Some(&HashSet::from([1])));
|
||||
assert_eq!(
|
||||
selected
|
||||
.removal_mask
|
||||
.iter()
|
||||
.filter(|remove| **remove)
|
||||
.count(),
|
||||
8
|
||||
);
|
||||
let document_wide =
|
||||
compute_layout_complexity(&selected.items, &selected.layout_items, &[], &[]);
|
||||
assert!(!document_wide.pages_with_columns.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_detect_encoding_issues_fffd() {
|
||||
assert!(detect_encoding_issues(
|
||||
|
||||
+165
-28
@@ -975,18 +975,54 @@ pub fn to_markdown_from_items_with_rects(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
) -> String {
|
||||
let document_page_count = items.iter().map(|item| item.page).max().unwrap_or(0);
|
||||
to_markdown_from_items_with_rects_and_page_count(items, options, rects, document_page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to Markdown with an authoritative PDF page count.
|
||||
///
|
||||
/// Use this overload when the owning PDF is available so trailing blank or
|
||||
/// unextracted pages are included in document-level header and folio coverage.
|
||||
/// Item-only callers can continue using [`to_markdown_from_items_with_rects`],
|
||||
/// which falls back to the highest observed item page.
|
||||
pub fn to_markdown_from_items_with_rects_and_page_count(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
document_page_count: u32,
|
||||
) -> String {
|
||||
to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
options,
|
||||
rects,
|
||||
&[],
|
||||
&HashMap::new(),
|
||||
None,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages: None,
|
||||
prefiltered_page_number_mask: None,
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) struct MarkdownDocumentContext<'a> {
|
||||
pub(crate) page_thresholds: &'a HashMap<u32, f32>,
|
||||
pub(crate) struct_roles:
|
||||
Option<&'a HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
pub(crate) struct_tables: &'a [crate::structure_tree::StructTable],
|
||||
pub(crate) page_count: u32,
|
||||
/// Pages where an upstream document-level pass removed folios. This keeps
|
||||
/// table-continuation classification consistent after masked items drop.
|
||||
pub(crate) prefiltered_page_number_pages: Option<&'a HashSet<u32>>,
|
||||
/// Document-level removal decisions aligned with this call's input items.
|
||||
/// Table detection consumes the original items; the mask is applied only
|
||||
/// after table claims have been established.
|
||||
pub(crate) prefiltered_page_number_mask: Option<&'a [bool]>,
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown, using rectangles and line segments for table detection.
|
||||
///
|
||||
/// Line-based detection runs first (strongest structural evidence), then rect-based,
|
||||
@@ -996,27 +1032,43 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
pdf_lines: &[crate::types::PdfLine],
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
struct_roles: Option<&HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
struct_tables: &[crate::structure_tree::StructTable],
|
||||
context: MarkdownDocumentContext<'_>,
|
||||
) -> String {
|
||||
use crate::tables::{
|
||||
detect_tables, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, try_build_rect_guided_table,
|
||||
content_width, detect_tables_from_lines, detect_tables_from_rects,
|
||||
detect_tables_from_struct_tree, detect_tables_with_page_width, try_build_rect_guided_table,
|
||||
};
|
||||
use crate::types::ItemType;
|
||||
|
||||
let MarkdownDocumentContext {
|
||||
page_thresholds,
|
||||
struct_roles,
|
||||
struct_tables,
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages,
|
||||
prefiltered_page_number_mask,
|
||||
} = context;
|
||||
|
||||
if items.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Table detection must retain the original collection because short
|
||||
// numeric table cells can be indistinguishable from folios until
|
||||
// structural context is available. A precomputed mask carries the
|
||||
// document-wide decision without removing items before table claims.
|
||||
debug_assert!(prefiltered_page_number_mask.is_none_or(|mask| mask.len() == items.len()));
|
||||
let has_precomputed_page_number_mask = prefiltered_page_number_mask.is_some();
|
||||
let removed_page_number_pages = prefiltered_page_number_pages.cloned().unwrap_or_default();
|
||||
|
||||
// Separate images and links from text items
|
||||
let mut images: Vec<TextItem> = Vec::new();
|
||||
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
let mut links: Vec<TextItem> = Vec::new();
|
||||
let mut text_items: Vec<TextItem> = Vec::new();
|
||||
let mut text_item_page_number_mask: Vec<bool> = Vec::new();
|
||||
|
||||
for item in items {
|
||||
for (input_index, item) in items.into_iter().enumerate() {
|
||||
match &item.item_type {
|
||||
ItemType::Image => {
|
||||
page_image_regions.entry(item.page).or_default().push((
|
||||
@@ -1035,6 +1087,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
ItemType::Text | ItemType::FormField => {
|
||||
text_item_page_number_mask.push(
|
||||
prefiltered_page_number_mask
|
||||
.and_then(|mask| mask.get(input_index))
|
||||
.copied()
|
||||
.unwrap_or(false),
|
||||
);
|
||||
text_items.push(item);
|
||||
}
|
||||
}
|
||||
@@ -1075,7 +1133,6 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
let mut pages: Vec<u32> = page_groups.keys().copied().collect();
|
||||
pages.sort();
|
||||
let page_count = pages.last().copied().unwrap_or(0) + 1;
|
||||
|
||||
// Track band splits per page so we can split non-table items later
|
||||
let mut page_band_splits: HashMap<u32, Vec<(f32, f32)>> = HashMap::new();
|
||||
@@ -1087,6 +1144,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
for page in pages {
|
||||
let group = page_groups.get(&page).unwrap();
|
||||
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
|
||||
let page_content_width = content_width(&page_items);
|
||||
|
||||
// Chart-bar regions: bar charts drawn as filled rects read as cell
|
||||
// rects or aligned text and get gridded into phantom tables. Their
|
||||
@@ -1128,7 +1186,10 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
});
|
||||
let chart_prose_columns = chart_prose_split.is_some();
|
||||
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
// Check for side-by-side table layout using the original items. Sparse
|
||||
// numeric cells need table context before they can be distinguished
|
||||
// safely from folios; cleaned evidence is reserved for column and
|
||||
// final non-table layout decisions.
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// A rect table crossing a proposed split boundary means the "gutter"
|
||||
// is really the gap between ruled and borderless table columns —
|
||||
@@ -1368,7 +1429,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// table can share the prose anchors. Reject only candidates
|
||||
// whose cells prove they are parallel prose fragments.
|
||||
let reject_parallel_prose = chart_prose_columns && !was_split;
|
||||
let tables = detect_tables(subset_items, base_size, false);
|
||||
let tables = detect_tables_with_page_width(
|
||||
subset_items,
|
||||
base_size,
|
||||
false,
|
||||
page_content_width,
|
||||
);
|
||||
for table in tables {
|
||||
if reject_parallel_prose && is_parallel_prose_table(&table) {
|
||||
log::debug!(
|
||||
@@ -1549,7 +1615,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// and reject chart-page prose candidates individually below.
|
||||
let skip_body_font =
|
||||
merged_retry_skips_body_font(detected_columns, !chart_regions.is_empty());
|
||||
let heuristic_tables = detect_tables(&chart_free, base_size, skip_body_font);
|
||||
let heuristic_tables = detect_tables_with_page_width(
|
||||
&chart_free,
|
||||
base_size,
|
||||
skip_body_font,
|
||||
page_content_width,
|
||||
);
|
||||
for table in &heuristic_tables {
|
||||
if !chart_regions.is_empty() && is_parallel_prose_table(table) {
|
||||
log::debug!(
|
||||
@@ -1634,16 +1705,20 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
};
|
||||
|
||||
// Filter out table items and process the rest
|
||||
let non_table_items: Vec<TextItem> = text_items
|
||||
let non_table_items: Vec<(usize, TextItem)> = text_items
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.filter(|(idx, _)| !table_items.contains(idx))
|
||||
.map(|(_, item)| item)
|
||||
.collect();
|
||||
|
||||
// Find pages that are table-only (no remaining non-table text)
|
||||
let table_only_pages: HashSet<u32> = {
|
||||
let pages_with_text: HashSet<u32> = non_table_items.iter().map(|i| i.page).collect();
|
||||
let mut pages_with_text: HashSet<u32> =
|
||||
non_table_items.iter().map(|(_, item)| item.page).collect();
|
||||
// Preserve the pre-filter continuation classification: a page that
|
||||
// originally also contained a folio does not become table-only merely
|
||||
// because an upstream document-level pass removed it.
|
||||
pages_with_text.extend(removed_page_number_pages);
|
||||
page_tables
|
||||
.keys()
|
||||
.filter(|p| !pages_with_text.contains(p))
|
||||
@@ -1658,11 +1733,25 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// column detection on pages where table column gaps would be misidentified.
|
||||
let table_page_set: HashSet<u32> = page_tables.keys().copied().collect();
|
||||
|
||||
let non_table_items = if has_precomputed_page_number_mask {
|
||||
non_table_items
|
||||
.into_iter()
|
||||
.filter(|(index, _)| !text_item_page_number_mask[*index])
|
||||
.map(|(_, item)| item)
|
||||
.collect()
|
||||
} else {
|
||||
crate::extractor::filter_markdown_page_numbers_with_removed_pages(
|
||||
non_table_items.into_iter().map(|(_, item)| item).collect(),
|
||||
document_page_count,
|
||||
)
|
||||
.0
|
||||
};
|
||||
|
||||
// Split non-table items by band boundaries before line grouping so that
|
||||
// items from different side-by-side zones (e.g. left/right month columns
|
||||
// in a calendar) don't merge into the same line.
|
||||
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
|
||||
crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
non_table_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1690,13 +1779,14 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
// Process unsplit pages normally
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
let mut all_lines =
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
// Process each split page's bands independently, then interleave
|
||||
// by Y position so paired zones (e.g. left/right months) appear together.
|
||||
let mut split_pages: Vec<u32> = split_page_items.keys().copied().collect();
|
||||
@@ -1714,7 +1804,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !band_items.is_empty() {
|
||||
page_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
band_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1748,7 +1838,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !column_items.is_empty() {
|
||||
zone_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
column_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1789,7 +1879,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
item.y >= low || item_is_in_chart_region(item, chart_regions)
|
||||
});
|
||||
all_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
chart_zone,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1808,7 +1898,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Strip repeated headers/footers before conversion
|
||||
let lines = if options.strip_headers_footers {
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
preprocess::strip_repeated_lines(lines, document_page_count)
|
||||
} else {
|
||||
lines
|
||||
};
|
||||
@@ -1921,6 +2011,53 @@ mod tests {
|
||||
it
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn precomputed_folio_mask_preserves_numeric_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for column in 0..2 {
|
||||
let mut item = make_item_w(
|
||||
110.0 + column as f32 * 100.0,
|
||||
30.0 + row as f32 * 20.0,
|
||||
20.0,
|
||||
1,
|
||||
);
|
||||
item.text = (row * 2 + column + 1).to_string();
|
||||
items.push(item);
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + column as f32 * 100.0,
|
||||
y: 20.0 + row as f32 * 20.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Simulate document-level folio decisions that would remove every
|
||||
// short numeric item if applied before structural table detection.
|
||||
let removal_mask = vec![true; items.len()];
|
||||
let removed_pages = HashSet::from([1]);
|
||||
let markdown = to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
MarkdownOptions::default(),
|
||||
&rects,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: 1,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(&removal_mask),
|
||||
},
|
||||
);
|
||||
|
||||
assert!(markdown.contains("|1|2|"), "{markdown}");
|
||||
assert!(markdown.contains("|7|8|"), "{markdown}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn early_layout_excludes_chart_items_before_column_detection() {
|
||||
let mut items = Vec::new();
|
||||
|
||||
+20
-64
@@ -3,6 +3,7 @@
|
||||
use regex::Regex;
|
||||
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
use crate::text_utils::is_page_number_line;
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
@@ -145,7 +146,7 @@ fn fix_hyphenation(text: &str) -> String {
|
||||
result
|
||||
}
|
||||
|
||||
/// Remove standalone page numbers (lines that are just 1-4 digit numbers)
|
||||
/// Remove isolated page-number expressions from Markdown.
|
||||
fn remove_page_numbers(text: &str) -> String {
|
||||
let mut result = Vec::new();
|
||||
let lines: Vec<&str> = text.lines().collect();
|
||||
@@ -183,69 +184,6 @@ fn remove_page_numbers(text: &str) -> String {
|
||||
result.join("\n")
|
||||
}
|
||||
|
||||
/// Check if a line looks like a page number
|
||||
fn is_page_number_line(trimmed: &str) -> bool {
|
||||
// Empty lines are not page numbers
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Pattern 1: Just a number (1-4 digits)
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Pattern 2: "Page X of Y" or "Page X" or "Page of" (placeholder)
|
||||
let lower = trimmed.to_lowercase();
|
||||
if let Some(rest) = lower.strip_prefix("page") {
|
||||
let rest = rest.trim();
|
||||
// "Page of" (empty page numbers)
|
||||
if rest == "of" || rest.starts_with("of ") {
|
||||
return true;
|
||||
}
|
||||
// "Page X" or "Page X of Y"
|
||||
if rest
|
||||
.chars()
|
||||
.next()
|
||||
.map(|c| c.is_ascii_digit())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Just "Page" followed by whitespace and maybe "of"
|
||||
if rest.is_empty()
|
||||
|| rest
|
||||
.split_whitespace()
|
||||
.all(|w| w == "of" || w.chars().all(|c| c.is_ascii_digit()))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 3: "X of Y" where X and Y are numbers
|
||||
if let Some(of_idx) = trimmed.find(" of ") {
|
||||
let before = trimmed[..of_idx].trim();
|
||||
let after = trimmed[of_idx + 4..].trim();
|
||||
if before.chars().all(|c| c.is_ascii_digit())
|
||||
&& after.chars().all(|c| c.is_ascii_digit())
|
||||
&& !before.is_empty()
|
||||
&& !after.is_empty()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 4: "- X -" centered page number
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if inner.chars().all(|c| c.is_ascii_digit()) && !inner.is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Convert URLs to markdown links
|
||||
fn format_urls(text: &str) -> String {
|
||||
use once_cell::sync::Lazy;
|
||||
@@ -489,12 +427,14 @@ mod tests {
|
||||
fn test_is_page_number_page_x() {
|
||||
assert!(is_page_number_line("Page 5"));
|
||||
assert!(is_page_number_line("page 12"));
|
||||
assert!(is_page_number_line("Page123"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_page_x_of_y() {
|
||||
assert!(is_page_number_line("Page 3 of 10"));
|
||||
assert!(is_page_number_line("page 1 of 5"));
|
||||
assert!(is_page_number_line("Page 3 of 10 Report header"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -526,6 +466,12 @@ mod tests {
|
||||
assert!(!is_page_number_line("Total: 500"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_labeled_running_header() {
|
||||
assert!(is_page_number_line("Page 42 Chapter 5"));
|
||||
assert!(is_page_number_line("Page 42 explains the result"));
|
||||
}
|
||||
|
||||
// --- remove_page_numbers ---
|
||||
|
||||
#[test]
|
||||
@@ -551,6 +497,16 @@ mod tests {
|
||||
assert!(result.contains("42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_labeled_header_with_content() {
|
||||
let input = "Content\n\nPage 42 explains the result\n---\nEnd";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert!(!result.contains("Page 42 explains the result"));
|
||||
assert!(result.contains("Content"));
|
||||
assert!(result.contains("End"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_multiple_patterns() {
|
||||
let input = "\n1\n\nContent\n\n2\n\n---\nMore\n\n3\n";
|
||||
|
||||
@@ -18,7 +18,15 @@ use super::{Table, TableDetectionMode};
|
||||
///
|
||||
/// Returns `(merged_items, index_map)` where `index_map[merged_idx]` contains
|
||||
/// the original item indices that were merged into that item.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Vec<usize>>) {
|
||||
merge_adjacent_items_preserving(items, &std::collections::HashSet::new())
|
||||
}
|
||||
|
||||
fn merge_adjacent_items_preserving(
|
||||
items: &[TextItem],
|
||||
preserved_indices: &std::collections::HashSet<usize>,
|
||||
) -> (Vec<TextItem>, Vec<Vec<usize>>) {
|
||||
if items.is_empty() {
|
||||
return (vec![], vec![]);
|
||||
}
|
||||
@@ -72,6 +80,23 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
break;
|
||||
}
|
||||
|
||||
// Proven replacement cells must retain their own decoration.
|
||||
// Otherwise an adjacent old/new pair inherits only the first
|
||||
// fragment's flags and can lose the live table evidence.
|
||||
let decoration_changes = indices.iter().any(|index| {
|
||||
let merged_item = &items[*index];
|
||||
next_item.is_underline != merged_item.is_underline
|
||||
|| next_item.is_strikeout != merged_item.is_strikeout
|
||||
});
|
||||
if decoration_changes
|
||||
&& (indices
|
||||
.iter()
|
||||
.any(|index| preserved_indices.contains(index))
|
||||
|| preserved_indices.contains(&next_idx))
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
let gap = next_item.x - end_x;
|
||||
// Stop if gap exceeds threshold (inter-column gap)
|
||||
if gap > x_gap_max {
|
||||
@@ -137,17 +162,319 @@ fn expand_consolidated_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<usize>)
|
||||
(expanded, index_map)
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct RedlineEditRegion {
|
||||
x_ranges: Vec<(f32, f32)>,
|
||||
y_min: f32,
|
||||
y_max: f32,
|
||||
spans_page_width: bool,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct UnderlinedTableColumn {
|
||||
x_min: f32,
|
||||
x_max: f32,
|
||||
}
|
||||
|
||||
pub(crate) fn content_width(items: &[TextItem]) -> f32 {
|
||||
let x_min = items
|
||||
.iter()
|
||||
.map(|item| item.x)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let x_max = items
|
||||
.iter()
|
||||
.map(|item| item.x + item.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
(x_max - x_min).max(1.0)
|
||||
}
|
||||
|
||||
/// Spatial regions where multiple strikeout rows indicate a redline edit block.
|
||||
///
|
||||
/// A lone deletion can occur inside or beside a real table, so it must not
|
||||
/// globally suppress underlined table cells. Closely spaced strikeout rows are
|
||||
/// different: together with nearby underlines they form the overlapping
|
||||
/// old/new text layers used by legislative redlines, and those decorations
|
||||
/// must not become heuristic column evidence.
|
||||
fn redline_edit_regions(items: &[TextItem], page_width: f32) -> Vec<RedlineEditRegion> {
|
||||
const ROW_DEDUP_TOLERANCE: f32 = 8.0;
|
||||
const MAX_CLUSTER_GAP: f32 = 64.0;
|
||||
const Y_PADDING: f32 = 36.0;
|
||||
const X_PADDING: f32 = 12.0;
|
||||
const PAGE_WIDTH_RATIO: f32 = 0.35;
|
||||
const MAX_HORIZONTAL_GAP_RATIO: f32 = 0.20;
|
||||
|
||||
let mut strikeouts: Vec<&TextItem> = items.iter().filter(|item| item.is_strikeout).collect();
|
||||
strikeouts.sort_by(|a, b| a.y.total_cmp(&b.y));
|
||||
|
||||
let mut rows: Vec<(f32, Vec<(f32, f32)>)> = Vec::new();
|
||||
for item in strikeouts {
|
||||
if let Some((_, x_ranges)) = rows
|
||||
.last_mut()
|
||||
.filter(|(y, _)| (item.y - *y).abs() <= ROW_DEDUP_TOLERANCE)
|
||||
{
|
||||
x_ranges.push((item.x, item.x + item.width));
|
||||
} else {
|
||||
rows.push((item.y, vec![(item.x, item.x + item.width)]));
|
||||
}
|
||||
}
|
||||
|
||||
let mut regions = Vec::new();
|
||||
let mut start = 0;
|
||||
while start < rows.len() {
|
||||
let mut end = start + 1;
|
||||
while end < rows.len() && rows[end].0 - rows[end - 1].0 <= MAX_CLUSTER_GAP {
|
||||
end += 1;
|
||||
}
|
||||
if end - start >= 2 {
|
||||
let mut x_ranges: Vec<(f32, f32)> = rows[start..end]
|
||||
.iter()
|
||||
.flat_map(|(_, x_ranges)| x_ranges.iter().copied())
|
||||
.collect();
|
||||
x_ranges.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
let mut merged_ranges: Vec<(f32, f32)> = Vec::new();
|
||||
for (x_min, x_max) in x_ranges {
|
||||
// Modest inline gaps can separate fragments of one flowing
|
||||
// edit; column-scale gaps must remain distinct spatial masks.
|
||||
if let Some((_, merged_max)) = merged_ranges.last_mut().filter(|(_, merged_max)| {
|
||||
x_min - *merged_max <= page_width * MAX_HORIZONTAL_GAP_RATIO
|
||||
}) {
|
||||
*merged_max = merged_max.max(x_max);
|
||||
} else {
|
||||
merged_ranges.push((x_min, x_max));
|
||||
}
|
||||
}
|
||||
let covered_width: f32 = merged_ranges
|
||||
.iter()
|
||||
.map(|(x_min, x_max)| x_max - x_min)
|
||||
.sum();
|
||||
for (x_min, x_max) in &mut merged_ranges {
|
||||
*x_min -= X_PADDING;
|
||||
*x_max += X_PADDING;
|
||||
}
|
||||
// Redlines spread across much of the text width are flowing prose,
|
||||
// so their whole Y-band is ambiguous. Compact edits can be scoped
|
||||
// to their actual horizontal spans without hiding content between
|
||||
// unrelated edits in separate columns.
|
||||
regions.push(RedlineEditRegion {
|
||||
x_ranges: merged_ranges,
|
||||
y_min: rows[start].0 - Y_PADDING,
|
||||
y_max: rows[end - 1].0 + Y_PADDING,
|
||||
spans_page_width: covered_width >= page_width * PAGE_WIDTH_RATIO,
|
||||
});
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
// Padding can make neighboring clusters overlap. Partition that overlap at
|
||||
// its midpoint so each Y position maps to one region without combining
|
||||
// horizontally unrelated edits.
|
||||
for index in 1..regions.len() {
|
||||
let (previous, current) = regions.split_at_mut(index);
|
||||
let previous = &mut previous[index - 1];
|
||||
let current = &mut current[0];
|
||||
if previous.y_max >= current.y_min {
|
||||
let boundary = (previous.y_max + current.y_min) / 2.0;
|
||||
previous.y_max = boundary;
|
||||
current.y_min = boundary;
|
||||
}
|
||||
}
|
||||
regions
|
||||
}
|
||||
|
||||
fn overlaps_redline_x(item: &TextItem, region: &RedlineEditRegion) -> bool {
|
||||
let range_index = region
|
||||
.x_ranges
|
||||
.partition_point(|(_, x_max)| *x_max < item.x);
|
||||
region
|
||||
.x_ranges
|
||||
.get(range_index)
|
||||
.is_some_and(|(x_min, _)| item.x + item.width >= *x_min)
|
||||
}
|
||||
|
||||
fn has_distinct_rows(
|
||||
items: &[&TextItem],
|
||||
underlined_only: bool,
|
||||
required: usize,
|
||||
tolerance: f32,
|
||||
) -> bool {
|
||||
debug_assert!(required <= 3);
|
||||
let mut rows = [0.0; 3];
|
||||
let mut row_count = 0;
|
||||
for item in items {
|
||||
if underlined_only && !item.is_underline {
|
||||
continue;
|
||||
}
|
||||
if rows[..row_count]
|
||||
.iter()
|
||||
.all(|row| (item.y - row).abs() > tolerance)
|
||||
{
|
||||
rows[row_count] = item.y;
|
||||
row_count += 1;
|
||||
if row_count == required {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Aligned live items seeded by underline evidence form revised table columns.
|
||||
/// Compact edits accept one replacement backed by surrounding live rows; wide
|
||||
/// prose-like edits require replacements on at least two distinct rows.
|
||||
fn underlined_table_columns(
|
||||
items: &[TextItem],
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
) -> Vec<Vec<UnderlinedTableColumn>> {
|
||||
const X_ALIGNMENT_TOLERANCE: f32 = 4.0;
|
||||
const ROW_DEDUP_TOLERANCE: f32 = 8.0;
|
||||
|
||||
// Sort once globally by X, then partition candidates into their unique Y
|
||||
// regions. Each regional vector remains X-sorted without another sort.
|
||||
let mut live_items: Vec<&TextItem> = items.iter().filter(|item| !item.is_strikeout).collect();
|
||||
live_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let mut candidates_by_region: Vec<Vec<&TextItem>> = vec![Vec::new(); redline_regions.len()];
|
||||
for item in live_items {
|
||||
if let Some(region_index) = redline_region_at_y(redline_regions, item.y) {
|
||||
if overlaps_redline_x(item, &redline_regions[region_index]) {
|
||||
candidates_by_region[region_index].push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let mut columns_by_region = Vec::with_capacity(redline_regions.len());
|
||||
for (region, candidates) in redline_regions.iter().zip(candidates_by_region) {
|
||||
let mut columns: Vec<UnderlinedTableColumn> = Vec::new();
|
||||
let mut start = 0;
|
||||
while start < candidates.len() {
|
||||
let mut end = start + 1;
|
||||
while end < candidates.len()
|
||||
&& candidates[end].x - candidates[start].x <= X_ALIGNMENT_TOLERANCE
|
||||
{
|
||||
end += 1;
|
||||
}
|
||||
let aligned_items = &candidates[start..end];
|
||||
let enough_live_rows = has_distinct_rows(aligned_items, false, 3, ROW_DEDUP_TOLERANCE);
|
||||
let required_underlined_rows = if region.spans_page_width { 2 } else { 1 };
|
||||
let enough_underlined_rows = has_distinct_rows(
|
||||
aligned_items,
|
||||
true,
|
||||
required_underlined_rows,
|
||||
ROW_DEDUP_TOLERANCE,
|
||||
);
|
||||
if enough_live_rows && enough_underlined_rows {
|
||||
let x_min = candidates[start].x - X_ALIGNMENT_TOLERANCE;
|
||||
let x_max = candidates[end - 1].x + X_ALIGNMENT_TOLERANCE;
|
||||
if let Some(column) = columns.last_mut().filter(|column| column.x_max >= x_min) {
|
||||
column.x_max = column.x_max.max(x_max);
|
||||
} else {
|
||||
columns.push(UnderlinedTableColumn { x_min, x_max });
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
columns_by_region.push(columns);
|
||||
}
|
||||
columns_by_region
|
||||
}
|
||||
|
||||
fn redline_region_at_y(redline_regions: &[RedlineEditRegion], y: f32) -> Option<usize> {
|
||||
let region_index = redline_regions.partition_point(|region| region.y_max < y);
|
||||
redline_regions
|
||||
.get(region_index)
|
||||
.filter(|region| y >= region.y_min)
|
||||
.map(|_| region_index)
|
||||
}
|
||||
|
||||
fn is_heuristic_table_evidence(
|
||||
item: &TextItem,
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
underlined_table_columns: &[Vec<UnderlinedTableColumn>],
|
||||
) -> bool {
|
||||
if item.is_strikeout {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(region_index) = redline_region_at_y(redline_regions, item.y) else {
|
||||
return true;
|
||||
};
|
||||
let region = &redline_regions[region_index];
|
||||
let columns = &underlined_table_columns[region_index];
|
||||
if region.spans_page_width && columns.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlaps_x = overlaps_redline_x(item, region);
|
||||
!overlaps_x || is_revised_table_cell(item, columns)
|
||||
}
|
||||
|
||||
fn is_revised_table_cell(
|
||||
item: &TextItem,
|
||||
underlined_table_columns: &[UnderlinedTableColumn],
|
||||
) -> bool {
|
||||
let column_index = underlined_table_columns.partition_point(|column| column.x_max < item.x);
|
||||
underlined_table_columns
|
||||
.get(column_index)
|
||||
.is_some_and(|column| item.x >= column.x_min)
|
||||
}
|
||||
|
||||
fn revised_table_cell_indices(
|
||||
items: &[TextItem],
|
||||
redline_regions: &[RedlineEditRegion],
|
||||
underlined_table_columns: &[Vec<UnderlinedTableColumn>],
|
||||
) -> std::collections::HashSet<usize> {
|
||||
items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(item_index, item)| {
|
||||
let region_index = redline_region_at_y(redline_regions, item.y)?;
|
||||
let region = &redline_regions[region_index];
|
||||
(item.is_underline
|
||||
&& overlaps_redline_x(item, region)
|
||||
&& is_revised_table_cell(item, &underlined_table_columns[region_index]))
|
||||
.then_some(item_index)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Detect tables in a set of text items from a single page
|
||||
pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bool) -> Vec<Table> {
|
||||
detect_tables_with_page_width(items, base_font_size, skip_body_font, content_width(items))
|
||||
}
|
||||
|
||||
/// Detect tables in a subset while using the full page's text width for
|
||||
/// page-spanning redline classification.
|
||||
pub(crate) fn detect_tables_with_page_width(
|
||||
items: &[TextItem],
|
||||
base_font_size: f32,
|
||||
skip_body_font: bool,
|
||||
page_width: f32,
|
||||
) -> Vec<Table> {
|
||||
if items.len() < 6 {
|
||||
return vec![];
|
||||
}
|
||||
// Compute these before consolidation: adjacent old/new text can merge and
|
||||
// inherit only the first fragment's decoration flags.
|
||||
let redline_regions = redline_edit_regions(items, page_width);
|
||||
let underlined_table_columns = underlined_table_columns(items, &redline_regions);
|
||||
let source_evidence: Vec<bool> = items
|
||||
.iter()
|
||||
.map(|item| is_heuristic_table_evidence(item, &redline_regions, &underlined_table_columns))
|
||||
.collect();
|
||||
let revised_table_cells =
|
||||
revised_table_cell_indices(items, &redline_regions, &underlined_table_columns);
|
||||
|
||||
// Step 1: Merge adjacent single-char items into words (handles per-character PDFs)
|
||||
let (merged_items, merge_map) = merge_adjacent_items(items);
|
||||
let (merged_items, merge_map) = merge_adjacent_items_preserving(items, &revised_table_cells);
|
||||
|
||||
// Step 2: Expand consolidated financial items (e.g. "$ 1,234 $ 5,678" → sub-items)
|
||||
let (expanded_items, expand_map) = expand_consolidated_items(&merged_items);
|
||||
let expanded_evidence: Vec<bool> = expand_map
|
||||
.iter()
|
||||
.map(|&merged_index| {
|
||||
merge_map[merged_index]
|
||||
.iter()
|
||||
.all(|&source_index| source_evidence[source_index])
|
||||
})
|
||||
.collect();
|
||||
let items = &expanded_items[..]; // shadow parameter — all detection uses processed items
|
||||
|
||||
let mut tables = Vec::new();
|
||||
@@ -159,7 +486,11 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
let table_candidates: Vec<(usize, &TextItem)> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| item.font_size <= table_font_threshold && item.font_size >= 6.0)
|
||||
.filter(|(index, item)| {
|
||||
expanded_evidence[*index]
|
||||
&& item.font_size <= table_font_threshold
|
||||
&& item.font_size >= 6.0
|
||||
})
|
||||
.collect();
|
||||
|
||||
if table_candidates.len() >= 6 {
|
||||
@@ -208,6 +539,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
.enumerate()
|
||||
.filter(|(idx, item)| {
|
||||
!claimed_indices.contains(idx)
|
||||
&& expanded_evidence[*idx]
|
||||
&& item.font_size >= body_font_low
|
||||
&& item.font_size <= body_font_high
|
||||
&& item.font_size >= 6.0
|
||||
@@ -1646,6 +1978,362 @@ fn try_add_label_column(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn body_item(text: &str, x: f32, y: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y,
|
||||
width: 90.0,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_preserves_redline_boundaries() {
|
||||
let old = body_item("old value", 220.0, 700.0, true);
|
||||
let mut replacement = body_item("new value", 312.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let preserved_indices = std::collections::HashSet::from([1]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[old, replacement], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0], vec![1]]);
|
||||
assert!(merged[0].is_strikeout);
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_keeps_boundary_after_preserved_fragment() {
|
||||
let mut prefix = body_item("prefix", 20.0, 700.0, false);
|
||||
prefix.is_underline = true;
|
||||
let mut replacement = body_item("replacement", 111.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let deleted = body_item("deleted", 202.0, 700.0, true);
|
||||
let preserved_indices = std::collections::HashSet::from([1]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[prefix, replacement, deleted], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0, 1], vec![2]]);
|
||||
assert!(merged[0].is_underline);
|
||||
assert!(merged[1].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_adjacent_items_keeps_preserved_fragment_out_of_mixed_run() {
|
||||
let mut prefix = body_item("prefix", 20.0, 700.0, false);
|
||||
prefix.is_underline = true;
|
||||
let deleted = body_item("deleted", 111.0, 700.0, true);
|
||||
let mut replacement = body_item("replacement", 202.0, 700.0, false);
|
||||
replacement.is_underline = true;
|
||||
let preserved_indices = std::collections::HashSet::from([2]);
|
||||
|
||||
let (merged, index_map) =
|
||||
merge_adjacent_items_preserving(&[prefix, deleted, replacement], &preserved_indices);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(index_map, vec![vec![0, 1], vec![2]]);
|
||||
assert!(merged[0].is_underline);
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn body_font_redline_deletions_do_not_create_a_table() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("live paragraph text", 50.0, y, false));
|
||||
items.push(body_item("deleted wording", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"aligned strikeout overlays are source edits, not table columns"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn underlined_body_font_table_without_deletions_is_still_detected() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"underline-only tables must keep their heuristic evidence"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn body_font_table_with_one_deletion_is_still_detected() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("row value", 220.0, y, row == 0));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"one revised cell must not suppress an otherwise complete table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unrelated_strikeout_does_not_remove_underlined_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
items.push(body_item("deleted prose", 50.0, 300.0, true));
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"a distant deletion must not suppress underlined table columns"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nearby_redline_rows_do_not_remove_plain_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("row value", 220.0, y, false));
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("deleted prose", 400.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"redline rows must suppress decorations, not nearby live table cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separate_strikeout_columns_do_not_suppress_content_between_them() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 160.0, y, false));
|
||||
items.push(body_item("row value", 280.0, y, false));
|
||||
}
|
||||
for (row, y) in [700.0, 684.0, 668.0, 652.0].into_iter().enumerate() {
|
||||
let x = if row % 2 == 0 { 50.0 } else { 420.0 };
|
||||
let mut deletion = body_item("old", x, y, true);
|
||||
deletion.width = 30.0;
|
||||
items.push(deletion);
|
||||
}
|
||||
|
||||
let regions = redline_edit_regions(&items, content_width(&items));
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert_eq!(regions[0].x_ranges.len(), 2);
|
||||
assert!(!regions[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"separate edit columns must not create a suppression bridge across the page"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_redline_rows_do_not_turn_fragmented_prose_into_a_table() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("line number", 50.0, y, false));
|
||||
items.push(body_item("live prose fragment", 160.0, y, false));
|
||||
let strikeout_x = if row % 2 == 0 { 280.0 } else { 430.0 };
|
||||
items.push(body_item("deleted prose", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"full-width redline prose must not retain table-shaped fragments"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_redline_prose_ignores_underlines_outside_edit_span() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
let mut line_number = body_item("line number", 50.0, y, false);
|
||||
line_number.is_underline = row < 2;
|
||||
items.push(line_number);
|
||||
items.push(body_item("live prose fragment", 160.0, y, false));
|
||||
let strikeout_x = if row % 2 == 0 { 280.0 } else { 430.0 };
|
||||
items.push(body_item("deleted prose", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(
|
||||
detect_tables(&items, 12.0, false).is_empty(),
|
||||
"underlines outside the edit span must not disable the wide-prose veto"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nearby_redline_rows_do_not_remove_separate_underlined_table_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("deleted prose", 400.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"redline rows must not suppress a horizontally separate underlined table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiple_revised_rows_do_not_remove_underlined_table_column() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let replacement_x = 220.0 + (row % 2) as f32 * 2.0;
|
||||
let mut value = body_item("new value", replacement_x, y, false);
|
||||
value.is_underline = true;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"aligned replacement cells must preserve a partially revised table"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn single_replacement_uses_aligned_live_table_rows() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = row == 0;
|
||||
items.push(value);
|
||||
}
|
||||
for y in [700.0, 652.0, 604.0] {
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"one replacement cell must retain its aligned live table column"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wide_revised_table_keeps_repeated_replacement_evidence() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
let mut value = body_item("row value", 220.0, y, false);
|
||||
value.is_underline = row < 2;
|
||||
items.push(value);
|
||||
let strikeout_x = if row % 2 == 0 { 220.0 } else { 400.0 };
|
||||
items.push(body_item("old value", strikeout_x, y, true));
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"wide edits must keep a table column with repeated replacements"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn revised_financial_columns_survive_item_expansion() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..4 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("old value", 180.0, y, true));
|
||||
let mut values = body_item("$ 100 $ 200 $ 300", 180.0, y, false);
|
||||
values.width = 300.0;
|
||||
values.is_underline = true;
|
||||
items.push(values);
|
||||
}
|
||||
|
||||
let tables = detect_tables(&items, 12.0, false);
|
||||
assert!(
|
||||
tables.iter().any(|table| table.columns.len() >= 4),
|
||||
"all expanded financial columns must inherit their source evidence"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_layout_band_uses_full_page_width_for_redline_scope() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..8 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 190.0, y, false));
|
||||
items.push(body_item("old value", 300.0, y, true));
|
||||
let mut replacement = body_item("new value", 300.0, y, false);
|
||||
replacement.is_underline = true;
|
||||
items.push(replacement);
|
||||
}
|
||||
|
||||
assert!(redline_edit_regions(&items, content_width(&items))[0].spans_page_width);
|
||||
assert!(!redline_edit_regions(&items, 500.0)[0].spans_page_width);
|
||||
assert!(
|
||||
!detect_tables_with_page_width(&items, 12.0, false, 500.0).is_empty(),
|
||||
"a narrow band must use full-page context to preserve revised table cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adjacent_revised_fragments_preserve_live_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..4 {
|
||||
let y = 700.0 - row as f32 * 16.0;
|
||||
items.push(body_item("row label", 50.0, y, false));
|
||||
items.push(body_item("old value", 220.0, y, true));
|
||||
let mut replacement = body_item("new value", 312.0, y, false);
|
||||
replacement.is_underline = true;
|
||||
items.push(replacement);
|
||||
}
|
||||
|
||||
assert!(
|
||||
!detect_tables(&items, 12.0, false).is_empty(),
|
||||
"adjacent old/new fragments must retain the live revised cells"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_rejects_toc() {
|
||||
|
||||
+27
-1
@@ -436,7 +436,8 @@ pub(crate) fn recover_header_row(
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, item)| {
|
||||
item.font_size > small_font_threshold
|
||||
!item.is_strikeout
|
||||
&& item.font_size > small_font_threshold
|
||||
&& item.y > first_row_y
|
||||
&& item.y <= first_row_y + row_gap_limit
|
||||
})
|
||||
@@ -804,6 +805,31 @@ mod tests {
|
||||
assert_eq!(table.rows.len(), rows_before);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_skips_strikeout_candidates() {
|
||||
let mut old_col1 = make_item("Old Col1", 100.0, 520.0, 12.0);
|
||||
old_col1.is_strikeout = true;
|
||||
let mut old_col2 = make_item("Old Col2", 200.0, 520.0, 12.0);
|
||||
old_col2.is_strikeout = true;
|
||||
let all_items = vec![
|
||||
old_col1,
|
||||
old_col2,
|
||||
make_item("A", 100.0, 500.0, 8.0),
|
||||
make_item("B", 200.0, 500.0, 8.0),
|
||||
];
|
||||
let mut table = Table {
|
||||
columns: vec![100.0, 200.0],
|
||||
rows: vec![500.0, 480.0],
|
||||
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||
item_indices: vec![2, 3],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
recover_header_row(&mut table, &all_items, 9.0);
|
||||
assert_eq!(table.rows.len(), 2);
|
||||
assert_eq!(table.cells[0], vec!["A", "B"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recover_header_row_too_far_above() {
|
||||
let all_items = vec![
|
||||
|
||||
+3
-1
@@ -12,7 +12,9 @@ mod grid;
|
||||
pub mod structured;
|
||||
|
||||
pub use detect_heuristic::detect_tables;
|
||||
pub(crate) use detect_heuristic::is_table_of_contents;
|
||||
pub(crate) use detect_heuristic::{
|
||||
content_width, detect_tables_with_page_width, is_table_of_contents,
|
||||
};
|
||||
pub use detect_lines::detect_tables_from_lines;
|
||||
pub(crate) use detect_lines::detect_vector_grid_tables_from_lines;
|
||||
pub(crate) use detect_rects::cluster_rects;
|
||||
|
||||
@@ -7,6 +7,76 @@
|
||||
use crate::types::TextItem;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
/// Return whether text is an explicit page-number expression.
|
||||
///
|
||||
/// This strict form is suitable before layout, where removing one numeric item
|
||||
/// from substantive text such as `Page 42 explains the result` would lose data.
|
||||
pub(crate) fn is_explicit_page_number_expression(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let is_number = |value: &str| {
|
||||
!value.is_empty() && value.chars().all(|character| character.is_ascii_digit())
|
||||
};
|
||||
|
||||
if trimmed.len() <= 4 && is_number(trimmed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if is_number(inner) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
let lowercase = trimmed.to_ascii_lowercase();
|
||||
if let Some(rest) = lowercase.strip_prefix("page") {
|
||||
let words: Vec<&str> = rest.split_whitespace().collect();
|
||||
if words.len() >= 3 && is_number(words[0]) && words[1] == "of" && is_number(words[2]) {
|
||||
return true;
|
||||
}
|
||||
if words.len() >= 2 && words[0] == "of" && is_number(words[1]) {
|
||||
return true;
|
||||
}
|
||||
return match words.as_slice() {
|
||||
[] | ["of"] => true,
|
||||
[number] => is_number(number),
|
||||
["of", total] => is_number(total),
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
};
|
||||
}
|
||||
|
||||
let words: Vec<&str> = lowercase.split_whitespace().collect();
|
||||
match words.as_slice() {
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return whether a completed Markdown line looks like a page number or a
|
||||
/// labeled running header.
|
||||
///
|
||||
/// At this stage the complete line and surrounding breaks are available, so a
|
||||
/// leading `Page N` remains compatible with the existing header cleanup even
|
||||
/// when the PDF appends a chapter or document title.
|
||||
pub(crate) fn is_page_number_line(text: &str) -> bool {
|
||||
if is_explicit_page_number_expression(text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let lowercase = text.trim().to_ascii_lowercase();
|
||||
lowercase.strip_prefix("page").is_some_and(|rest| {
|
||||
rest.trim_start()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
/// Check if a character is CJK (Chinese, Japanese, Korean).
|
||||
/// CJK languages don't use spaces between words, so word-boundary
|
||||
/// heuristics should not apply when CJK characters are involved.
|
||||
|
||||
+113
-10
@@ -148,14 +148,17 @@ impl TextLine {
|
||||
self.text_with_formatting(false, false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
/// Get text with optional bold/italic/decorative markdown formatting.
|
||||
///
|
||||
/// `format_decorations` enables both geometrically detected source
|
||||
/// decorations: underline (`<u>`) and strikeout (`<s>`).
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_underline: bool,
|
||||
format_decorations: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
if !format_bold && !format_italic && !format_decorations {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -165,6 +168,7 @@ impl TextLine {
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
let mut current_strikeout = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -190,13 +194,16 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
// Check for style changes. Source decorations are exclusive:
|
||||
// `<u>`/`<s>` content stays free of `**`/`*` markers — consumers
|
||||
// (and the eval harnesses this feeds) match tag content literally,
|
||||
// and mixed nesting breaks that. A struck-and-underlined item is
|
||||
// emitted as struck text because deletion is the stronger semantic
|
||||
// distinction in redline documents.
|
||||
let item_strikeout = format_decorations && item.is_strikeout;
|
||||
let item_underline = format_decorations && item.is_underline && !item_strikeout;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline && !item_strikeout;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline && !item_strikeout;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -211,6 +218,10 @@ impl TextLine {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
if current_strikeout && !item_strikeout {
|
||||
result.push_str("</s>");
|
||||
current_strikeout = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -222,6 +233,10 @@ impl TextLine {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_strikeout && !current_strikeout {
|
||||
result.push_str("<s>");
|
||||
current_strikeout = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -244,6 +259,9 @@ impl TextLine {
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
if current_strikeout {
|
||||
result.push_str("</s>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
@@ -309,3 +327,88 @@ impl TextLine {
|
||||
|| space_already_exists)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod formatting_tests {
|
||||
use super::{ItemType, TextItem, TextLine};
|
||||
|
||||
fn item(text: &str, x: f32, width: f32, strikeout: bool) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y: 100.0,
|
||||
width,
|
||||
height: 12.0,
|
||||
font: "F1".to_string(),
|
||||
font_size: 12.0,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: strikeout,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn line(items: Vec<TextItem>) -> TextLine {
|
||||
TextLine {
|
||||
items,
|
||||
y: 100.0,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_emits_semantic_strikeout() {
|
||||
let line = line(vec![item("deleted", 10.0, 42.0, true)]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_closes_strikeout_before_live_text() {
|
||||
let line = line(vec![
|
||||
item("keep", 10.0, 24.0, false),
|
||||
item("remove", 40.0, 42.0, true),
|
||||
item("keep", 88.0, 24.0, false),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"keep <s>remove</s> keep"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn formatting_coalesces_adjacent_struck_items() {
|
||||
let line = line(vec![
|
||||
item("deleted", 10.0, 42.0, true),
|
||||
item("words", 58.0, 30.0, true),
|
||||
]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted words</s>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strikeout_takes_precedence_over_other_styles() {
|
||||
let mut decorated = item("deleted", 10.0, 42.0, true);
|
||||
decorated.is_bold = true;
|
||||
decorated.is_italic = true;
|
||||
decorated.is_underline = true;
|
||||
let line = line(vec![decorated]);
|
||||
|
||||
assert_eq!(
|
||||
line.text_with_formatting(true, true, true),
|
||||
"<s>deleted</s>"
|
||||
);
|
||||
assert_eq!(line.text(), "deleted");
|
||||
}
|
||||
}
|
||||
|
||||
+276
-4
@@ -8,12 +8,13 @@ use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
process_pdf_mem, process_pdf_mem_with_options, process_pdf_with_options, to_markdown,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
@@ -40,10 +41,11 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [{media_box}] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>"
|
||||
),
|
||||
);
|
||||
|
||||
let content = "BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
@@ -79,6 +81,186 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_recurring_contextual_folio_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R 7 0 R 9 0 R] /Count 4 >>",
|
||||
);
|
||||
for page_index in 0..4 {
|
||||
let page_id = 3 + page_index * 2;
|
||||
let content_id = page_id + 1;
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
page_id,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 11 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
),
|
||||
);
|
||||
let page_number = page_index + 1;
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 1 0 0 1 25 30 Tm ({page_number}) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Body page {page_number}) Tj ET"
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
content_id,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
}
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
11,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_malformed_unselected_page() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R] /Count 2 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
let content = "BT /F1 12 Tf 1 0 0 1 25 30 Tm (1) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Selected page text) Tj 0 -16 Td (More selected text) Tj 0 -16 Td (Still selected text) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 6 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Length 3 >>\nstream\nBI \nendstream",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
make_text_pdf(
|
||||
"BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET",
|
||||
"0 0 612 792",
|
||||
)
|
||||
}
|
||||
|
||||
fn make_digit_run_repro_pdf() -> Vec<u8> {
|
||||
let content = r#"BT
|
||||
/F1 12 Tf
|
||||
1 0 0 1 72 780 Tm (A\)) Tj
|
||||
1 0 0 1 96 780 Tm (The) Tj
|
||||
1 0 0 1 126 780 Tm (total) Tj
|
||||
1 0 0 1 166 780 Tm (of) Tj
|
||||
1 0 0 1 186 780 Tm (730) Tj
|
||||
1 0 0 1 220 780 Tm (seats) Tj
|
||||
1 0 0 1 262 780 Tm (was) Tj
|
||||
1 0 0 1 296 780 Tm (approved.) Tj
|
||||
1 0 0 1 72 755 Tm (B\)) Tj
|
||||
1 0 0 1 96 755 Tm (let) Tj
|
||||
1 0 0 1 120 755 Tm (log) Tj
|
||||
1 0 0 1 150 755 Tm (2) Tj
|
||||
1 0 0 1 164 755 Tm (=) Tj
|
||||
1 0 0 1 180 755 Tm (a) Tj
|
||||
1 0 0 1 72 720 Tm (C\) Control: The total of 730 seats was approved. let log 2 = a) Tj
|
||||
ET"#;
|
||||
make_text_pdf(content, "0 0 595 842")
|
||||
}
|
||||
|
||||
fn truncate_eof_marker(mut pdf: Vec<u8>) -> Vec<u8> {
|
||||
assert!(pdf.ends_with(b"%%EOF"));
|
||||
pdf.pop();
|
||||
@@ -330,6 +512,21 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
assert_eq!(lines[0].text(), "First Second Third");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_digit_only_text_runs_are_preserved_in_markdown() {
|
||||
let pdf = make_digit_run_repro_pdf();
|
||||
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract positioned text");
|
||||
assert!(items.iter().any(|item| item.text == "730"));
|
||||
assert!(items.iter().any(|item| item.text == "2"));
|
||||
|
||||
let result = process_pdf_mem(&pdf).expect("convert PDF to markdown");
|
||||
assert_eq!(
|
||||
result.markdown.expect("markdown output").trim(),
|
||||
"A) The total of 730 seats was approved.\nB) let log 2 = a\nC) Control: The total of 730 seats was approved. let log 2 = a"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MarkdownOptions Tests
|
||||
// ============================================================================
|
||||
@@ -547,6 +744,30 @@ fn test_markdown_from_items_page_breaks() {
|
||||
assert!(md.contains("Content on second page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_page_count_overload_includes_trailing_blank_pages_in_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
items.push(make_text_item(value, 25.0, 30.0, 12.0, page));
|
||||
items.push(make_text_item(
|
||||
"Company report footer",
|
||||
41.0,
|
||||
30.0,
|
||||
12.0,
|
||||
page,
|
||||
));
|
||||
}
|
||||
let options = MarkdownOptions {
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = to_markdown_from_items_with_rects_and_page_count(items, options, &[], 20);
|
||||
|
||||
assert!(md.contains("1 Company report footer"));
|
||||
assert!(md.contains("4 Company report footer"));
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Markdown From Lines Tests
|
||||
// ============================================================================
|
||||
@@ -2927,6 +3148,57 @@ fn test_extract_pages_markdown_basic() {
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = extract_pages_markdown_mem(&pdf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 4);
|
||||
for (index, page) in result.pages.iter().enumerate() {
|
||||
assert!(page.markdown.contains("Company report footer"));
|
||||
assert!(
|
||||
!page
|
||||
.markdown
|
||||
.contains(&format!("{} Company report footer", index + 1)),
|
||||
"recurring contextual folio survived on page {}: {}",
|
||||
index + 1,
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_process_pdf_page_filter_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
|
||||
assert!(markdown.contains("Company report footer"));
|
||||
assert!(!markdown.contains("1 Company report footer"), "{markdown}");
|
||||
assert!(markdown.contains("Body page 1"));
|
||||
assert!(!markdown.contains("Body page 2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_selected_page_ignores_context_only_extraction_failure() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
let pages = extract_pages_markdown_mem(&pdf, Some(&[0])).unwrap();
|
||||
assert_eq!(pages.pages.len(), 1);
|
||||
assert!(pages.pages[0].markdown.contains("Selected page text"));
|
||||
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
assert!(markdown.contains("Selected page text"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_requested_page_extraction_failure_remains_fatal() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
assert!(extract_pages_markdown_mem(&pdf, Some(&[1])).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
@@ -56,7 +56,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
Form **4070** Employee’s Report (Rev. July 1996)
|
||||
|
||||
## of Tips to EmployerOMB No. 1545-0065
|
||||
|
||||
@@ -81,4 +81,3 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ J. W. Tukey. A device with two stable positions, such as a relay or a flip-flop
|
||||
|
||||
log₂*M* = log₁₀*M*= log₁₀2 = 3:32 log₁₀*M*;
|
||||
|
||||
1 Nyquist, H., “Certain Factors Affecting Telegraph Speed,” *Bell System Technical Journal,* April 1924, p. 324; “Certain Topics in Telegraph Transmission Theory,” *A.I.E.E. Trans.,* v. 47, April 1928, p. 617. Hartley, R. V. L., “Transmission of Information,” *Bell System Technical Journal,* July 1928, p. 535.
|
||||
1 Nyquist, H., “Certain Factors Affecting Telegraph Speed,” *Bell System Technical Journal,* April 1924, p. 324; “Certain Topics in Telegraph Transmission Theory,” *A.I.E.E. Trans.,* v. 47, April 1928, p. 617. 2 Hartley, R. V. L., “Transmission of Information,” *Bell System Technical Journal,* July 1928, p. 535.
|
||||
|
||||
INFORMATION SOURCE TRANSMITTER RECEIVER DESTINATION
|
||||
|
||||
|
||||
Generated
+2
-2
@@ -605,9 +605,9 @@ checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.41.0"
|
||||
version = "0.42.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags 2.13.1",
|
||||
|
||||
Reference in New Issue
Block a user