Compare commits
119
Commits
@@ -20,17 +20,7 @@ jobs:
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/bin/
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
target/
|
||||
key: ${{ runner.os }}-cargo-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-
|
||||
uses: Swatinem/rust-cache@v2
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose
|
||||
@@ -61,17 +51,9 @@ jobs:
|
||||
components: clippy
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/bin/
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
target/
|
||||
key: ${{ runner.os }}-cargo-clippy-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-clippy-
|
||||
key: clippy
|
||||
|
||||
- name: Run clippy
|
||||
run: cargo clippy -- -D warnings
|
||||
@@ -89,17 +71,9 @@ jobs:
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/bin/
|
||||
~/.cargo/registry/index/
|
||||
~/.cargo/registry/cache/
|
||||
~/.cargo/git/db/
|
||||
target/
|
||||
key: ${{ runner.os }}-cargo-build-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-cargo-build-
|
||||
key: build
|
||||
|
||||
- name: Build
|
||||
run: cargo build --release --verbose
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
name: Publish Rust crate
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['Cargo.toml']
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
|
||||
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
|
||||
"https://crates.io/api/v1/crates/pdf-inspector/$NEW_VERSION")
|
||||
|
||||
case "$HTTP_STATUS" in
|
||||
200)
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
echo "pdf-inspector v$NEW_VERSION is already published"
|
||||
;;
|
||||
404)
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
*)
|
||||
cat /tmp/crate-version.json
|
||||
echo "Unexpected crates.io response: $HTTP_STATUS" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
publish:
|
||||
name: Publish to crates.io
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
environment: crates-io
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
env:
|
||||
CARGO_REGISTRY_TOKEN: ${{ steps.auth.outputs.token }}
|
||||
@@ -2,14 +2,41 @@ name: Publish npm package
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ['v*']
|
||||
branches: [main]
|
||||
paths: ['napi/package.json']
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(node -p "require('./napi/package.json').version")
|
||||
OLD_VERSION=$(git show HEAD~1:napi/package.json | node -p "JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')).version")
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
build:
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true'
|
||||
name: Build ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
@@ -19,6 +46,8 @@ jobs:
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
@@ -68,7 +97,7 @@ jobs:
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: build
|
||||
needs: [check-version, build]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -35,3 +35,8 @@ scripts/
|
||||
# Test output
|
||||
test_output/
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast PDF text extraction to structured Markdown. CLI binary: `pdf2md`. Detection binary: `detect-pdf`.
|
||||
|
||||
## Build & Test
|
||||
|
||||
```bash
|
||||
cargo fmt # format
|
||||
cargo clippy -- -D warnings # lint (enforced, zero warnings)
|
||||
cargo test # unit + integration tests (267+ unit, 73+ integration)
|
||||
cargo build --release # release binary for benchmarks
|
||||
```
|
||||
|
||||
All three must pass before committing.
|
||||
|
||||
## Binaries
|
||||
|
||||
- `pdf2md` — extract PDF → Markdown. Supports `--json` for structured output.
|
||||
- `detect-pdf` — classify PDF type (TextBased/Scanned/Mixed/ImageBased). Supports `--analyze --json`.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
src/
|
||||
lib.rs – public API, process_pdf_with_options, encoding issue detection
|
||||
detector.rs – PDF type classification, tiled-scan detection, page sampling
|
||||
types.rs – TextItem, TextLine, PdfRect, PdfLine
|
||||
tounicode.rs – CMap/ToUnicode parsing, CID decoding
|
||||
text_utils.rs – CJK/RTL handling, Otsu threshold, ligature expansion, NFKC
|
||||
extractor/
|
||||
mod.rs – top-level extraction orchestrator
|
||||
content_stream.rs – PDF operator state machine (Tj/TJ/Td/Tm/q/Q)
|
||||
fonts.rs – font width/encoding, CMapDecisionCache, TrueType cmap fallback
|
||||
layout.rs – column detection (histogram), newspaper/tabular classification,
|
||||
spanning-line pre-masking, sidebar detection
|
||||
tables/
|
||||
detect_rects.rs – rect-based table detection (union-find clustering)
|
||||
detect_heuristic.rs – heuristic table detection (gap-histogram, body-font tables)
|
||||
detect_lines.rs – line-based table detection (H/V line grids)
|
||||
grid.rs – column/row boundaries, cell assignment
|
||||
format.rs – table→Markdown formatting, continuation row merging
|
||||
markdown/
|
||||
convert.rs – core line→Markdown loop, struct-tree role support
|
||||
analysis.rs – font stats, heading tiers, paragraph thresholds
|
||||
classify.rs – line classification (header, list, code, caption)
|
||||
preprocess.rs – drop cap merging, heading line merging
|
||||
postprocess.rs – dot leaders, hyphenation, page numbers, URL formatting
|
||||
```
|
||||
|
||||
## Key design decisions
|
||||
|
||||
- **Primary audience is AI agents.** Output optimized for token efficiency and semantic quality, not visual formatting. No cosmetic padding.
|
||||
- **Three table detection strategies** run in priority order: rect-based → line-based → heuristic. First valid result wins.
|
||||
- **Column detection** uses horizontal projection histograms with valley detection. Multi-item spanning lines (titles, headers) are pre-masked using column-aware thresholds before column assignment.
|
||||
- **Newspaper vs tabular** classification determines reading order: newspaper reads columns sequentially, tabular Y-interleaves them.
|
||||
- **Tiled-scan detection** catches scanned PDFs with JBIG2/strip images where no single tile exceeds the template threshold but aggregate area does (≥2M pixels).
|
||||
- **Garbage text upgrade** reclassifies Mixed PDFs as Scanned when extracted text is <50% alphanumeric.
|
||||
- **Tagged PDF support** uses structure tree roles (H1-H6, P, L, Code, BlockQuote) when available, falling back to font-size heuristics.
|
||||
|
||||
## Testing
|
||||
|
||||
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
|
||||
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 179+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
|
||||
## Debugging
|
||||
|
||||
```bash
|
||||
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf
|
||||
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf
|
||||
RUST_LOG=pdf_inspector::detector=debug cargo run --release --bin detect-pdf -- file.pdf
|
||||
```
|
||||
|
||||
## Conventions
|
||||
|
||||
- Clippy: use `is_some_and(...)` not `map_or(false, ...)`
|
||||
- lopdf quirk: `ParseError` is private — match by string for `InvalidFileHeader`
|
||||
- Column limit for tables: 25 (wide statistical tables)
|
||||
- `propagate_merged_cells` skipped for >10 columns (spanning rects = background fills)
|
||||
@@ -61,7 +61,8 @@ src/
|
||||
|
||||
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
|
||||
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 179+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 187+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
- **Semantic quality**: run `bench.py score` in `pdf-evals` for the semantic verdict (TEDS + MHS + reading order + char/word + list preservation, composited). Character-level diff alone misclassifies structural improvements (e.g., column-detection rewrites) as regressions — `score` is the tie-breaker. See `pdf-evals/CLAUDE.md` "Semantic scoring".
|
||||
|
||||
## Debugging
|
||||
|
||||
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
version = "0.1.4"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
@@ -17,7 +17,7 @@ crate-type = ["lib", "cdylib"]
|
||||
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "052674053814a9f4897af94f0b8e46a545c9b329", features = ["rayon"] }
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -1,5 +1,9 @@
|
||||
# pdf-inspector
|
||||
|
||||
[](https://crates.io/crates/pdf-inspector)
|
||||
[](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
@@ -16,6 +20,23 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | 0.78 | 0.87 | 0.59 | 0.57 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus.
|
||||
|
||||
**Where we do well:** Speed (fastest of all engines), reading order, table detection vs other direct-text tools.
|
||||
|
||||
**Where we lag:** Heading detection trails opendataloader — many PDFs use bold text at body font size for headings, or headings that are only slightly larger than body text. Table detection trails OCR-based engines that can see visual table structure.
|
||||
|
||||
## Quick start
|
||||
|
||||
### Python
|
||||
@@ -38,12 +59,12 @@ print(result.markdown) # Markdown string or None
|
||||
### Node.js
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-js
|
||||
npm install @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
```javascript
|
||||
import { readFileSync } from 'fs';
|
||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector-js';
|
||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector';
|
||||
|
||||
const result = processPdf(readFileSync('document.pdf'));
|
||||
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
||||
@@ -54,9 +75,17 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
|
||||
```bash
|
||||
cargo add pdf-inspector
|
||||
```
|
||||
|
||||
Or add it manually:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
pdf-inspector = "0.1"
|
||||
```
|
||||
|
||||
```rust
|
||||
@@ -74,29 +103,37 @@ if let Some(markdown) = &result.markdown {
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
# Install the CLI tools
|
||||
cargo install pdf-inspector
|
||||
|
||||
# Convert PDF to Markdown
|
||||
cargo run --bin pdf2md -- document.pdf
|
||||
pdf2md document.pdf
|
||||
|
||||
# JSON output (for piping)
|
||||
cargo run --bin pdf2md -- document.pdf --json
|
||||
pdf2md document.pdf --json
|
||||
|
||||
# Positioned TextItem JSON, including is_underline metadata
|
||||
pdf2md document.pdf --items-json
|
||||
|
||||
# Raw markdown only (no headers)
|
||||
cargo run --bin pdf2md -- document.pdf --raw
|
||||
pdf2md document.pdf --raw
|
||||
|
||||
# Insert page break markers (<!-- Page N -->)
|
||||
cargo run --bin pdf2md -- document.pdf --pages
|
||||
pdf2md document.pdf --pages
|
||||
|
||||
# Process only specific pages
|
||||
cargo run --bin pdf2md -- document.pdf --select-pages 1,3,5-10
|
||||
pdf2md document.pdf --select-pages 1,3,5-10
|
||||
|
||||
# Detection only (no extraction)
|
||||
cargo run --bin detect-pdf -- document.pdf
|
||||
cargo run --bin detect-pdf -- document.pdf --json
|
||||
detect-pdf document.pdf
|
||||
detect-pdf document.pdf --json
|
||||
|
||||
# Detection + layout analysis (tables, columns)
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze --json
|
||||
detect-pdf document.pdf --analyze --json
|
||||
```
|
||||
|
||||
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
@@ -206,4 +243,4 @@ See [docs/debugging.md](docs/debugging.md) for `RUST_LOG` environment variable u
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
[MIT](LICENSE)
|
||||
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
# Security Policy
|
||||
|
||||
## Reporting a Vulnerability
|
||||
|
||||
If you believe you've found a security vulnerability in pdf-inspector, please
|
||||
report it privately so we can fix it before public disclosure.
|
||||
|
||||
**Preferred:** Email **help@firecrawl.dev** with:
|
||||
|
||||
- A description of the issue and its impact
|
||||
- Steps to reproduce (a minimal PDF or input that triggers the bug is ideal)
|
||||
- The version or commit hash of pdf-inspector you tested against
|
||||
|
||||
**Alternative:** Use GitHub's private vulnerability reporting under the
|
||||
[Security tab](https://github.com/firecrawl/pdf-inspector/security/advisories/new).
|
||||
|
||||
We'll acknowledge your report in a timely manner and keep you updated on
|
||||
remediation progress. Please do not open a public GitHub issue for security
|
||||
bugs.
|
||||
|
||||
## Scope
|
||||
|
||||
In scope:
|
||||
- Memory-safety issues (panics, OOB reads, UB) reachable from a crafted PDF
|
||||
- Denial-of-service vectors (unbounded allocation, infinite loops) on
|
||||
reasonably-sized inputs
|
||||
- Bugs in the `pdf2md` / `detect-pdf` binaries or the `pdf-inspector` crate
|
||||
that affect downstream consumers
|
||||
|
||||
Out of scope:
|
||||
- Bugs in upstream dependencies (`lopdf`, etc.) — please report those upstream
|
||||
- Extraction quality issues (wrong text, missing tables) — open a regular
|
||||
GitHub issue instead
|
||||
@@ -0,0 +1,21 @@
|
||||
# Publishing
|
||||
|
||||
The Rust crate is published to [crates.io](https://crates.io/crates/pdf-inspector) with trusted publishing from GitHub Actions. The first release was published manually; future releases publish from `.github/workflows/publish-crate.yml` when a `Cargo.toml` version change lands on `main`.
|
||||
|
||||
## crates.io Trusted Publisher
|
||||
|
||||
Configure the trusted publisher for the `pdf-inspector` crate with:
|
||||
|
||||
- Repository: `firecrawl/pdf-inspector`
|
||||
- Workflow: `publish-crate.yml`
|
||||
- Environment: `crates-io`
|
||||
|
||||
The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC token for a short-lived crates.io token, then passes it to `cargo publish`.
|
||||
|
||||
## Release Steps
|
||||
|
||||
1. Update `version` in `Cargo.toml`.
|
||||
2. Merge the version bump to `main`.
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
@@ -42,6 +42,14 @@ text = pdf_inspector.extract_text("document.pdf")
|
||||
items = pdf_inspector.extract_text_with_positions("document.pdf")
|
||||
for item in items[:5]:
|
||||
print(f"'{item.text}' at ({item.x:.0f}, {item.y:.0f}) size={item.font_size}")
|
||||
|
||||
# Per-page markdown (one Markdown string per page, plus layout metadata)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf")
|
||||
for page in result.pages:
|
||||
print(f"Page {page.page}: {len(page.markdown)} chars, needs_ocr={page.needs_ocr}")
|
||||
|
||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
```
|
||||
|
||||
## API reference
|
||||
@@ -60,6 +68,8 @@ for item in items[:5]:
|
||||
| `extract_text_with_positions_bytes(data, pages=None)` | Text with positions from bytes |
|
||||
| `extract_text_in_regions(path, page_regions)` | Extract text in bounding-box regions |
|
||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
|
||||
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
|
||||
|
||||
## Types
|
||||
|
||||
@@ -72,3 +82,7 @@ for item in items[:5]:
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
|
||||
**`PageMarkdown` fields:** `page` (0-indexed), `markdown`, `needs_ocr`
|
||||
|
||||
**`PagesExtractionResult` fields:** `pages` (list of PageMarkdown), `pages_with_tables` (1-indexed), `pages_with_columns` (1-indexed), `pages_needing_ocr` (1-indexed), `is_complex`
|
||||
|
||||
@@ -79,6 +79,27 @@ let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
|
||||
Extract per-page Markdown (one string per page, plus document-wide layout
|
||||
metadata):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::extract_pages_markdown;
|
||||
|
||||
// Pass `None` for every page in document order, or a slice of 0-indexed
|
||||
// pages to restrict the output (caller-supplied order is preserved).
|
||||
let result = extract_pages_markdown("document.pdf", None)?;
|
||||
|
||||
for page in &result.pages {
|
||||
if page.needs_ocr {
|
||||
// Route this page to OCR
|
||||
} else {
|
||||
println!("Page {}: {}", page.page, page.markdown);
|
||||
}
|
||||
}
|
||||
|
||||
println!("Complex layout? {}", result.is_complex);
|
||||
```
|
||||
|
||||
## Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
@@ -102,6 +123,8 @@ let result = process_pdf_mem(&bytes)?;
|
||||
| `to_markdown(text, options)` | Convert plain text to Markdown |
|
||||
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
|
||||
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
@@ -119,4 +142,6 @@ Low-level detection functions are also available via the `detector` module (`det
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
|
||||
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
|
||||
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
|
||||
|
||||
Generated
+5
-22
@@ -129,12 +129,6 @@ version = "3.20.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb"
|
||||
|
||||
[[package]]
|
||||
name = "bytecount"
|
||||
version = "0.6.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "175812e0be2bccb6abe50bb8d566126198344f707e304f45c648fd8f2cc0365e"
|
||||
|
||||
[[package]]
|
||||
name = "cbc"
|
||||
version = "0.1.2"
|
||||
@@ -678,8 +672,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.40.0"
|
||||
source = "git+https://github.com/J-F-Liu/lopdf?rev=052674053814a9f4897af94f0b8e46a545c9b329#052674053814a9f4897af94f0b8e46a545c9b329"
|
||||
version = "0.41.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -695,7 +690,6 @@ dependencies = [
|
||||
"log",
|
||||
"md-5",
|
||||
"nom",
|
||||
"nom_locate",
|
||||
"rand",
|
||||
"rangemap",
|
||||
"rayon",
|
||||
@@ -807,17 +801,6 @@ dependencies = [
|
||||
"memchr",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "nom_locate"
|
||||
version = "5.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b577e2d69827c4740cba2b52efaad1c4cc7c73042860b199710b3575c68438d"
|
||||
dependencies = [
|
||||
"bytecount",
|
||||
"memchr",
|
||||
"nom",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-conv"
|
||||
version = "0.2.1"
|
||||
@@ -847,7 +830,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
version = "0.1.4"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"log",
|
||||
@@ -862,7 +845,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.0"
|
||||
version = "0.2.2"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.0"
|
||||
version = "0.2.2"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
+7
-6
@@ -1,4 +1,4 @@
|
||||
# firecrawl-pdf-inspector
|
||||
# PDF Inspector
|
||||
|
||||
Fast PDF classification and region-based text extraction for Node.js/Bun. Native Rust performance via [napi-rs](https://napi.rs).
|
||||
|
||||
@@ -7,9 +7,9 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
## Install
|
||||
|
||||
```bash
|
||||
npm install firecrawl-pdf-inspector
|
||||
npm install @firecrawl/pdf-inspector
|
||||
# or
|
||||
bun add firecrawl-pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolchain needed.
|
||||
@@ -21,7 +21,7 @@ Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolch
|
||||
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
|
||||
|
||||
```typescript
|
||||
import { classifyPdf } from 'firecrawl-pdf-inspector'
|
||||
import { classifyPdf } from '@firecrawl/pdf-inspector'
|
||||
import { readFileSync } from 'fs'
|
||||
|
||||
const pdf = readFileSync('document.pdf')
|
||||
@@ -37,10 +37,10 @@ console.log(result.confidence) // 0.875
|
||||
|
||||
Extract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR.
|
||||
|
||||
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues).
|
||||
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues). When the cause is a suspected garbled text layer, `ocrReason` is set to `"suspected_garbled_text"`.
|
||||
|
||||
```typescript
|
||||
import { extractTextInRegions } from 'firecrawl-pdf-inspector'
|
||||
import { extractTextInRegions } from '@firecrawl/pdf-inspector'
|
||||
|
||||
const result = extractTextInRegions(pdf, [
|
||||
{
|
||||
@@ -84,6 +84,7 @@ interface PageRegionTexts {
|
||||
interface RegionText {
|
||||
text: string
|
||||
needsOcr: boolean // true when text is unreliable
|
||||
ocrReason?: string // "suspected_garbled_text" when known
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
Executable
+131
@@ -0,0 +1,131 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
import { readFileSync, writeFileSync } from "fs";
|
||||
import { createRequire } from "module";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const { version } = require("../package.json");
|
||||
|
||||
const HELP = `pdf-inspector v${version} — Fast PDF text extraction to Markdown
|
||||
|
||||
Usage:
|
||||
pdf-inspector <file> Extract markdown (default)
|
||||
pdf-inspector detect <file> Classify PDF type
|
||||
|
||||
Options:
|
||||
--json Output as JSON
|
||||
--pages <pages> Comma-separated page numbers (e.g. 1,3,5)
|
||||
-o, --output <file> Write output to file instead of stdout
|
||||
-h, --help Show this help
|
||||
-v, --version Show version
|
||||
|
||||
Examples:
|
||||
pdf-inspector document.pdf
|
||||
pdf-inspector document.pdf --json
|
||||
pdf-inspector document.pdf --pages 1,2,3
|
||||
pdf-inspector detect document.pdf --json
|
||||
cat document.pdf | pdf-inspector -`;
|
||||
|
||||
function die(msg) {
|
||||
process.stderr.write(`error: ${msg}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
function parseArgs(argv) {
|
||||
const opts = { json: false, pages: null, output: null, file: null, command: "extract" };
|
||||
let i = 0;
|
||||
|
||||
// Check for subcommand
|
||||
if (argv[0] === "detect") {
|
||||
opts.command = "detect";
|
||||
i = 1;
|
||||
}
|
||||
|
||||
while (i < argv.length) {
|
||||
const arg = argv[i];
|
||||
if (arg === "-h" || arg === "--help") {
|
||||
process.stdout.write(HELP + "\n");
|
||||
process.exit(0);
|
||||
} else if (arg === "-v" || arg === "--version") {
|
||||
process.stdout.write(`${version}\n`);
|
||||
process.exit(0);
|
||||
} else if (arg === "--json") {
|
||||
opts.json = true;
|
||||
} else if (arg === "--pages") {
|
||||
i++;
|
||||
if (!argv[i]) die("--pages requires a value (e.g. 1,3,5)");
|
||||
opts.pages = argv[i].split(",").map((p) => {
|
||||
const n = parseInt(p.trim(), 10);
|
||||
if (Number.isNaN(n) || n < 1) die(`invalid page number: ${p}`);
|
||||
return n;
|
||||
});
|
||||
} else if (arg === "-o" || arg === "--output") {
|
||||
i++;
|
||||
if (!argv[i]) die("-o requires a filename");
|
||||
opts.output = argv[i];
|
||||
} else if (arg === "-" || !arg.startsWith("-")) {
|
||||
if (opts.file) die(`unexpected argument: ${arg}`);
|
||||
opts.file = arg;
|
||||
} else {
|
||||
die(`unknown option: ${arg}`);
|
||||
}
|
||||
i++;
|
||||
}
|
||||
|
||||
return opts;
|
||||
}
|
||||
|
||||
function readInput(file) {
|
||||
if (file === "-") {
|
||||
return readFileSync(0); // stdin fd
|
||||
}
|
||||
try {
|
||||
return readFileSync(file);
|
||||
} catch (err) {
|
||||
if (err.code === "ENOENT") die(`file not found: ${file}`);
|
||||
die(err.message);
|
||||
}
|
||||
}
|
||||
|
||||
function output(text, outputPath) {
|
||||
if (outputPath) {
|
||||
writeFileSync(outputPath, text);
|
||||
} else {
|
||||
process.stdout.write(text);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- main ----
|
||||
|
||||
const opts = parseArgs(process.argv.slice(2));
|
||||
|
||||
if (!opts.file) {
|
||||
// Check if stdin is piped
|
||||
if (process.stdin.isTTY !== false) {
|
||||
process.stderr.write(HELP + "\n");
|
||||
process.exit(1);
|
||||
}
|
||||
opts.file = "-";
|
||||
}
|
||||
|
||||
const { processPdf, classifyPdf } = await import("../index.js");
|
||||
const buffer = readInput(opts.file);
|
||||
|
||||
if (opts.command === "detect") {
|
||||
const result = classifyPdf(buffer);
|
||||
if (opts.json) {
|
||||
output(JSON.stringify(result, null, 2) + "\n", opts.output);
|
||||
} else {
|
||||
const ocr = result.pagesNeedingOcr.length > 0
|
||||
? `, ${result.pagesNeedingOcr.length} pages need OCR`
|
||||
: "";
|
||||
output(`${result.pdfType} (${result.pageCount} pages, confidence: ${result.confidence.toFixed(2)}${ocr})\n`, opts.output);
|
||||
}
|
||||
} else {
|
||||
const result = processPdf(buffer, opts.pages ?? undefined);
|
||||
if (opts.json) {
|
||||
output(JSON.stringify(result, null, 2) + "\n", opts.output);
|
||||
} else {
|
||||
output((result.markdown ?? "") + "\n", opts.output);
|
||||
}
|
||||
}
|
||||
+8
-3
@@ -1,9 +1,12 @@
|
||||
{
|
||||
"name": "firecrawl-pdf-inspector",
|
||||
"version": "0.3.2",
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.10.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
"bin": {
|
||||
"pdf-inspector": "bin/pdf-inspector.mjs"
|
||||
},
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"pdf",
|
||||
@@ -20,6 +23,7 @@
|
||||
"index.js",
|
||||
"index.d.ts",
|
||||
"*.node",
|
||||
"bin/",
|
||||
"README.md"
|
||||
],
|
||||
"repository": {
|
||||
@@ -34,7 +38,8 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"aarch64-apple-darwin"
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
],
|
||||
"package": {
|
||||
"name": "@firecrawl/pdf-inspector-js"
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
import { readFileSync } from "node:fs";
|
||||
import { createRequire } from "node:module";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const { detectVectorGridInRegion } = require("./index.js");
|
||||
|
||||
const pdfPath =
|
||||
process.argv[2] ?? "/tmp/pdf_inspector_indent_fixtures/cis_edge_benchmark.pdf";
|
||||
const pdf = readFileSync(pdfPath);
|
||||
const dpi = Number(process.argv[3] ?? 200);
|
||||
|
||||
const crops = [
|
||||
{ pageIdx: 29, box: [0, 0, 612, 792], label: "page30-full" },
|
||||
{ pageIdx: 16, box: [0, 0, 612, 792], label: "page17-full" },
|
||||
{ pageIdx: 23, box: [0, 0, 612, 792], label: "page24-full" },
|
||||
];
|
||||
|
||||
for (const { pageIdx, box, label } of crops) {
|
||||
const result = detectVectorGridInRegion(pdf, pageIdx, box, dpi);
|
||||
if (!result) {
|
||||
console.log(`${label}: null`);
|
||||
continue;
|
||||
}
|
||||
const rows = result.structureTokens.filter((token) => token === "<tr>").length;
|
||||
const cols = rows > 0 ? result.cellBboxes.length / rows : 0;
|
||||
console.log(
|
||||
`${label}: cells=${result.cellBboxes.length} rows=${rows} cols=${cols}`,
|
||||
);
|
||||
}
|
||||
+447
-48
@@ -5,6 +5,28 @@ use napi_derive::napi;
|
||||
use std::collections::HashSet;
|
||||
use std::panic;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Enums
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// PDF document type classification.
|
||||
#[napi(string_enum)]
|
||||
pub enum PdfType {
|
||||
TextBased,
|
||||
Scanned,
|
||||
ImageBased,
|
||||
Mixed,
|
||||
}
|
||||
|
||||
/// Type of a positioned text item.
|
||||
#[napi(string_enum)]
|
||||
pub enum ItemType {
|
||||
Text,
|
||||
Image,
|
||||
Link,
|
||||
FormField,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result types
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -12,12 +34,14 @@ use std::panic;
|
||||
/// Full PDF processing result with markdown and metadata.
|
||||
#[napi(object)]
|
||||
pub struct PdfResult {
|
||||
pub pdf_type: String,
|
||||
pub pdf_type: PdfType,
|
||||
pub markdown: Option<String>,
|
||||
pub page_count: u32,
|
||||
pub processing_time_ms: u32,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
pub title: Option<String>,
|
||||
pub confidence: f64,
|
||||
pub is_complex_layout: bool,
|
||||
@@ -26,10 +50,17 @@ pub struct PdfResult {
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
/// OCR reasons for a single 1-indexed page.
|
||||
#[napi(object)]
|
||||
pub struct PageOcrReasons {
|
||||
pub page: u32,
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[napi(object)]
|
||||
pub struct PdfClassification {
|
||||
pub pdf_type: String,
|
||||
pub pdf_type: PdfType,
|
||||
pub page_count: u32,
|
||||
/// 0-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
@@ -49,7 +80,15 @@ pub struct TextItem {
|
||||
pub page: u32,
|
||||
pub is_bold: bool,
|
||||
pub is_italic: bool,
|
||||
pub item_type: String,
|
||||
/// Underline detected geometrically (drawn rule/thin rect under the
|
||||
/// baseline) — PDFs carry no underline font flag.
|
||||
pub is_underline: bool,
|
||||
/// Strikeout detected geometrically (rule crossing the glyphs at mid
|
||||
/// x-height).
|
||||
pub is_strikeout: bool,
|
||||
pub item_type: ItemType,
|
||||
/// URL for link items, `None` for other types.
|
||||
pub link_url: Option<String>,
|
||||
}
|
||||
|
||||
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
||||
@@ -66,6 +105,8 @@ pub struct RegionText {
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
@@ -75,26 +116,34 @@ pub struct PageRegionTexts {
|
||||
pub regions: Vec<RegionText>,
|
||||
}
|
||||
|
||||
/// Vector-grid detection result compatible with `extractTablesWithStructure*`.
|
||||
#[napi(object)]
|
||||
pub struct VectorGridDetectionJs {
|
||||
pub structure_tokens: Vec<String>,
|
||||
pub cell_bboxes: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn pdf_type_string(t: pdf_inspector::PdfType) -> String {
|
||||
fn convert_pdf_type(t: pdf_inspector::PdfType) -> PdfType {
|
||||
match t {
|
||||
pdf_inspector::PdfType::TextBased => "TextBased".to_string(),
|
||||
pdf_inspector::PdfType::Scanned => "Scanned".to_string(),
|
||||
pdf_inspector::PdfType::ImageBased => "ImageBased".to_string(),
|
||||
pdf_inspector::PdfType::Mixed => "Mixed".to_string(),
|
||||
pdf_inspector::PdfType::TextBased => PdfType::TextBased,
|
||||
pdf_inspector::PdfType::Scanned => PdfType::Scanned,
|
||||
pdf_inspector::PdfType::ImageBased => PdfType::ImageBased,
|
||||
pdf_inspector::PdfType::Mixed => PdfType::Mixed,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
PdfResult {
|
||||
pdf_type: pdf_type_string(r.pdf_type),
|
||||
pdf_type: convert_pdf_type(r.pdf_type),
|
||||
markdown: r.markdown,
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms as u32,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
title: r.title,
|
||||
confidence: r.confidence as f64,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
@@ -104,12 +153,24 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn item_type_string(t: &pdf_inspector::types::ItemType) -> String {
|
||||
fn to_napi_page_ocr_reasons(
|
||||
reasons: Vec<pdf_inspector::PageOcrReasons>,
|
||||
) -> Vec<PageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PageOcrReasons {
|
||||
page: reason.page,
|
||||
reasons: reason.reasons,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => "text".into(),
|
||||
pdf_inspector::types::ItemType::Image => "image".into(),
|
||||
pdf_inspector::types::ItemType::Link(url) => format!("link:{url}"),
|
||||
pdf_inspector::types::ItemType::FormField => "form_field".into(),
|
||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||
pdf_inspector::types::ItemType::Image => (ItemType::Image, None),
|
||||
pdf_inspector::types::ItemType::Link(url) => (ItemType::Link, Some(url.clone())),
|
||||
pdf_inspector::types::ItemType::FormField => (ItemType::FormField, None),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -181,7 +242,7 @@ pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: pdf_type_string(result.pdf_type),
|
||||
pdf_type: convert_pdf_type(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
@@ -222,18 +283,24 @@ pub fn extract_text_with_positions(
|
||||
|
||||
Ok(items
|
||||
.into_iter()
|
||||
.map(|item| TextItem {
|
||||
text: item.text,
|
||||
x: item.x as f64,
|
||||
y: item.y as f64,
|
||||
width: item.width as f64,
|
||||
height: item.height as f64,
|
||||
font: item.font,
|
||||
font_size: item.font_size as f64,
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
item_type: item_type_string(&item.item_type),
|
||||
.map(|item| {
|
||||
let (item_type, link_url) = convert_item_type(&item.item_type);
|
||||
TextItem {
|
||||
text: item.text,
|
||||
x: item.x as f64,
|
||||
y: item.y as f64,
|
||||
width: item.width as f64,
|
||||
height: item.height as f64,
|
||||
font: item.font,
|
||||
font_size: item.font_size as f64,
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type,
|
||||
link_url,
|
||||
}
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
@@ -255,7 +322,341 @@ pub fn extract_text_in_regions(
|
||||
page_regions: Vec<PageRegions>,
|
||||
) -> Result<Vec<PageRegionTexts>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let regions: Vec<(u32, Vec<[f32; 4]>)> = page_regions
|
||||
let regions = parse_page_regions(&page_regions);
|
||||
|
||||
catch_panic("extract_text_in_regions", move || {
|
||||
let results = pdf_inspector::extract_text_in_regions_mem(&bytes, ®ions)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_in_regions"))?;
|
||||
Ok(to_page_region_texts(results))
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract markdown tables within bounding-box regions from a PDF.
|
||||
///
|
||||
/// Like `extractTextInRegions` but runs table detection on items within each
|
||||
/// region and returns markdown pipe-tables instead of flat text.
|
||||
///
|
||||
/// When table structure is detected, `text` contains a markdown pipe-table and
|
||||
/// `needsOcr` is `false`. When no table is found, `text` is empty and
|
||||
/// `needsOcr` is `true` so the caller can fall back to GPU OCR.
|
||||
///
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
#[napi]
|
||||
pub fn extract_tables_in_regions(
|
||||
buffer: Buffer,
|
||||
page_regions: Vec<PageRegions>,
|
||||
) -> Result<Vec<PageRegionTexts>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let regions = parse_page_regions(&page_regions);
|
||||
|
||||
catch_panic("extract_tables_in_regions", move || {
|
||||
let results = pdf_inspector::extract_tables_in_regions_mem(&bytes, ®ions)
|
||||
.map_err(|e| to_napi_err(e, "extract_tables_in_regions"))?;
|
||||
Ok(to_page_region_texts(results))
|
||||
})
|
||||
}
|
||||
|
||||
/// Detect a vector ruled-line / rectangle grid inside one page region.
|
||||
///
|
||||
/// Returns TSR-compatible structure tokens plus crop-pixel cell bboxes, or
|
||||
/// `null` when the region does not contain a valid vector grid.
|
||||
///
|
||||
/// `pageIdx` is 0-indexed. `regionPdfPtBbox` is `[x1,y1,x2,y2]` in PDF
|
||||
/// points with top-left origin. `renderDpi` is the DPI of the crop image that
|
||||
/// will consume the returned cell bboxes.
|
||||
#[napi]
|
||||
pub fn detect_vector_grid_in_region(
|
||||
buffer: Buffer,
|
||||
page_idx: u32,
|
||||
region_pdf_pt_bbox: Vec<f64>,
|
||||
render_dpi: f64,
|
||||
) -> Result<Option<VectorGridDetectionJs>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let region = if region_pdf_pt_bbox.len() == 4 {
|
||||
[
|
||||
region_pdf_pt_bbox[0] as f32,
|
||||
region_pdf_pt_bbox[1] as f32,
|
||||
region_pdf_pt_bbox[2] as f32,
|
||||
region_pdf_pt_bbox[3] as f32,
|
||||
]
|
||||
} else {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
};
|
||||
|
||||
catch_panic("detect_vector_grid_in_region", move || {
|
||||
let result = pdf_inspector::detect_vector_grid_in_region_mem(
|
||||
&bytes,
|
||||
page_idx,
|
||||
region,
|
||||
render_dpi as f32,
|
||||
)
|
||||
.map_err(|e| to_napi_err(e, "detect_vector_grid_in_region"))?;
|
||||
|
||||
Ok(result.map(|r| VectorGridDetectionJs {
|
||||
structure_tokens: r.structure_tokens,
|
||||
cell_bboxes: r
|
||||
.cell_bboxes
|
||||
.into_iter()
|
||||
.map(|bbox| bbox.into_iter().map(|v| v as f64).collect())
|
||||
.collect(),
|
||||
}))
|
||||
})
|
||||
}
|
||||
|
||||
/// One cropped table region plus its raw structure-recovery output, for
|
||||
/// `extractTablesWithStructure`.
|
||||
///
|
||||
/// `structureTokens` and `cellBboxes` are typically produced by an external
|
||||
/// table-structure recognition model (e.g. SLANet on PaddleOCR) running on
|
||||
/// a rendered crop of the page. pdf-inspector uses the structure to lay out
|
||||
/// the cells and pulls the cell text from the native PDF — no OCR involved.
|
||||
#[napi(object)]
|
||||
pub struct TsrTableInputJs {
|
||||
/// 0-indexed page number where the crop was taken from.
|
||||
pub page: u32,
|
||||
/// Crop bbox on the page, `[x1, y1, x2, y2]` in PDF points with
|
||||
/// top-left origin.
|
||||
pub crop_pdf_pt_bbox: Vec<f64>,
|
||||
/// DPI the crop image was rendered at (e.g. `200.0`).
|
||||
pub render_dpi: f64,
|
||||
/// Raw structure tokens emitted by the TSR model, in document order.
|
||||
pub structure_tokens: Vec<String>,
|
||||
/// One bbox per cell (in document order). May be 4-element
|
||||
/// `[x1,y1,x2,y2]` or 8-element 4-corner polygon, in crop image-pixel
|
||||
/// space.
|
||||
pub cell_bboxes: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
/// Extract markdown tables using externally-supplied structure recovery.
|
||||
///
|
||||
/// For each input, pairs structure tokens with cell bboxes (rowspan/colspan
|
||||
/// aware), converts each cell bbox from crop image-pixels into page PDF
|
||||
/// points, pulls the cell's text from the native PDF, and emits a markdown
|
||||
/// pipe-table.
|
||||
///
|
||||
/// Returns one markdown string per input, in input order.
|
||||
#[napi]
|
||||
pub fn extract_tables_with_structure(
|
||||
buffer: Buffer,
|
||||
inputs: Vec<TsrTableInputJs>,
|
||||
) -> Result<Vec<String>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let parsed = parse_tsr_inputs(&inputs);
|
||||
|
||||
catch_panic("extract_tables_with_structure", move || {
|
||||
pdf_inspector::extract_tables_with_structure_mem(&bytes, &parsed)
|
||||
.map_err(|e| to_napi_err(e, "extract_tables_with_structure"))
|
||||
})
|
||||
}
|
||||
|
||||
/// One resolved cell from `extractTablesWithStructureCells`.
|
||||
#[napi(object)]
|
||||
pub struct StructuredCellJs {
|
||||
/// 0-indexed grid row.
|
||||
pub row: u32,
|
||||
/// 0-indexed grid column.
|
||||
pub col: u32,
|
||||
/// 1 for a normal cell.
|
||||
pub rowspan: u32,
|
||||
/// 1 for a normal cell.
|
||||
pub colspan: u32,
|
||||
/// `true` when the cell is a `<th>` or sits inside `<thead>`.
|
||||
pub is_header: bool,
|
||||
/// Text extracted from the native PDF for this cell (may be empty).
|
||||
pub text: String,
|
||||
/// Axis-aligned bbox `[x1, y1, x2, y2]` in page PDF-points, top-left
|
||||
/// origin. Useful for debug overlays or per-cell post-processing.
|
||||
pub page_pt_bbox: Vec<f64>,
|
||||
}
|
||||
|
||||
/// Extract structured cells using externally-supplied structure recovery.
|
||||
///
|
||||
/// Lower-level sibling of [`extractTablesWithStructure`]: instead of
|
||||
/// rendering markdown, returns the resolved cells (row, col, rowspan,
|
||||
/// colspan, isHeader, text, pagePtBbox) so callers can drive their own
|
||||
/// rendering, debug overlays, or per-cell post-processing.
|
||||
///
|
||||
/// Returns one `Array<StructuredCellJs>` per input, in input order.
|
||||
#[napi]
|
||||
pub fn extract_tables_with_structure_cells(
|
||||
buffer: Buffer,
|
||||
inputs: Vec<TsrTableInputJs>,
|
||||
) -> Result<Vec<Vec<StructuredCellJs>>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let parsed = parse_tsr_inputs(&inputs);
|
||||
|
||||
catch_panic("extract_tables_with_structure_cells", move || {
|
||||
let result = pdf_inspector::extract_tables_with_structure_cells_mem(&bytes, &parsed)
|
||||
.map_err(|e| to_napi_err(e, "extract_tables_with_structure_cells"))?;
|
||||
Ok(result
|
||||
.into_iter()
|
||||
.map(|cells| {
|
||||
cells
|
||||
.into_iter()
|
||||
.map(|c| StructuredCellJs {
|
||||
row: c.row as u32,
|
||||
col: c.col as u32,
|
||||
rowspan: c.rowspan as u32,
|
||||
colspan: c.colspan as u32,
|
||||
is_header: c.is_header,
|
||||
text: c.text,
|
||||
page_pt_bbox: c.page_pt_bbox.iter().map(|v| *v as f64).collect(),
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// One result from `extractTablesWithStructureAuto` — markdown plus a
|
||||
/// diagnostic flag identifying which path produced it.
|
||||
///
|
||||
/// `fallbackReason` is `null` when the TSR-hybrid path produced the
|
||||
/// markdown directly. When stage 1's quality check fires (the cells
|
||||
/// look like a SLANet detection pathology — phantom rows or multi-row
|
||||
/// content in a single cell), the auto path may expand the TSR cells
|
||||
/// in-place or run the heuristic table extractor on the same region.
|
||||
/// `fallbackReason` carries the diagnostic label (for example
|
||||
/// `"multi_row_in_cell_expanded"` or `"phantom_empty_row"`).
|
||||
#[napi(object)]
|
||||
pub struct TableExtractionResultJs {
|
||||
pub markdown: String,
|
||||
pub fallback_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Auto-fallback variant of [`extractTablesWithStructure`].
|
||||
///
|
||||
/// Runs the TSR-hybrid path, checks the resulting cells for known
|
||||
/// SLANet detection pathologies, expands multi-row cells in-place when
|
||||
/// possible, and otherwise falls back to the heuristic
|
||||
/// `extractTablesInRegions` for inputs where the TSR path looks
|
||||
/// compromised.
|
||||
///
|
||||
/// On clean inputs this returns identical markdown to
|
||||
/// `extractTablesWithStructure`; on flagged inputs `fallbackReason` is
|
||||
/// set to the recovery path that produced the result.
|
||||
#[napi]
|
||||
pub fn extract_tables_with_structure_auto(
|
||||
buffer: Buffer,
|
||||
inputs: Vec<TsrTableInputJs>,
|
||||
) -> Result<Vec<TableExtractionResultJs>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let parsed = parse_tsr_inputs(&inputs);
|
||||
|
||||
catch_panic("extract_tables_with_structure_auto", move || {
|
||||
let result = pdf_inspector::extract_tables_with_structure_auto_mem(&bytes, &parsed)
|
||||
.map_err(|e| to_napi_err(e, "extract_tables_with_structure_auto"))?;
|
||||
Ok(result
|
||||
.into_iter()
|
||||
.map(|r| TableExtractionResultJs {
|
||||
markdown: r.markdown,
|
||||
fallback_reason: r.fallback_reason,
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_tsr_inputs(inputs: &[TsrTableInputJs]) -> Vec<pdf_inspector::TsrTableInput> {
|
||||
inputs
|
||||
.iter()
|
||||
.map(|i| {
|
||||
let crop = if i.crop_pdf_pt_bbox.len() == 4 {
|
||||
[
|
||||
i.crop_pdf_pt_bbox[0] as f32,
|
||||
i.crop_pdf_pt_bbox[1] as f32,
|
||||
i.crop_pdf_pt_bbox[2] as f32,
|
||||
i.crop_pdf_pt_bbox[3] as f32,
|
||||
]
|
||||
} else {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
};
|
||||
let cell_bboxes: Vec<Vec<f32>> = i
|
||||
.cell_bboxes
|
||||
.iter()
|
||||
.map(|bb| bb.iter().map(|v| *v as f32).collect())
|
||||
.collect();
|
||||
pdf_inspector::TsrTableInput {
|
||||
page: i.page,
|
||||
crop_pdf_pt_bbox: crop,
|
||||
render_dpi: i.render_dpi as f32,
|
||||
structure_tokens: i.structure_tokens.clone(),
|
||||
cell_bboxes,
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Per-page markdown extraction result.
|
||||
#[napi(object)]
|
||||
pub struct PageMarkdownResult {
|
||||
/// 0-indexed page number.
|
||||
pub page: u32,
|
||||
/// Formatted markdown for this page.
|
||||
pub markdown: String,
|
||||
/// `true` when text on this page is unreliable.
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Combined per-page markdown extraction and layout classification result.
|
||||
#[napi(object)]
|
||||
pub struct PagesExtractionResult {
|
||||
/// Per-page markdown results.
|
||||
pub pages: Vec<PageMarkdownResult>,
|
||||
/// 1-indexed pages where tables were detected.
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// 1-indexed pages where multi-column layout was detected.
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// 1-indexed pages that need OCR (scanned/image-based).
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
/// True if any page has tables or columns.
|
||||
pub is_complex: bool,
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF, with layout classification
|
||||
/// metadata.
|
||||
///
|
||||
/// Returns per-page markdown and classification data (tables, columns,
|
||||
/// OCR needs) from a single parse. Font statistics are computed from the
|
||||
/// full document so header detection is consistent across pages.
|
||||
///
|
||||
/// Omit `pages` (or pass `undefined`) to return every page in document
|
||||
/// order. Pass an array of 0-indexed page numbers to restrict output to
|
||||
/// those pages, in caller-supplied order.
|
||||
#[napi]
|
||||
pub fn extract_pages_markdown(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_pages_markdown", move || {
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|r| PageMarkdownResult {
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_page_regions(page_regions: &[PageRegions]) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||
page_regions
|
||||
.iter()
|
||||
.map(|pr| {
|
||||
let bboxes: Vec<[f32; 4]> = pr
|
||||
@@ -271,25 +672,23 @@ pub fn extract_text_in_regions(
|
||||
.collect();
|
||||
(pr.page, bboxes)
|
||||
})
|
||||
.collect();
|
||||
.collect()
|
||||
}
|
||||
|
||||
catch_panic("extract_text_in_regions", move || {
|
||||
let results = pdf_inspector::extract_text_in_regions_mem(&bytes, ®ions)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_in_regions"))?;
|
||||
|
||||
Ok(results
|
||||
.into_iter()
|
||||
.map(|page_result| PageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<PageRegionTexts> {
|
||||
results
|
||||
.into_iter()
|
||||
.map(|page_result| PageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
@@ -7,6 +7,8 @@ import {
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractTextInRegions,
|
||||
detectVectorGridInRegion,
|
||||
extractPagesMarkdown,
|
||||
} from './index.js';
|
||||
|
||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||
@@ -89,6 +91,39 @@ assert.equal(typeof regionResults[0].regions[0].text, 'string');
|
||||
assert.equal(typeof regionResults[0].regions[0].needsOcr, 'boolean');
|
||||
console.log(' extractTextInRegions: OK');
|
||||
|
||||
// --- detectVectorGridInRegion ---
|
||||
console.log('Testing detectVectorGridInRegion...');
|
||||
const vectorGrid = detectVectorGridInRegion(fixture, 0, [0, 0, 600, 800], 72);
|
||||
assert.ok(vectorGrid === null || typeof vectorGrid === 'object');
|
||||
if (vectorGrid) {
|
||||
assert.ok(Array.isArray(vectorGrid.structureTokens));
|
||||
assert.ok(Array.isArray(vectorGrid.cellBboxes));
|
||||
assert.ok(vectorGrid.cellBboxes.every(bbox => Array.isArray(bbox) && bbox.length === 4));
|
||||
}
|
||||
console.log(' detectVectorGridInRegion: OK');
|
||||
|
||||
// --- extractPagesMarkdown ---
|
||||
console.log('Testing extractPagesMarkdown...');
|
||||
|
||||
// omit pages → every page in document order
|
||||
const allPages = extractPagesMarkdown(fixture);
|
||||
assert.equal(allPages.pages.length, 3);
|
||||
assert.deepEqual(allPages.pages.map(p => p.page), [0, 1, 2]);
|
||||
assert.ok(typeof allPages.pages[0].markdown === 'string');
|
||||
assert.equal(typeof allPages.pages[0].needsOcr, 'boolean');
|
||||
assert.ok(Array.isArray(allPages.pagesWithTables));
|
||||
assert.ok(Array.isArray(allPages.pagesWithColumns));
|
||||
assert.ok(Array.isArray(allPages.pagesNeedingOcr));
|
||||
assert.equal(typeof allPages.isComplex, 'boolean');
|
||||
console.log(' extractPagesMarkdown (no pages arg): OK');
|
||||
|
||||
// selected pages preserve caller order
|
||||
const picked = extractPagesMarkdown(fixture, [2, 0]);
|
||||
assert.equal(picked.pages.length, 2);
|
||||
assert.equal(picked.pages[0].page, 2);
|
||||
assert.equal(picked.pages[1].page, 0);
|
||||
console.log(' extractPagesMarkdown with pages: OK');
|
||||
|
||||
// --- Error handling ---
|
||||
console.log('Testing error handling...');
|
||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||
|
||||
@@ -38,6 +38,8 @@ class TextItem:
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText:
|
||||
@@ -52,6 +54,28 @@ class PageRegionTexts:
|
||||
"""0-indexed page number."""
|
||||
regions: list[RegionText]
|
||||
|
||||
class PageMarkdown:
|
||||
"""Per-page markdown extraction result."""
|
||||
page: int
|
||||
"""0-indexed page number."""
|
||||
markdown: str
|
||||
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
|
||||
needs_ocr: bool
|
||||
"""True when text on this page is unreliable and OCR should be used instead."""
|
||||
|
||||
class PagesExtractionResult:
|
||||
"""Per-page markdown output with document-wide layout classification."""
|
||||
pages: list[PageMarkdown]
|
||||
"""Per-page markdown results, in the order requested."""
|
||||
pages_with_tables: list[int]
|
||||
"""1-indexed pages where tables were detected."""
|
||||
pages_with_columns: list[int]
|
||||
"""1-indexed pages where multi-column layout was detected."""
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed pages that need OCR."""
|
||||
is_complex: bool
|
||||
"""True if any page has tables or multi-column layout."""
|
||||
|
||||
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF: detect type, extract text, convert to Markdown."""
|
||||
...
|
||||
@@ -115,3 +139,31 @@ def extract_text_in_regions_bytes(
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_pages_markdown(
|
||||
path: str,
|
||||
pages: Optional[list[int]] = None,
|
||||
) -> PagesExtractionResult:
|
||||
"""Extract formatted markdown for pages of a PDF, with layout classification.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
pages: Optional list of 0-indexed pages. When ``None`` (default), every
|
||||
page is returned in document order. Otherwise, output matches the
|
||||
caller-supplied order.
|
||||
|
||||
Returns:
|
||||
PagesExtractionResult with per-page markdown and document-wide layout
|
||||
classification (tables, columns, OCR needs).
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_pages_markdown_bytes(
|
||||
data: bytes,
|
||||
pages: Optional[list[int]] = None,
|
||||
) -> PagesExtractionResult:
|
||||
"""Extract formatted markdown for pages of a PDF from bytes.
|
||||
|
||||
See :func:`extract_pages_markdown` for details.
|
||||
"""
|
||||
...
|
||||
|
||||
+3
-1
@@ -4,7 +4,9 @@ build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
# Version is sourced from Cargo.toml [package] version by maturin so the Python
|
||||
# artifact always tracks the crate release instead of drifting on its own.
|
||||
dynamic = ["version"]
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
|
||||
+33
-11
@@ -1,8 +1,12 @@
|
||||
//! CLI tool for detecting PDF type (text-based vs scanned)
|
||||
|
||||
use pdf_inspector::{detect_pdf_type, process_pdf_with_options, PdfOptions, PdfType, ProcessMode};
|
||||
use pdf_inspector::{
|
||||
detect_pdf_type, detector::estimate_page_count_from_bytes, process_pdf_with_options,
|
||||
PdfOptions, PdfType, ProcessMode,
|
||||
};
|
||||
use std::env;
|
||||
use std::fmt::Write;
|
||||
use std::fs;
|
||||
use std::process;
|
||||
use std::time::Instant;
|
||||
|
||||
@@ -64,6 +68,32 @@ fn pdf_type_str(pdf_type: &PdfType) -> &'static str {
|
||||
}
|
||||
}
|
||||
|
||||
fn page_count_hint(pdf_path: &str) -> Option<u32> {
|
||||
fs::read(pdf_path)
|
||||
.ok()
|
||||
.map(|bytes| estimate_page_count_from_bytes(&bytes))
|
||||
.filter(|&count| count > 0)
|
||||
}
|
||||
|
||||
fn print_error(e: &pdf_inspector::PdfError, pdf_path: &str, json_output: bool) {
|
||||
if json_output {
|
||||
if let Some(count) = page_count_hint(pdf_path) {
|
||||
println!(
|
||||
r#"{{"error":"{}","page_count_hint":{}}}"#,
|
||||
json_escape(&e.to_string()),
|
||||
count
|
||||
);
|
||||
} else {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
}
|
||||
} else {
|
||||
eprintln!("Error: {}", e);
|
||||
if let Some(count) = page_count_hint(pdf_path) {
|
||||
eprintln!("Page count hint: {}", count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
match process_pdf_with_options(pdf_path, PdfOptions::new().mode(ProcessMode::Analyze)) {
|
||||
Ok(result) => {
|
||||
@@ -135,11 +165,7 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
if json_output {
|
||||
println!(r#"{{"error":"{}"}}"#, e);
|
||||
} else {
|
||||
eprintln!("Error: {}", e);
|
||||
}
|
||||
print_error(&e, pdf_path, json_output);
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
@@ -236,11 +262,7 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
if json_output {
|
||||
println!(r#"{{"error":"{}"}}"#, e);
|
||||
} else {
|
||||
eprintln!("Error: {}", e);
|
||||
}
|
||||
print_error(&e, pdf_path, json_output);
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
+129
-3
@@ -1,6 +1,10 @@
|
||||
//! CLI tool for PDF to Markdown conversion
|
||||
|
||||
use pdf_inspector::{process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode};
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
|
||||
PdfType, ProcessMode, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
use std::env;
|
||||
use std::fmt::Write;
|
||||
@@ -31,6 +35,110 @@ fn json_escape(s: &str) -> String {
|
||||
out
|
||||
}
|
||||
|
||||
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|entry| {
|
||||
let reasons_json = entry
|
||||
.reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn item_type_label(item_type: &ItemType) -> &'static str {
|
||||
match item_type {
|
||||
ItemType::Text => "text",
|
||||
ItemType::Image => "image",
|
||||
ItemType::Link(_) => "link",
|
||||
ItemType::FormField => "form_field",
|
||||
}
|
||||
}
|
||||
|
||||
fn format_items_json(items: &[TextItem]) -> String {
|
||||
let underlined_count = items.iter().filter(|item| item.is_underline).count();
|
||||
let items_json = items
|
||||
.iter()
|
||||
.map(|item| {
|
||||
let mcid = item
|
||||
.mcid
|
||||
.map(|value| value.to_string())
|
||||
.unwrap_or_else(|| "null".to_string());
|
||||
let link_url = match &item.item_type {
|
||||
ItemType::Link(url) => format!(r#","url":"{}""#, json_escape(url)),
|
||||
_ => String::new(),
|
||||
};
|
||||
format!(
|
||||
r#"{{"text":"{}","page":{},"x":{:.2},"y":{:.2},"width":{:.2},"height":{:.2},"font":"{}","font_size":{:.2},"is_bold":{},"is_italic":{},"is_underline":{},"is_strikeout":{},"item_type":"{}","mcid":{}{}}}"#,
|
||||
json_escape(&item.text),
|
||||
item.page,
|
||||
item.x,
|
||||
item.y,
|
||||
item.width,
|
||||
item.height,
|
||||
json_escape(&item.font),
|
||||
item.font_size,
|
||||
item.is_bold,
|
||||
item.is_italic,
|
||||
item.is_underline,
|
||||
item.is_strikeout,
|
||||
item_type_label(&item.item_type),
|
||||
mcid,
|
||||
link_url,
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
|
||||
format!(
|
||||
r#"{{"total_items":{},"underlined_count":{},"items":[{}]}}"#,
|
||||
items.len(),
|
||||
underlined_count,
|
||||
items_json
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::format_items_json;
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
#[test]
|
||||
fn items_json_includes_position_and_underline_metadata() {
|
||||
let items = vec![TextItem {
|
||||
text: "A \"quoted\" item".to_string(),
|
||||
x: 12.345,
|
||||
y: 67.891,
|
||||
width: 23.456,
|
||||
height: 9.876,
|
||||
font: "F1".to_string(),
|
||||
font_size: 10.0,
|
||||
page: 2,
|
||||
is_bold: false,
|
||||
is_italic: true,
|
||||
is_underline: true,
|
||||
is_strikeout: true,
|
||||
item_type: ItemType::Text,
|
||||
mcid: Some(7),
|
||||
}];
|
||||
|
||||
let json = format_items_json(&items);
|
||||
|
||||
assert!(json.contains(r#""text":"A \"quoted\" item""#));
|
||||
assert!(json.contains(r#""page":2"#));
|
||||
assert!(json.contains(r#""x":12.35"#));
|
||||
assert!(json.contains(r#""is_underline":true"#));
|
||||
assert!(json.contains(r#""item_type":"text""#));
|
||||
assert!(json.contains(r#""mcid":7"#));
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
fn parse_page_spec(spec: &str) -> Result<HashSet<u32>, String> {
|
||||
let mut pages = HashSet::new();
|
||||
@@ -88,6 +196,7 @@ fn main() {
|
||||
if args.len() < 2 {
|
||||
eprintln!("Usage: {} <pdf_file> [output_file]", args[0]);
|
||||
eprintln!(" {} <pdf_file> --json", args[0]);
|
||||
eprintln!(" {} <pdf_file> --items-json", args[0]);
|
||||
eprintln!(" {} <pdf_file> --raw", args[0]);
|
||||
eprintln!();
|
||||
eprintln!("Converts PDF to Markdown with smart type detection.");
|
||||
@@ -95,6 +204,7 @@ fn main() {
|
||||
eprintln!();
|
||||
eprintln!("Options:");
|
||||
eprintln!(" --json Output result as JSON");
|
||||
eprintln!(" --items-json Output positioned TextItem JSON");
|
||||
eprintln!(" --raw Output only markdown (no headers)");
|
||||
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
|
||||
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
||||
@@ -105,6 +215,7 @@ fn main() {
|
||||
|
||||
let pdf_path = &args[1];
|
||||
let json_output = args.iter().any(|a| a == "--json");
|
||||
let items_json_output = args.iter().any(|a| a == "--items-json");
|
||||
let raw_output = args.iter().any(|a| a == "--raw");
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
@@ -129,6 +240,17 @@ fn main() {
|
||||
})
|
||||
});
|
||||
|
||||
if items_json_output {
|
||||
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
|
||||
Ok(items) => println!("{}", format_items_json(&items)),
|
||||
Err(e) => {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
@@ -177,12 +299,14 @@ fn main() {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
|
||||
pdf_type_str,
|
||||
result.page_count,
|
||||
result.processing_time_ms,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
@@ -223,8 +347,9 @@ fn main() {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
|
||||
match result.pdf_type {
|
||||
PdfType::TextBased => "text_based",
|
||||
PdfType::Scanned => "scanned",
|
||||
@@ -236,6 +361,7 @@ fn main() {
|
||||
result.processing_time_ms,
|
||||
result.markdown.as_ref().map(|m| m.len()).unwrap_or(0),
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
|
||||
+2238
-133
File diff suppressed because it is too large
Load Diff
+633
-49
File diff suppressed because it is too large
Load Diff
+699
-16
@@ -4,7 +4,7 @@ use crate::glyph_names::glyph_to_char;
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{FontEncodingMap, FontWidthInfo, PageFontEncodings, PageFontWidths};
|
||||
use log::debug;
|
||||
use lopdf::{Document, Encoding, Object};
|
||||
use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
|
||||
@@ -526,6 +526,11 @@ pub(crate) fn parse_font_encoding(
|
||||
font_dict: &lopdf::Dictionary,
|
||||
) -> Option<EncodingResult> {
|
||||
let encoding_obj = font_dict.get(b"Encoding").ok()?;
|
||||
let base_font_name = font_dict
|
||||
.get(b"BaseFont")
|
||||
.ok()
|
||||
.and_then(|o| o.as_name().ok())
|
||||
.map(|n| String::from_utf8_lossy(n).to_string());
|
||||
|
||||
// Encoding can be a name or a dictionary
|
||||
match encoding_obj {
|
||||
@@ -538,12 +543,14 @@ pub(crate) fn parse_font_encoding(
|
||||
Object::Reference(obj_ref) => {
|
||||
// Reference to encoding dictionary
|
||||
if let Ok(enc_dict) = doc.get_dictionary(*obj_ref) {
|
||||
parse_encoding_dictionary(doc, enc_dict)
|
||||
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
Object::Dictionary(enc_dict) => parse_encoding_dictionary(doc, enc_dict),
|
||||
Object::Dictionary(enc_dict) => {
|
||||
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -562,6 +569,7 @@ pub(crate) struct EncodingResult {
|
||||
pub(crate) fn parse_encoding_dictionary(
|
||||
doc: &Document,
|
||||
enc_dict: &lopdf::Dictionary,
|
||||
base_font_name: Option<&str>,
|
||||
) -> Option<EncodingResult> {
|
||||
let differences = enc_dict.get(b"Differences").ok()?;
|
||||
|
||||
@@ -591,11 +599,9 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
Object::Name(name) => {
|
||||
// Map current code to glyph name -> Unicode
|
||||
let glyph_name = String::from_utf8_lossy(&name).to_string();
|
||||
if glyph_name == "fi"
|
||||
|| glyph_name == "fl"
|
||||
|| glyph_name == "ffi"
|
||||
|| glyph_name == "ffl"
|
||||
{
|
||||
let mapped_char = glyph_to_char(&glyph_name)
|
||||
.or_else(|| private_glyph_to_char(&glyph_name, base_font_name));
|
||||
if mapped_char.is_some_and(is_ligature_char) {
|
||||
debug!(
|
||||
" Differences: code=0x{:02X} glyph={:?} (ligature)",
|
||||
current_code, glyph_name
|
||||
@@ -610,7 +616,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
{
|
||||
gid_glyph_count += 1;
|
||||
}
|
||||
if let Some(ch) = glyph_to_char(&glyph_name) {
|
||||
if let Some(ch) = mapped_char {
|
||||
encoding_map.insert(current_code, ch);
|
||||
} else {
|
||||
debug!(
|
||||
@@ -645,6 +651,31 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
})
|
||||
}
|
||||
|
||||
fn private_glyph_to_char(glyph_name: &str, base_font_name: Option<&str>) -> Option<char> {
|
||||
let base_font_name = strip_subset_prefix(base_font_name?);
|
||||
|
||||
// Aptos CFF subsets from Office PDFs can expose the ff ligature as /g431
|
||||
// without a ToUnicode map. Keep this font-scoped because /gNNN names are private.
|
||||
if base_font_name.eq_ignore_ascii_case("Aptos") && glyph_name == "g431" {
|
||||
Some('\u{FB00}')
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn strip_subset_prefix(font_name: &str) -> &str {
|
||||
font_name
|
||||
.split_once('+')
|
||||
.map_or(font_name, |(_, stripped)| stripped)
|
||||
}
|
||||
|
||||
fn is_ligature_char(ch: char) -> bool {
|
||||
matches!(
|
||||
ch,
|
||||
'\u{FB00}' | '\u{FB01}' | '\u{FB02}' | '\u{FB03}' | '\u{FB04}'
|
||||
)
|
||||
}
|
||||
|
||||
/// Get the CMap lookup key for an Identity-H/V CID font without ToUnicode.
|
||||
/// Returns the object number used by `collect_cmaps_from_fonts` to store the CMap:
|
||||
/// - FontFile2 or FontFile3 obj_num (for embedded font cmap)
|
||||
@@ -708,6 +739,171 @@ pub(crate) fn get_font_file2_obj_num(doc: &Document, font_dict: &lopdf::Dictiona
|
||||
.map(|r| r.0)
|
||||
}
|
||||
|
||||
/// Document-scoped memo of embedded-font style flags, keyed by the
|
||||
/// FontFile2/FontFile3 stream's object id. The same font program is
|
||||
/// referenced from every page that uses the font, and decompressing +
|
||||
/// parsing it dominates `descriptor_style_flags` — without the memo that
|
||||
/// cost repeats per page whenever the descriptor leaves a flag unset
|
||||
/// (the common case: regular fonts report neither italic nor bold).
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct FontStyleCache {
|
||||
by_font_file: HashMap<ObjectId, (bool, bool)>,
|
||||
}
|
||||
|
||||
impl FontStyleCache {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
}
|
||||
|
||||
/// Style flags from the FontDescriptor, which survive subset fonts whose
|
||||
/// BaseFont names are opaque tags ("Tc1", "ABCDEF+F1") that defeat the
|
||||
/// name-based bold/italic heuristics.
|
||||
///
|
||||
/// Italic: `ItalicAngle` beyond a few degrees, or Flags bit 7 (Italic,
|
||||
/// value 64). Bold: Flags bit 19 (ForceBold, value 1<<18). The small
|
||||
/// ItalicAngle threshold skips fonts that declare a token slant.
|
||||
pub(crate) fn descriptor_style_flags(
|
||||
doc: &Document,
|
||||
font_dict: &lopdf::Dictionary,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> (bool, bool) {
|
||||
let descriptor = font_dict
|
||||
.get(b"FontDescriptor")
|
||||
.ok()
|
||||
.and_then(|obj| resolve_dict(doc, obj))
|
||||
.or_else(|| {
|
||||
// Type0 fonts hang the descriptor off DescendantFonts[0].
|
||||
let desc_fonts = font_dict.get(b"DescendantFonts").ok()?;
|
||||
let desc_fonts = resolve_array(doc, desc_fonts)?;
|
||||
let cid_font_dict = resolve_dict(doc, desc_fonts.first()?)?;
|
||||
resolve_dict(doc, cid_font_dict.get(b"FontDescriptor").ok()?)
|
||||
});
|
||||
let Some(descriptor) = descriptor else {
|
||||
return (false, false);
|
||||
};
|
||||
|
||||
let italic_angle = descriptor
|
||||
.get(b"ItalicAngle")
|
||||
.ok()
|
||||
.and_then(|obj| match obj {
|
||||
Object::Integer(i) => Some(*i as f32),
|
||||
Object::Real(r) => Some(*r),
|
||||
_ => None,
|
||||
})
|
||||
.unwrap_or(0.0);
|
||||
let flags = descriptor
|
||||
.get(b"Flags")
|
||||
.ok()
|
||||
.and_then(|obj| obj.as_i64().ok())
|
||||
.unwrap_or(0);
|
||||
|
||||
let mut italic = italic_angle.abs() >= 4.0 || flags & (1 << 6) != 0;
|
||||
let mut bold = flags & (1 << 18) != 0;
|
||||
|
||||
// Descriptors lie: subset generators write ItalicAngle 0 for genuinely
|
||||
// italic faces. The embedded font file keeps the truth — OS/2
|
||||
// fsSelection (via `Face::is_italic`) and the post table's italicAngle.
|
||||
if !italic || !bold {
|
||||
if let Some(ff_ref) = font_file_ref(descriptor) {
|
||||
let (emb_italic, emb_bold) = *style_cache
|
||||
.by_font_file
|
||||
.entry(ff_ref)
|
||||
.or_insert_with(|| embedded_style_flags(doc, ff_ref));
|
||||
italic = italic || emb_italic;
|
||||
bold = bold || emb_bold;
|
||||
}
|
||||
}
|
||||
(italic, bold)
|
||||
}
|
||||
|
||||
/// Style flags parsed from an embedded font program stream.
|
||||
fn embedded_style_flags(doc: &Document, ff_ref: ObjectId) -> (bool, bool) {
|
||||
let Some(data) = font_file_data(doc, ff_ref) else {
|
||||
return (false, false);
|
||||
};
|
||||
if let Ok(face) = ttf_parser::Face::parse(&data, 0) {
|
||||
(
|
||||
face.is_italic() || face.italic_angle().abs() >= 4.0,
|
||||
face.is_bold(),
|
||||
)
|
||||
} else if let Some(name) = cff_font_name(&data) {
|
||||
// FontFile3 is bare CFF (no sfnt container) — ttf_parser
|
||||
// can't open it, but the CFF Name INDEX keeps the real
|
||||
// PostScript name ("XXXXXX+Amplitude-LightItalic") even
|
||||
// when the descriptor was rewritten to claim upright.
|
||||
(
|
||||
crate::text_utils::is_italic_font(&name),
|
||||
crate::text_utils::is_bold_font(&name),
|
||||
)
|
||||
} else {
|
||||
(false, false)
|
||||
}
|
||||
}
|
||||
|
||||
/// First PostScript name from a bare CFF font's Name INDEX (CFF spec §7).
|
||||
fn cff_font_name(data: &[u8]) -> Option<String> {
|
||||
// Header: major(1) minor(1) hdrSize(1) offSize(1); major must be 1.
|
||||
if data.len() < 4 || data[0] != 1 {
|
||||
return None;
|
||||
}
|
||||
let hdr_size = data[2] as usize;
|
||||
// Name INDEX: count(u16) offSize(u8) offsets[count+1] data
|
||||
let count = u16::from_be_bytes([*data.get(hdr_size)?, *data.get(hdr_size + 1)?]) as usize;
|
||||
if count == 0 {
|
||||
return None;
|
||||
}
|
||||
let off_size = *data.get(hdr_size + 2)? as usize;
|
||||
if !(1..=4).contains(&off_size) {
|
||||
return None;
|
||||
}
|
||||
let read_offset = |idx: usize| -> Option<usize> {
|
||||
let at = hdr_size + 3 + idx * off_size;
|
||||
let bytes = data.get(at..at + off_size)?;
|
||||
let mut v = 0usize;
|
||||
for b in bytes {
|
||||
v = (v << 8) | *b as usize;
|
||||
}
|
||||
Some(v)
|
||||
};
|
||||
let start = read_offset(0)?;
|
||||
let end = read_offset(1)?;
|
||||
if start == 0 || end < start {
|
||||
return None;
|
||||
}
|
||||
// Offsets are 1-based from the byte before the object data.
|
||||
let objects_base = hdr_size + 3 + (count + 1) * off_size - 1;
|
||||
let name = data.get(objects_base + start..objects_base + end)?;
|
||||
Some(String::from_utf8_lossy(name).to_string())
|
||||
}
|
||||
|
||||
/// FontFile2/FontFile3 stream reference from a FontDescriptor.
|
||||
fn font_file_ref(descriptor: &lopdf::Dictionary) -> Option<ObjectId> {
|
||||
descriptor
|
||||
.get(b"FontFile2")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.or_else(|| {
|
||||
descriptor
|
||||
.get(b"FontFile3")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
})
|
||||
}
|
||||
|
||||
/// Decompressed embedded font program bytes.
|
||||
fn font_file_data(doc: &Document, ff_ref: ObjectId) -> Option<Vec<u8>> {
|
||||
let stream = doc
|
||||
.get_object(ff_ref)
|
||||
.and_then(lopdf::Object::as_stream)
|
||||
.ok()?;
|
||||
Some(
|
||||
stream
|
||||
.decompressed_content()
|
||||
.unwrap_or_else(|_| stream.content.clone()),
|
||||
)
|
||||
}
|
||||
|
||||
/// Decode text from a PDF string operand using font CMaps, encodings, and fallbacks.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn extract_text_from_operand(
|
||||
@@ -720,7 +916,13 @@ pub(crate) fn extract_text_from_operand(
|
||||
font_encodings: &PageFontEncodings,
|
||||
encoding_cache: &HashMap<String, Encoding<'_>>,
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
font_widths: &PageFontWidths,
|
||||
) -> Option<String> {
|
||||
let is_type0_cid_font = font_widths
|
||||
.get(current_font)
|
||||
.is_some_and(|info| info.is_cid);
|
||||
let use_cp1252_fallback =
|
||||
should_use_cp1252_single_byte_fallback(base_font_name, is_type0_cid_font);
|
||||
let result = (|| -> Option<String> {
|
||||
if let Object::String(bytes, _) = obj {
|
||||
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
|
||||
@@ -751,9 +953,12 @@ pub(crate) fn extract_text_from_operand(
|
||||
return Some(ch.to_string());
|
||||
}
|
||||
}
|
||||
// 4. Printable ASCII/Latin-1 fallback
|
||||
// 4. Printable single-byte fallback
|
||||
if b >= 0x20 {
|
||||
return Some((b as char).to_string());
|
||||
return Some(
|
||||
decode_single_byte_fallback_char(b, use_cp1252_fallback)
|
||||
.to_string(),
|
||||
);
|
||||
}
|
||||
None
|
||||
})
|
||||
@@ -854,6 +1059,13 @@ pub(crate) fn extract_text_from_operand(
|
||||
// unmapped. Don't fall through to text-interpretation fallbacks
|
||||
// (Latin-1, UTF-16, etc.) which would misinterpret CID bytes as
|
||||
// character codes (e.g. CID 0x01A9 → Latin-1 "©").
|
||||
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
|
||||
// 2-byte CIDs (Identity-H) are by far the common case; for
|
||||
// an odd byte count we still emit at least one marker so
|
||||
// detection downstream fires.
|
||||
let cid_count = (bytes.len() / 2).max(1);
|
||||
return Some("\u{FFFD}".repeat(cid_count));
|
||||
}
|
||||
|
||||
// Try our custom encoding map from Differences arrays.
|
||||
// The Differences array overrides specific codes in a base encoding (typically
|
||||
@@ -869,8 +1081,9 @@ pub(crate) fn extract_text_from_operand(
|
||||
Some(ch)
|
||||
} else if b >= 0x20 {
|
||||
// Base encoding fallback for printable bytes.
|
||||
// For codes 0x20-0x7E this matches all standard PDF encodings.
|
||||
Some(b as char)
|
||||
// Most PDFs with simple fonts use WinAnsi/PDFDocEncoding
|
||||
// semantics, not ISO-8859-1 C1 controls.
|
||||
Some(decode_single_byte_fallback_char(b, use_cp1252_fallback))
|
||||
} else {
|
||||
None // Skip unmapped control characters
|
||||
}
|
||||
@@ -926,6 +1139,7 @@ pub(crate) fn extract_text_from_operand(
|
||||
// Try to decode using cached font encoding from lopdf
|
||||
if let Some(encoding) = encoding_cache.get(current_font) {
|
||||
if let Ok(text) = Document::decode_text(encoding, bytes) {
|
||||
let text = normalize_cp1252_controls(text, use_cp1252_fallback);
|
||||
if text.contains('\u{FFFD}') {
|
||||
debug!(
|
||||
"decode_text produced replacement for font={} bytes_len={}",
|
||||
@@ -962,13 +1176,119 @@ pub(crate) fn extract_text_from_operand(
|
||||
return Some(symbol_text);
|
||||
}
|
||||
|
||||
// Latin-1 fallback
|
||||
Some(bytes.iter().map(|&b| b as char).collect())
|
||||
// Non-CID (Type1 / TrueType / Type3) fonts use single-byte
|
||||
// encodings. In practice the fallback should follow WinAnsi for
|
||||
// 0x80..=0x9F so bytes like 0x92 become smart punctuation instead
|
||||
// of C1 controls that look like CID mojibake.
|
||||
Some(decode_single_byte_fallback(bytes, use_cp1252_fallback))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})();
|
||||
result.map(clean_symbol_pua)
|
||||
result.map(|text| {
|
||||
let text = clean_symbol_pua(text);
|
||||
normalize_cp1252_controls(text, use_cp1252_fallback)
|
||||
})
|
||||
}
|
||||
|
||||
fn decode_single_byte_fallback(bytes: &[u8], use_cp1252_fallback: bool) -> String {
|
||||
bytes
|
||||
.iter()
|
||||
.map(|&b| decode_single_byte_fallback_char(b, use_cp1252_fallback))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn decode_single_byte_fallback_char(byte: u8, use_cp1252_fallback: bool) -> char {
|
||||
if !use_cp1252_fallback {
|
||||
return byte as char;
|
||||
}
|
||||
|
||||
match byte {
|
||||
0x80 => '\u{20AC}',
|
||||
0x82 => '\u{201A}',
|
||||
0x83 => '\u{0192}',
|
||||
0x84 => '\u{201E}',
|
||||
0x85 => '\u{2026}',
|
||||
0x86 => '\u{2020}',
|
||||
0x87 => '\u{2021}',
|
||||
0x88 => '\u{02C6}',
|
||||
0x89 => '\u{2030}',
|
||||
0x8A => '\u{0160}',
|
||||
0x8B => '\u{2039}',
|
||||
0x8C => '\u{0152}',
|
||||
0x8E => '\u{017D}',
|
||||
0x91 => '\u{2018}',
|
||||
0x92 => '\u{2019}',
|
||||
0x93 => '\u{201C}',
|
||||
0x94 => '\u{201D}',
|
||||
0x95 => '\u{2022}',
|
||||
0x96 => '\u{2013}',
|
||||
0x97 => '\u{2014}',
|
||||
0x98 => '\u{02DC}',
|
||||
0x99 => '\u{2122}',
|
||||
0x9A => '\u{0161}',
|
||||
0x9B => '\u{203A}',
|
||||
0x9C => '\u{0153}',
|
||||
0x9E => '\u{017E}',
|
||||
0x9F => '\u{0178}',
|
||||
_ => byte as char,
|
||||
}
|
||||
}
|
||||
|
||||
fn normalize_cp1252_controls(text: String, use_cp1252_fallback: bool) -> String {
|
||||
if !use_cp1252_fallback {
|
||||
return text;
|
||||
}
|
||||
if !text
|
||||
.chars()
|
||||
.any(|ch| ('\u{0080}'..='\u{009F}').contains(&ch))
|
||||
{
|
||||
return text;
|
||||
}
|
||||
|
||||
text.chars()
|
||||
.map(|ch| {
|
||||
if ('\u{0080}'..='\u{009F}').contains(&ch) {
|
||||
decode_single_byte_fallback_char(ch as u8, true)
|
||||
} else {
|
||||
ch
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn should_use_cp1252_single_byte_fallback(
|
||||
base_font_name: Option<&str>,
|
||||
is_type0_cid_font: bool,
|
||||
) -> bool {
|
||||
if is_type0_cid_font {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(base_font_name) = base_font_name else {
|
||||
return true;
|
||||
};
|
||||
let font_name = base_font_name
|
||||
.rsplit_once('+')
|
||||
.map_or(base_font_name, |(_, stripped)| stripped)
|
||||
.to_ascii_lowercase();
|
||||
|
||||
// TeX/Computer Modern and math/symbol fonts often place ligatures or
|
||||
// symbols in the C1 byte range. Treating those bytes as Windows-1252 makes
|
||||
// words like "deficiente" become "de…ciente" and "fluid" become "‡uid".
|
||||
let non_cp1252_prefixes = [
|
||||
"cmr", "cmb", "cmmi", "cmsy", "cmex", "cmtt", "cmss", "cmti", "ecrm", "ecbx", "ecti",
|
||||
"tcrm", "tctt", "msam", "msbm", "ttdc",
|
||||
];
|
||||
if non_cp1252_prefixes
|
||||
.iter()
|
||||
.any(|prefix| font_name.starts_with(prefix))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let non_cp1252_names = ["math", "symbol", "dingbat", "emoji"];
|
||||
!non_cp1252_names.iter().any(|name| font_name.contains(name))
|
||||
}
|
||||
|
||||
/// Replace PUA characters in the F000-F0FF range with standard Unicode equivalents.
|
||||
@@ -1090,6 +1410,7 @@ fn score_text(text: &str) -> i32 {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::dictionary;
|
||||
|
||||
fn make_font_info(widths: &[(u16, u16)], default_width: u16, is_cid: bool) -> FontWidthInfo {
|
||||
FontWidthInfo {
|
||||
@@ -1106,6 +1427,172 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn doc_with_descriptor(descriptor: lopdf::Dictionary) -> (Document, lopdf::Dictionary) {
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let desc_id = doc.add_object(descriptor);
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "Tc1",
|
||||
"FontDescriptor" => desc_id,
|
||||
};
|
||||
(doc, font_dict)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn descriptor_italic_angle_sets_italic() {
|
||||
// Subset font with an opaque BaseFont name ("Tc1") — the name
|
||||
// heuristic sees nothing, the descriptor carries the truth.
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => -12,
|
||||
"Flags" => 32,
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(true, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn descriptor_italic_flag_bit_sets_italic() {
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => 0,
|
||||
"Flags" => 64, // bit 7: Italic
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(true, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn descriptor_force_bold_flag_sets_bold() {
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => 0,
|
||||
"Flags" => 1 << 18, // ForceBold
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, true)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiny_italic_angle_is_not_italic() {
|
||||
// A token 1-degree slant is optical correction, not italic.
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => lopdf::Object::Real(-1.0),
|
||||
"Flags" => 32,
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_descriptor_yields_no_flags() {
|
||||
let doc = Document::with_version("1.4");
|
||||
let font_dict = dictionary! { "Type" => "Font", "BaseFont" => "Tc1" };
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type0_descendant_descriptor_is_resolved() {
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let desc_id = doc.add_object(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "ABCDEF+F1",
|
||||
"ItalicAngle" => -15,
|
||||
});
|
||||
let cid_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "CIDFontType2",
|
||||
"FontDescriptor" => desc_id,
|
||||
});
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type0",
|
||||
"BaseFont" => "ABCDEF+F1",
|
||||
"DescendantFonts" => vec![lopdf::Object::Reference(cid_id)],
|
||||
};
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(true, false)
|
||||
);
|
||||
}
|
||||
|
||||
/// Bare CFF: header + Name INDEX only — enough for `cff_font_name`.
|
||||
fn bare_cff_with_name(name: &str) -> Vec<u8> {
|
||||
let mut data = vec![1, 0, 4, 1]; // major, minor, hdrSize, offSize
|
||||
data.extend_from_slice(&1u16.to_be_bytes()); // Name INDEX count
|
||||
data.push(1); // offSize
|
||||
data.push(1); // offset of first name
|
||||
data.push(1 + name.len() as u8); // offset past last name
|
||||
data.extend_from_slice(name.as_bytes());
|
||||
data
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_font_style_is_cached_by_font_file_object() {
|
||||
use lopdf::{Object, Stream};
|
||||
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let ff_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
bare_cff_with_name("ABCDEF+Test-BoldItalic"),
|
||||
)));
|
||||
let desc_id = doc.add_object(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "ABCDEF+Test-BoldItalic",
|
||||
"ItalicAngle" => 0,
|
||||
"Flags" => 32,
|
||||
"FontFile3" => ff_id,
|
||||
});
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Tc1",
|
||||
"FontDescriptor" => desc_id,
|
||||
};
|
||||
|
||||
let mut cache = FontStyleCache::new();
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut cache),
|
||||
(true, true)
|
||||
);
|
||||
assert_eq!(cache.by_font_file.len(), 1);
|
||||
|
||||
// Replace the font program with garbage: a repeat call must serve
|
||||
// the memo instead of re-reading the stream — repeated per-page
|
||||
// decompression is exactly what the cache exists to avoid.
|
||||
doc.objects.insert(
|
||||
ff_id,
|
||||
Object::Stream(Stream::new(dictionary! {}, vec![0u8; 4])),
|
||||
);
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut cache),
|
||||
(true, true)
|
||||
);
|
||||
// A cold cache parses the (now garbage) stream, proving the warm
|
||||
// call above answered from the memo.
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compute_string_width_ts_no_tc_tw() {
|
||||
// Without Tc/Tw (both 0), width = glyph widths only
|
||||
@@ -1213,4 +1700,200 @@ mod tests {
|
||||
let bad = "###!!!@@@$$$";
|
||||
assert!(score_text(good) > score_text(bad));
|
||||
}
|
||||
|
||||
fn doc_with_private_differences() -> (Document, lopdf::ObjectId) {
|
||||
let mut doc = Document::with_version("1.7");
|
||||
let encoding_id = doc.add_object(dictionary! {
|
||||
"Differences" => Object::Array(vec![
|
||||
Object::Integer(0x88),
|
||||
Object::Name(b"g431".to_vec()),
|
||||
Object::Name(b"fi".to_vec()),
|
||||
Object::Integer(0xAD),
|
||||
Object::Name(b"fl".to_vec()),
|
||||
]),
|
||||
});
|
||||
|
||||
(doc, encoding_id)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn aptos_private_g431_maps_to_ff_ligature() {
|
||||
let (doc, encoding_id) = doc_with_private_differences();
|
||||
let font_dict = dictionary! {
|
||||
"BaseFont" => Object::Name(b"NJEQOD+Aptos".to_vec()),
|
||||
"Encoding" => Object::Reference(encoding_id),
|
||||
};
|
||||
|
||||
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
|
||||
|
||||
assert_eq!(result.map.get(&0x88u8), Some(&'\u{FB00}'));
|
||||
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
|
||||
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn private_g431_does_not_map_for_unrelated_fonts() {
|
||||
let (doc, encoding_id) = doc_with_private_differences();
|
||||
let font_dict = dictionary! {
|
||||
"BaseFont" => Object::Name(b"ABCDEF+OtherFont".to_vec()),
|
||||
"Encoding" => Object::Reference(encoding_id),
|
||||
};
|
||||
|
||||
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
|
||||
|
||||
assert!(!result.map.contains_key(&0x88u8));
|
||||
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
|
||||
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() {
|
||||
// Type0/CID font (font_widths reports `is_cid=true`) where the
|
||||
// ToUnicode CMap couldn't be parsed (FontCMaps doesn't have the
|
||||
// obj_num). Bytes are a 2-byte CID stream containing high bytes
|
||||
// that aren't valid UTF-8 — exactly the case in the production
|
||||
// samples (Identity-H text where the ToUnicode CMap was missing
|
||||
// or malformed, scrape_id 019de78c-..., e.g. "Í Ù Z)¿").
|
||||
//
|
||||
// Without the guard, the function falls through to the byte-by-byte
|
||||
// Latin-1 fallback and produces "ÍÙ" (U+00CD U+00D9). The correct
|
||||
// behavior is to emit U+FFFD per CID so downstream
|
||||
// `detect_encoding_issues` flags the page for OCR.
|
||||
let bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9];
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
let font_cmaps = FontCMaps::default();
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
font_tounicode_refs.insert("F0".to_string(), 999);
|
||||
let inline_cmaps = HashMap::new();
|
||||
let font_encodings: PageFontEncodings = HashMap::new();
|
||||
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||
let mut decisions = CMapDecisionCache::new();
|
||||
let mut font_widths: PageFontWidths = HashMap::new();
|
||||
font_widths.insert("F0".to_string(), make_font_info(&[], 1000, true));
|
||||
|
||||
let result = extract_text_from_operand(
|
||||
&obj,
|
||||
"F0",
|
||||
None,
|
||||
&font_cmaps,
|
||||
&font_tounicode_refs,
|
||||
&inline_cmaps,
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut decisions,
|
||||
&font_widths,
|
||||
);
|
||||
|
||||
let text = result.expect("CID font fallback should still emit a marker");
|
||||
assert!(
|
||||
!text.contains('\u{00CD}') && !text.contains('\u{00D9}'),
|
||||
"CID font with unparseable CMap leaked Latin-1 mojibake: {text:?}"
|
||||
);
|
||||
assert!(
|
||||
text.contains('\u{FFFD}'),
|
||||
"CID font with unparseable CMap should emit U+FFFD so detect_encoding_issues fires: {text:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn simple_font_single_byte_fallback_passes_high_bytes_through() {
|
||||
// A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode`
|
||||
// reference but no usable CMap and no `/Differences` map.
|
||||
// Per-byte fallback is the canonical interpretation here — these
|
||||
// bytes are character codes, not CIDs. The CID guard must NOT strip
|
||||
// them. Reproduces the false positive that an earlier version of the
|
||||
// guard introduced for fonts in PDFs like pdf-evals/Navigating-
|
||||
// Artificial-Intelligence-..., where bytes like 0xB6 are legitimate
|
||||
// single-byte character codes.
|
||||
let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V"
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
let font_cmaps = FontCMaps::default();
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
font_tounicode_refs.insert("F1".to_string(), 999);
|
||||
let inline_cmaps = HashMap::new();
|
||||
let font_encodings: PageFontEncodings = HashMap::new();
|
||||
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||
let mut decisions = CMapDecisionCache::new();
|
||||
let mut font_widths: PageFontWidths = HashMap::new();
|
||||
font_widths.insert("F1".to_string(), make_font_info(&[], 1000, false));
|
||||
|
||||
let text = extract_text_from_operand(
|
||||
&obj,
|
||||
"F1",
|
||||
None,
|
||||
&font_cmaps,
|
||||
&font_tounicode_refs,
|
||||
&inline_cmaps,
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut decisions,
|
||||
&font_widths,
|
||||
)
|
||||
.expect("simple font should round-trip Latin-1 bytes");
|
||||
assert_eq!(text, "$G\u{00B6}V");
|
||||
assert!(
|
||||
!text.contains('\u{FFFD}'),
|
||||
"simple font fallback must not stamp FFFD over legitimate bytes: {text:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn simple_font_single_byte_fallback_maps_cp1252_punctuation() {
|
||||
let bytes = vec![b'l', 0x92_u8, b'a', b'c', b'a', b'd'];
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
let font_cmaps = FontCMaps::default();
|
||||
let font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
let inline_cmaps = HashMap::new();
|
||||
let font_encodings: PageFontEncodings = HashMap::new();
|
||||
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||
let mut decisions = CMapDecisionCache::new();
|
||||
let font_widths: PageFontWidths = HashMap::new();
|
||||
|
||||
let text = extract_text_from_operand(
|
||||
&obj,
|
||||
"F1",
|
||||
None,
|
||||
&font_cmaps,
|
||||
&font_tounicode_refs,
|
||||
&inline_cmaps,
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut decisions,
|
||||
&font_widths,
|
||||
)
|
||||
.expect("simple font should decode CP1252 punctuation");
|
||||
|
||||
assert_eq!(text, "l’acad");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cached_encoding_decode_normalizes_cp1252_controls() {
|
||||
let text = normalize_cp1252_controls("d\u{92}un \u{96} test".to_string(), true);
|
||||
assert_eq!(text, "d’un – test");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tex_font_decode_keeps_c1_ligature_bytes_unmodified() {
|
||||
let text = normalize_cp1252_controls("de\u{85}ciente \u{87}uid".to_string(), false);
|
||||
assert_eq!(text, "de\u{85}ciente \u{87}uid");
|
||||
assert!(!should_use_cp1252_single_byte_fallback(
|
||||
Some("TTdcr10"),
|
||||
false
|
||||
));
|
||||
assert!(!should_use_cp1252_single_byte_fallback(
|
||||
Some("cmr10"),
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn winansi_text_font_uses_cp1252_fallback() {
|
||||
assert!(should_use_cp1252_single_byte_fallback(
|
||||
Some("BJPQNQ+Times-Roman"),
|
||||
false
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
+296
-10
@@ -29,12 +29,17 @@ pub(crate) fn detect_columns(
|
||||
const MIN_ITEMS_PER_COLUMN: usize = 10;
|
||||
const NOISE_FRACTION: f32 = 0.15;
|
||||
|
||||
// Get items for this page
|
||||
let page_items: Vec<&TextItem> = items.iter().filter(|i| i.page == page).collect();
|
||||
// Get items for this page. Strip Image placeholders — an image's left edge
|
||||
// would otherwise count toward the column projection profile.
|
||||
let page_items: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|i| i.page == page && crate::extractor::is_text_layout_item(i))
|
||||
.collect();
|
||||
|
||||
if page_items.is_empty() {
|
||||
return vec![];
|
||||
}
|
||||
debug!("page {}: detect_columns: {} items", page, page_items.len());
|
||||
|
||||
// Find page bounds
|
||||
let x_min = page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
@@ -163,10 +168,17 @@ pub(crate) fn detect_columns(
|
||||
}
|
||||
}
|
||||
}
|
||||
// Try XY-cut fallback before giving up
|
||||
if let Some(columns) = try_xy_cut_split(&page_items, x_min, x_max, page) {
|
||||
return columns;
|
||||
}
|
||||
return vec![ColumnRegion { x_min, x_max }];
|
||||
}
|
||||
|
||||
return validate_and_build_columns(
|
||||
// Try center-based assignment first (handles asymmetric layouts / sidebars
|
||||
// better than edge-based). Fall back to edge-based if center produces
|
||||
// a degenerate split (one side empty).
|
||||
let result = validate_and_build_columns(
|
||||
&valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
@@ -175,8 +187,177 @@ pub(crate) fn detect_columns(
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
page,
|
||||
false, // edge-based assignment for absolute valleys
|
||||
true, // center-based assignment
|
||||
);
|
||||
if result.len() > 1 {
|
||||
return result;
|
||||
}
|
||||
let result = validate_and_build_columns(
|
||||
&valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
BIN_WIDTH,
|
||||
x_max,
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
page,
|
||||
false, // edge-based fallback
|
||||
);
|
||||
if result.len() > 1 {
|
||||
return result;
|
||||
}
|
||||
|
||||
// Fallback: XY-cut style gap detection. When the histogram finds no
|
||||
// clear valleys (common with asymmetric/sidebar layouts), look for the
|
||||
// largest horizontal gap between item edges. This is a simplified
|
||||
// single-level XY-cut inspired by opendataloader's XY-Cut++ algorithm.
|
||||
if page_items.len() >= 20 && !page_has_table {
|
||||
if let Some(columns) = try_xy_cut_split(&page_items, x_min, x_max, page) {
|
||||
return columns;
|
||||
}
|
||||
}
|
||||
|
||||
vec![ColumnRegion { x_min, x_max }]
|
||||
}
|
||||
|
||||
/// Simplified single-level XY-cut: find the largest horizontal gap between
|
||||
/// item right-edges and left-edges. If the gap is wide enough and both sides
|
||||
/// have sufficient items with vertical overlap, split into two columns.
|
||||
///
|
||||
/// Inspired by opendataloader's XY-Cut++ algorithm but without full recursion.
|
||||
/// Handles asymmetric layouts (sidebars) that the histogram misses because
|
||||
/// the narrow column has too few items to register in the occupancy profile.
|
||||
fn try_xy_cut_split(
|
||||
page_items: &[&TextItem],
|
||||
page_x_min: f32,
|
||||
page_x_max: f32,
|
||||
page: u32,
|
||||
) -> Option<Vec<ColumnRegion>> {
|
||||
const MIN_GAP: f32 = 15.0; // minimum gap to consider a split
|
||||
const MIN_ITEMS_MAJOR: usize = 10; // major column must have ≥10 items
|
||||
const MIN_ITEMS_MINOR: usize = 3; // minor column (sidebar) must have ≥3
|
||||
|
||||
let page_width = page_x_max - page_x_min;
|
||||
if page_width < 200.0 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Collect all item edges: (right_edge, left_edge) pairs sorted by right_edge
|
||||
// The gap between one item's right edge and the next item's left edge
|
||||
// reveals column gutters.
|
||||
let mut edges: Vec<(f32, f32)> = page_items
|
||||
.iter()
|
||||
.map(|i| (i.x, i.x + effective_width(i)))
|
||||
.collect();
|
||||
edges.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
// Find the largest gap between consecutive items (by left edge).
|
||||
// Use a sweep: sort left edges, find max gap between sorted right edges
|
||||
// of items to the left and left edges of items to the right.
|
||||
let mut left_edges: Vec<f32> = page_items.iter().map(|i| i.x).collect();
|
||||
left_edges.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// Build prefix max of right edges (for items sorted by left edge)
|
||||
let mut sorted_by_left: Vec<(f32, f32)> = page_items
|
||||
.iter()
|
||||
.map(|i| (i.x, i.x + effective_width(i)))
|
||||
.collect();
|
||||
sorted_by_left.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
let mut best_gap = 0.0f32;
|
||||
let mut best_split = 0.0f32;
|
||||
let mut max_right_so_far = f32::NEG_INFINITY;
|
||||
|
||||
for i in 0..sorted_by_left.len() - 1 {
|
||||
let (_, right) = sorted_by_left[i];
|
||||
max_right_so_far = max_right_so_far.max(right);
|
||||
|
||||
let (next_left, _) = sorted_by_left[i + 1];
|
||||
let gap = next_left - max_right_so_far;
|
||||
if gap > best_gap {
|
||||
best_gap = gap;
|
||||
best_split = (max_right_so_far + next_left) / 2.0;
|
||||
}
|
||||
}
|
||||
|
||||
if best_gap < MIN_GAP {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Don't split at page margins (within 10% of edges)
|
||||
let margin = page_width * 0.10;
|
||||
if best_split - page_x_min < margin || page_x_max - best_split < margin {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Count items on each side
|
||||
let left_count = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 <= best_split)
|
||||
.count();
|
||||
let right_count = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 > best_split)
|
||||
.count();
|
||||
|
||||
let (minor, major) = if left_count <= right_count {
|
||||
(left_count, right_count)
|
||||
} else {
|
||||
(right_count, left_count)
|
||||
};
|
||||
|
||||
if major < MIN_ITEMS_MAJOR || minor < MIN_ITEMS_MINOR {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Check vertical overlap — both sides should span a meaningful Y range
|
||||
let left_items: Vec<&&TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 <= best_split)
|
||||
.collect();
|
||||
let right_items: Vec<&&TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 > best_split)
|
||||
.collect();
|
||||
|
||||
let l_y_min = left_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
let l_y_max = left_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let r_y_min = right_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let r_y_max = right_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
|
||||
let overlap_min = l_y_min.max(r_y_min);
|
||||
let overlap_max = l_y_max.min(r_y_max);
|
||||
let overlap = (overlap_max - overlap_min).max(0.0);
|
||||
let y_range = (l_y_max.max(r_y_max) - l_y_min.min(r_y_min)).max(1.0);
|
||||
|
||||
if overlap / y_range < 0.20 {
|
||||
return None;
|
||||
}
|
||||
|
||||
debug!(
|
||||
"page {}: XY-cut split at x={:.1} (gap={:.1}pt, left={}, right={})",
|
||||
page, best_split, best_gap, left_count, right_count
|
||||
);
|
||||
|
||||
Some(vec![
|
||||
ColumnRegion {
|
||||
x_min: page_x_min,
|
||||
x_max: best_split,
|
||||
},
|
||||
ColumnRegion {
|
||||
x_min: best_split,
|
||||
x_max: page_x_max,
|
||||
},
|
||||
])
|
||||
}
|
||||
|
||||
/// Check whether each proposed column contains paragraph-like content.
|
||||
@@ -449,6 +630,33 @@ fn find_relative_valleys(
|
||||
valleys
|
||||
}
|
||||
|
||||
/// Detect whether a side of a gutter consists predominantly of list-marker
|
||||
/// glyphs (•, ●, ○, ◦, ▪, ▫, ◆, ◇). A column of bullets on the left margin
|
||||
/// creates a spurious histogram valley between the bullet and the content.
|
||||
/// Treating it as a real column splits each list item's text across two
|
||||
/// "columns," so we reject these candidates.
|
||||
fn is_list_marker_column(items: &[&&TextItem]) -> bool {
|
||||
const LIST_MARKERS: &[char] = &['•', '●', '○', '◦', '▪', '▫', '◆', '◇', '■', '□'];
|
||||
if items.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let marker_count = items
|
||||
.iter()
|
||||
.filter(|i| {
|
||||
let t = i.text.trim();
|
||||
let mut chars = t.chars();
|
||||
match (chars.next(), chars.next()) {
|
||||
(Some(c), None) => LIST_MARKERS.contains(&c),
|
||||
_ => false,
|
||||
}
|
||||
})
|
||||
.count();
|
||||
// Require ≥80% of items on this side to be standalone markers. A handful
|
||||
// of non-marker items (stray page numbers, footnote refs) shouldn't
|
||||
// defeat the check.
|
||||
marker_count as f32 / items.len() as f32 >= 0.8
|
||||
}
|
||||
|
||||
/// Validate valley candidates with vertical consistency checks and build column regions.
|
||||
///
|
||||
/// When `center_assign` is true, items are assigned to columns based on their
|
||||
@@ -505,7 +713,28 @@ fn validate_and_build_columns(
|
||||
})
|
||||
.collect();
|
||||
|
||||
if left_items.len() < min_items || right_items.len() < min_items {
|
||||
// Require both sides to have items. Symmetric layout needs min_items
|
||||
// on each side. Asymmetric layouts (sidebars) are accepted when the
|
||||
// dominant side has ≥ min_items and the smaller side has ≥ 3 items.
|
||||
let (smaller, larger) = if left_items.len() <= right_items.len() {
|
||||
(left_items.len(), right_items.len())
|
||||
} else {
|
||||
(right_items.len(), left_items.len())
|
||||
};
|
||||
if larger < min_items || smaller < 3 {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Reject valleys where the smaller side is just a column of list
|
||||
// markers (bullets aligned at the left margin). This is a common
|
||||
// pattern in PDFs where ● starts each list item: histogram detection
|
||||
// sees the gap between bullet and content as a gutter.
|
||||
let smaller_items: &[&&TextItem] = if left_items.len() <= right_items.len() {
|
||||
&left_items
|
||||
} else {
|
||||
&right_items
|
||||
};
|
||||
if is_list_marker_column(smaller_items) {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -1005,11 +1234,7 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
ci,
|
||||
item.x,
|
||||
item.y,
|
||||
if item.text.len() > 60 {
|
||||
&item.text[..60]
|
||||
} else {
|
||||
&item.text
|
||||
}
|
||||
super::trace_text_preview(&item.text, 60)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1269,6 +1494,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1398,6 +1625,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -1595,6 +1824,63 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bullet_marker_column_not_detected_as_column() {
|
||||
// Pattern: every line is `● <content>`, with ● at x=90 and content
|
||||
// starting at x=104. Histogram detection sees a gutter between them
|
||||
// and would split the page into a "bullet column" and "content column",
|
||||
// scrambling every list item.
|
||||
let mut items = Vec::new();
|
||||
for i in 0..15 {
|
||||
let y = 750.0 - i as f32 * 30.0;
|
||||
items.push(make_item(1, 90.0, y, "●"));
|
||||
items.push(make_item(
|
||||
1,
|
||||
104.0,
|
||||
y,
|
||||
"FullContentLineTextHere________________",
|
||||
));
|
||||
}
|
||||
// Pad with content to satisfy min item count for column detection.
|
||||
for i in 0..15 {
|
||||
let y = 300.0 - i as f32 * 14.0;
|
||||
items.push(make_item(1, 72.0, y, "FootnoteText_____________________"));
|
||||
}
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert_eq!(
|
||||
cols.len(),
|
||||
1,
|
||||
"Bullet markers aligned at left margin should not be treated as their own column"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_list_marker_column_detects_bullets() {
|
||||
let items = vec![
|
||||
make_item(1, 90.0, 100.0, "●"),
|
||||
make_item(1, 90.0, 114.0, "●"),
|
||||
make_item(1, 90.0, 128.0, "●"),
|
||||
make_item(1, 90.0, 142.0, "●"),
|
||||
];
|
||||
let refs: Vec<&TextItem> = items.iter().collect();
|
||||
let wrapped: Vec<&&TextItem> = refs.iter().collect();
|
||||
assert!(is_list_marker_column(&wrapped));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_list_marker_column_rejects_prose() {
|
||||
let items = vec![
|
||||
make_item(1, 30.0, 100.0, "Regular prose line"),
|
||||
make_item(1, 30.0, 114.0, "Another sentence"),
|
||||
make_item(1, 30.0, 128.0, "Third line"),
|
||||
make_item(1, 30.0, 142.0, "Fourth line"),
|
||||
];
|
||||
let refs: Vec<&TextItem> = items.iter().collect();
|
||||
let wrapped: Vec<&&TextItem> = refs.iter().collect();
|
||||
assert!(!is_list_marker_column(&wrapped));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn premask_narrow_line_not_masked() {
|
||||
// Items that form a line spanning only ~40% of column width → not masked
|
||||
|
||||
@@ -78,6 +78,8 @@ pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> V
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Link(url),
|
||||
mcid: None,
|
||||
});
|
||||
@@ -316,6 +318,8 @@ pub(crate) fn walk_form_fields(
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::FormField,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
+585
-50
@@ -6,11 +6,12 @@ pub(crate) mod content_stream;
|
||||
mod fonts;
|
||||
mod layout;
|
||||
mod links;
|
||||
pub(crate) mod underline;
|
||||
mod xobjects;
|
||||
|
||||
use crate::text_utils::is_rtl_text;
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{PageExtraction, TextItem};
|
||||
use crate::types::{PageExtraction, PdfLine, PdfRect, TextItem};
|
||||
use crate::PdfError;
|
||||
use log::debug;
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
@@ -23,6 +24,7 @@ use links::{extract_form_fields, extract_page_links};
|
||||
// Re-export public types so existing `crate::extractor::X` paths keep working.
|
||||
pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
pub use layout::group_into_lines;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
@@ -33,29 +35,24 @@ pub(crate) use layout::ColumnRegion;
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub(crate) fn trace_text_preview(text: &str, max_chars: usize) -> &str {
|
||||
match text.char_indices().nth(max_chars) {
|
||||
Some((idx, _)) => &text[..idx],
|
||||
None => text,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract text from PDF file as plain string
|
||||
pub fn extract_text<P: AsRef<Path>>(path: P) -> Result<String, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let doc = match Document::load(&path) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_with_password(&path, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
extract_text_from_doc(&doc)
|
||||
}
|
||||
|
||||
/// Extract text from PDF memory buffer
|
||||
pub fn extract_text_mem(buffer: &[u8]) -> Result<String, PdfError> {
|
||||
crate::validate_pdf_bytes(buffer)?;
|
||||
let doc = match Document::load_mem(buffer) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_mem_with_options(buffer, lopdf::LoadOptions::with_password(""))?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_mem(buffer)?;
|
||||
extract_text_from_doc(&doc)
|
||||
}
|
||||
|
||||
@@ -91,13 +88,7 @@ pub(crate) fn extract_text_with_positions_and_rects<P: AsRef<Path>>(
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let doc = match Document::load(&path) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_with_password(&path, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
@@ -124,13 +115,7 @@ pub(crate) fn extract_text_with_positions_mem_and_rects(
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_bytes(buffer)?;
|
||||
let doc = match Document::load_mem(buffer) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_mem_with_options(buffer, lopdf::LoadOptions::with_password(""))?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
@@ -177,6 +162,9 @@ fn extract_positioned_text_impl(
|
||||
let mut all_lines = Vec::new();
|
||||
let mut page_thresholds: PageThresholds = HashMap::new();
|
||||
let mut gid_encoded_pages: HashSet<u32> = HashSet::new();
|
||||
// Embedded-font style flags are document-scoped: the same font program
|
||||
// is shared across pages, so parse it once, not once per page.
|
||||
let mut style_cache = FontStyleCache::new();
|
||||
|
||||
// Build page ObjectId → page number map for form field extraction
|
||||
let page_id_to_num: HashMap<ObjectId, u32> =
|
||||
@@ -188,8 +176,14 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, rects, lines), has_gid_fonts) =
|
||||
extract_page_text_items(doc, page_id, *page_num, font_cmaps, include_invisible)?;
|
||||
let ((mut items, rects, lines), has_gid_fonts, _coords_rotated) = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
if has_gid_fonts {
|
||||
gid_encoded_pages.insert(*page_num);
|
||||
}
|
||||
@@ -197,6 +191,7 @@ fn extract_positioned_text_impl(
|
||||
if threshold > 0.10 {
|
||||
page_thresholds.insert(*page_num, threshold);
|
||||
}
|
||||
suppress_table_underlines(&mut items, &rects, &lines, *page_num);
|
||||
debug!(
|
||||
"page {}: {} text items, {} rects, {} lines{}",
|
||||
page_num,
|
||||
@@ -219,11 +214,7 @@ fn extract_positioned_text_impl(
|
||||
item.width,
|
||||
item.font_size,
|
||||
item.font,
|
||||
if item.text.len() > 80 {
|
||||
&item.text[..80]
|
||||
} else {
|
||||
&item.text
|
||||
}
|
||||
trace_text_preview(&item.text, 80)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -247,10 +238,109 @@ fn extract_positioned_text_impl(
|
||||
))
|
||||
}
|
||||
|
||||
fn suppress_table_underlines(
|
||||
items: &mut [TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[PdfLine],
|
||||
page: u32,
|
||||
) {
|
||||
if !items
|
||||
.iter()
|
||||
.any(|item| item.is_underline || item.is_strikeout)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
let mut table_item_indices: HashSet<usize> = HashSet::new();
|
||||
|
||||
if !rects.is_empty() {
|
||||
let (rect_tables, _) = crate::tables::detect_tables_from_rects(items, rects, page);
|
||||
for table in rect_tables {
|
||||
table_item_indices.extend(table.item_indices);
|
||||
}
|
||||
}
|
||||
|
||||
if !lines.is_empty() {
|
||||
for table in crate::tables::detect_tables_from_lines(items, lines, page) {
|
||||
table_item_indices.extend(table.item_indices);
|
||||
}
|
||||
}
|
||||
|
||||
for index in table_item_indices {
|
||||
if let Some(item) = items.get_mut(index) {
|
||||
item.is_underline = false;
|
||||
item.is_strikeout = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Shared helpers (used by submodules via `super::`)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Return true when this item should participate in text-layout
|
||||
/// heuristics (column detection, table grid detection, line grouping).
|
||||
///
|
||||
/// Image XObjects emit a positional placeholder via
|
||||
/// `extract_text_with_positions` (so layout-aware callers can crop +
|
||||
/// caption figures), but their bboxes don't carry text glyphs and would
|
||||
/// skew column/row clustering if they reached the heuristics. Hyperlinks
|
||||
/// and form fields *do* participate — the existing logic treats them as
|
||||
/// text-like and we keep that.
|
||||
pub(crate) fn is_text_layout_item(item: &crate::types::TextItem) -> bool {
|
||||
!matches!(item.item_type, crate::types::ItemType::Image)
|
||||
}
|
||||
|
||||
/// Map a (u, v) point in unit-square coordinates through the 6-element CTM
|
||||
/// to page-space. CTM format is `[a, b, c, d, e, f]` per
|
||||
/// [`multiply_matrices`].
|
||||
fn apply_ctm_point(ctm: &[f32; 6], u: f32, v: f32) -> (f32, f32) {
|
||||
(
|
||||
u * ctm[0] + v * ctm[2] + ctm[4],
|
||||
u * ctm[1] + v * ctm[3] + ctm[5],
|
||||
)
|
||||
}
|
||||
|
||||
/// Compute the page-space axis-aligned bounding box of an Image XObject
|
||||
/// invoked under the given CTM.
|
||||
///
|
||||
/// Per the PDF spec, an image XObject is always rendered into a unit
|
||||
/// square `(0,0)–(1,1)` in its local coordinate system, and the `Do`
|
||||
/// operator applies the current CTM to position/scale/rotate that square
|
||||
/// onto the page. For the common axis-aligned case (no rotation/shear),
|
||||
/// the CTM reduces to `[w, 0, 0, h, x, y]` and the bbox is just
|
||||
/// `(x, y, w, h)`. For rotated/sheared images we transform all four
|
||||
/// corners and return their axis-aligned bbox so the caller always gets
|
||||
/// an upright rectangle.
|
||||
///
|
||||
/// Coordinates are PDF user space (origin at bottom-left, y-up). Width
|
||||
/// and height are non-negative.
|
||||
pub(crate) fn image_bbox_from_ctm(ctm: &[f32; 6]) -> (f32, f32, f32, f32) {
|
||||
let corners = [
|
||||
apply_ctm_point(ctm, 0.0, 0.0),
|
||||
apply_ctm_point(ctm, 1.0, 0.0),
|
||||
apply_ctm_point(ctm, 1.0, 1.0),
|
||||
apply_ctm_point(ctm, 0.0, 1.0),
|
||||
];
|
||||
let (mut x_min, mut x_max) = (corners[0].0, corners[0].0);
|
||||
let (mut y_min, mut y_max) = (corners[0].1, corners[0].1);
|
||||
for (cx, cy) in corners.iter().skip(1) {
|
||||
if *cx < x_min {
|
||||
x_min = *cx;
|
||||
}
|
||||
if *cx > x_max {
|
||||
x_max = *cx;
|
||||
}
|
||||
if *cy < y_min {
|
||||
y_min = *cy;
|
||||
}
|
||||
if *cy > y_max {
|
||||
y_max = *cy;
|
||||
}
|
||||
}
|
||||
(x_min, y_min, x_max - x_min, y_max - y_min)
|
||||
}
|
||||
|
||||
/// Multiply two 2D transformation matrices
|
||||
/// Matrix format: [a, b, c, d, e, f] representing:
|
||||
/// | a b 0 |
|
||||
@@ -310,6 +400,133 @@ fn effective_merge_width(item: &TextItem) -> f32 {
|
||||
}
|
||||
}
|
||||
|
||||
fn is_standalone_bullet_text(text: &str) -> bool {
|
||||
matches!(text.trim(), "•" | "○" | "●" | "◦")
|
||||
}
|
||||
|
||||
fn first_text_char(text: &str) -> Option<char> {
|
||||
text.trim_start().chars().next()
|
||||
}
|
||||
|
||||
fn is_short_alpha_fragment(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
let char_count = trimmed.chars().count();
|
||||
(1..=4).contains(&char_count) && trimmed.chars().all(char::is_alphabetic)
|
||||
}
|
||||
|
||||
fn has_phrase_continuation_shape(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
trimmed
|
||||
.chars()
|
||||
.take(24)
|
||||
.any(|ch| ch.is_whitespace() || matches!(ch, '-'))
|
||||
}
|
||||
|
||||
fn should_preserve_overlapping_stream_order(group: &[&TextItem]) -> bool {
|
||||
if group.len() < 3 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(first) = group.iter().find(|item| !item.text.trim().is_empty()) else {
|
||||
return false;
|
||||
};
|
||||
if group.iter().all(|item| item.mcid.is_none()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut nonempty_count = 0;
|
||||
let mut saw_backtrack = false;
|
||||
let mut nonspace_chars = 0;
|
||||
let mut math_symbol_chars = 0;
|
||||
let mut max_font_size = first.font_size;
|
||||
|
||||
for item in group {
|
||||
if !item.text.trim().is_empty() {
|
||||
nonempty_count += 1;
|
||||
}
|
||||
if (item.font_size - first.font_size).abs() > first.font_size * 0.25 {
|
||||
return false;
|
||||
}
|
||||
max_font_size = max_font_size.max(item.font_size);
|
||||
for ch in item.text.chars().filter(|ch| !ch.is_whitespace()) {
|
||||
nonspace_chars += 1;
|
||||
if matches!(
|
||||
ch,
|
||||
'*' | 'ˆ' | '^' | '=' | '+' | '_' | '[' | ']' | '{' | '}' | '|' | '<' | '>'
|
||||
) {
|
||||
math_symbol_chars += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if nonempty_count < 2 {
|
||||
return false;
|
||||
}
|
||||
if nonspace_chars > 0 && math_symbol_chars * 4 > nonspace_chars {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut sorted_by_x = group.to_vec();
|
||||
sorted_by_x.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let cluster_start = sorted_by_x[0].x;
|
||||
let mut cluster_end = cluster_start + effective_merge_width(sorted_by_x[0]);
|
||||
for item in sorted_by_x.iter().skip(1) {
|
||||
let gap = item.x - cluster_end;
|
||||
if gap > max_font_size * 2.5 {
|
||||
return false;
|
||||
}
|
||||
cluster_end = cluster_end.max(item.x + effective_merge_width(item));
|
||||
}
|
||||
if cluster_end - cluster_start > max_font_size * 36.0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
for index in 0..group.len() - 1 {
|
||||
let previous = group[index];
|
||||
let next = group[index + 1];
|
||||
let font_size = previous.font_size.max(next.font_size);
|
||||
let backtrack_threshold = font_size * 0.25;
|
||||
let previous_start = previous.x;
|
||||
let next_start = next.x;
|
||||
let next_end = next.x + effective_merge_width(next);
|
||||
if next_start < previous_start - backtrack_threshold
|
||||
&& next_end > previous_start + backtrack_threshold
|
||||
{
|
||||
let has_near_prefix = group[..=index].iter().rev().take(4).any(|item| {
|
||||
is_short_alpha_fragment(&item.text)
|
||||
&& item.x >= next_start - font_size * 0.5
|
||||
&& item.x <= next_start + font_size * 4.0
|
||||
});
|
||||
let starts_lowercase = first_text_char(&next.text).is_some_and(char::is_lowercase);
|
||||
let phrase_continuation = has_phrase_continuation_shape(&next.text);
|
||||
let has_near_bullet = group[..=index]
|
||||
.iter()
|
||||
.position(|item| {
|
||||
is_standalone_bullet_text(&item.text) && next_start <= item.x + font_size * 3.0
|
||||
})
|
||||
.is_some_and(|bullet_index| {
|
||||
if bullet_index >= index {
|
||||
return false;
|
||||
}
|
||||
group[bullet_index + 1..=index]
|
||||
.iter()
|
||||
.rev()
|
||||
.find(|item| !item.text.trim().is_empty())
|
||||
.is_some_and(|item| {
|
||||
item.text.trim().chars().count() <= 8
|
||||
&& has_phrase_continuation_shape(&next.text)
|
||||
})
|
||||
});
|
||||
if (has_near_prefix && starts_lowercase && phrase_continuation) || has_near_bullet {
|
||||
saw_backtrack = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
saw_backtrack
|
||||
}
|
||||
|
||||
pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
if items.is_empty() {
|
||||
return items;
|
||||
@@ -330,28 +547,32 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
}
|
||||
}
|
||||
|
||||
// Sort each group by X position (direction-aware)
|
||||
for (_, _, group) in &mut line_groups {
|
||||
let mut ordered_line_groups: Vec<(u32, f32, Vec<&TextItem>, bool)> = Vec::new();
|
||||
|
||||
// Sort each group by X position (direction-aware), except for lines whose
|
||||
// content stream intentionally backtracks to overlay ActualText fragments.
|
||||
for (page, y, mut group) in line_groups {
|
||||
let rtl = is_rtl_text(group.iter().map(|i| &i.text));
|
||||
let preserve_stream_order = !rtl && should_preserve_overlapping_stream_order(&group);
|
||||
if rtl {
|
||||
group.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
} else if !preserve_stream_order {
|
||||
group.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
ordered_line_groups.push((page, y, group, preserve_stream_order));
|
||||
}
|
||||
|
||||
// Sort groups by page then Y descending (top of page first)
|
||||
line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
|
||||
ordered_line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
|
||||
|
||||
let mut merged = Vec::new();
|
||||
|
||||
for (_, _, group) in &line_groups {
|
||||
for (_, _, group, preserve_stream_order) in &ordered_line_groups {
|
||||
let mut i = 0;
|
||||
while i < group.len() {
|
||||
let first = group[i];
|
||||
let mut text = first.text.clone();
|
||||
let mut end_x = first.x + effective_merge_width(first);
|
||||
let x_gap_max = first.font_size * 0.5;
|
||||
|
||||
let mut j = i + 1;
|
||||
while j < group.len() {
|
||||
@@ -360,11 +581,29 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
if (next.font_size - first.font_size).abs() > first.font_size * 0.20 {
|
||||
break;
|
||||
}
|
||||
// Never merge across style boundaries: the merged item
|
||||
// carries `first`'s flags, so absorbing a styled run into a
|
||||
// plain neighbor (or vice versa) silently erases the styling
|
||||
// that markdown emission and downstream inline-styling need —
|
||||
// and OR-ing underline instead would stretch `<u>` spans over
|
||||
// neighboring plain text.
|
||||
if next.is_bold != first.is_bold
|
||||
|| next.is_italic != first.is_italic
|
||||
|| next.is_underline != first.is_underline
|
||||
|| next.is_strikeout != first.is_strikeout
|
||||
{
|
||||
break;
|
||||
}
|
||||
let gap = next.x - end_x;
|
||||
let x_gap_max = if *preserve_stream_order && is_standalone_bullet_text(&text) {
|
||||
first.font_size * 1.2
|
||||
} else {
|
||||
first.font_size * 0.5
|
||||
};
|
||||
if gap > x_gap_max {
|
||||
break;
|
||||
}
|
||||
if gap < -first.font_size * 0.5 {
|
||||
if gap < -first.font_size * 0.5 && !preserve_stream_order {
|
||||
break;
|
||||
}
|
||||
// Insert space at word boundaries.
|
||||
@@ -386,11 +625,19 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
first.font_size * 0.08
|
||||
}
|
||||
};
|
||||
if gap > threshold {
|
||||
let needs_bullet_space = *preserve_stream_order
|
||||
&& is_standalone_bullet_text(&text)
|
||||
&& !next.text.trim().is_empty();
|
||||
if needs_bullet_space || gap > threshold {
|
||||
text.push(' ');
|
||||
}
|
||||
text.push_str(&next.text);
|
||||
end_x = next.x + effective_merge_width(next);
|
||||
let next_end = next.x + effective_merge_width(next);
|
||||
end_x = if *preserve_stream_order {
|
||||
end_x.max(next_end)
|
||||
} else {
|
||||
next_end
|
||||
};
|
||||
j += 1;
|
||||
}
|
||||
|
||||
@@ -405,6 +652,8 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
page: first.page,
|
||||
is_bold: first.is_bold,
|
||||
is_italic: first.is_italic,
|
||||
is_underline: first.is_underline,
|
||||
is_strikeout: first.is_strikeout,
|
||||
item_type: first.item_type.clone(),
|
||||
mcid: first.mcid,
|
||||
});
|
||||
@@ -482,12 +731,24 @@ pub(crate) fn merge_subscript_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
.chars()
|
||||
.last()
|
||||
.is_some_and(|c| c.is_alphabetic());
|
||||
if parent.font_size >= sub_threshold && ends_with_letter {
|
||||
let same_marks = parent.is_underline == item.is_underline
|
||||
&& parent.is_strikeout == item.is_strikeout;
|
||||
if parent.font_size >= sub_threshold && ends_with_letter && same_marks {
|
||||
let parent_right = parent.x + parent.width;
|
||||
let gap = item.x - parent_right;
|
||||
// Subscripts must be tightly adjacent (within ~1pt)
|
||||
if gap < parent.font_size * 0.2 && gap > -parent.font_size * 0.3 {
|
||||
parent.text.push_str(&item.text);
|
||||
// Preserve the script when absorbing it: map the
|
||||
// digits to Unicode sub/superscript forms so the
|
||||
// raised/lowered rendering survives in extracted
|
||||
// text ("H"+"2" → "H₂", "word"+"2" → "word²").
|
||||
// NFKC/NFKD normalization folds these back to
|
||||
// plain digits, so text matching downstream is
|
||||
// unaffected. Direction from the baseline offset
|
||||
// (y-up here): raised → superscript (footnote
|
||||
// refs), lowered/level → subscript (chemistry).
|
||||
let raised = item.y > parent.y + parent.font_size * 0.1;
|
||||
parent.text.push_str(&map_script_digits(&item.text, raised));
|
||||
parent.width = (item.x + item.width) - parent.x;
|
||||
continue;
|
||||
}
|
||||
@@ -502,6 +763,21 @@ pub(crate) fn merge_subscript_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
result
|
||||
}
|
||||
|
||||
/// Map ASCII digits to their Unicode superscript (`raised`) or subscript
|
||||
/// forms. Callers guarantee digit-only input (see `merge_subscript_items`);
|
||||
/// anything else passes through unchanged.
|
||||
fn map_script_digits(text: &str, raised: bool) -> String {
|
||||
const SUP: [char; 10] = ['⁰', '¹', '²', '³', '⁴', '⁵', '⁶', '⁷', '⁸', '⁹'];
|
||||
const SUB: [char; 10] = ['₀', '₁', '₂', '₃', '₄', '₅', '₆', '₇', '₈', '₉'];
|
||||
text.chars()
|
||||
.map(|c| match c.to_digit(10) {
|
||||
Some(d) if raised => SUP[d as usize],
|
||||
Some(d) => SUB[d as usize],
|
||||
None => c,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Helper to get f32 from Object
|
||||
pub(crate) fn get_number(obj: &Object) -> Option<f32> {
|
||||
match obj {
|
||||
@@ -515,7 +791,7 @@ pub(crate) fn get_number(obj: &Object) -> Option<f32> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::text_utils::{is_cjk_char, is_rtl_char, is_rtl_text, sort_line_items};
|
||||
use crate::types::{ItemType, TextLine};
|
||||
use crate::types::{ItemType, PdfLine, TextLine};
|
||||
use layout::{detect_columns, is_newspaper_layout, ColumnRegion};
|
||||
|
||||
fn make_merge_item(text: &str, x: f32, width: f32) -> TextItem {
|
||||
@@ -530,11 +806,60 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn with_mcid(mut item: TextItem) -> TextItem {
|
||||
item.mcid = Some(1);
|
||||
item
|
||||
}
|
||||
|
||||
fn make_line(x1: f32, y1: f32, x2: f32, y2: f32) -> PdfLine {
|
||||
PdfLine {
|
||||
x1,
|
||||
y1,
|
||||
x2,
|
||||
y2,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trace_text_preview_truncates_on_char_boundary() {
|
||||
let text = format!("{}{}tail", "a".repeat(79), '\u{FFFD}');
|
||||
let preview = trace_text_preview(&text, 80);
|
||||
|
||||
assert_eq!(preview.chars().count(), 80);
|
||||
assert!(text.is_char_boundary(preview.len()));
|
||||
assert!(preview.ends_with('\u{FFFD}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_breaks_at_style_boundaries() {
|
||||
// A styled run adjacent to plain text must stay a separate item —
|
||||
// merging would erase the flags (italic) or stretch the span
|
||||
// (underline) before markdown emission sees them.
|
||||
let mut italic = make_merge_item("emphasis", 150.0, 40.0);
|
||||
italic.is_italic = true;
|
||||
let mut underlined = make_merge_item("term", 195.0, 20.0);
|
||||
underlined.is_underline = true;
|
||||
let items = vec![
|
||||
make_merge_item("plain lead", 100.0, 48.0),
|
||||
italic,
|
||||
underlined,
|
||||
make_merge_item("plain tail", 218.0, 45.0),
|
||||
];
|
||||
let merged = merge_text_items(items);
|
||||
assert_eq!(merged.len(), 4);
|
||||
assert!(merged[1].is_italic && !merged[1].is_underline);
|
||||
assert!(merged[2].is_underline && !merged[2].is_italic);
|
||||
assert!(!merged[3].is_underline && !merged[3].is_italic);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_no_space_before_period() {
|
||||
// Simulate Tc/Tw-adjusted width: "date" width is smaller than the gap
|
||||
@@ -573,6 +898,158 @@ mod tests {
|
||||
assert_eq!(merged[0].text, "hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_underline_from_later_fragment() {
|
||||
// Fragments with differing underline stay separate items — OR-merging
|
||||
// would stretch the eventual `<u>` span over the plain fragment.
|
||||
// Line-level text assembly still joins them without a space (tight
|
||||
// gap), so the rendered word is unchanged: `pre<u>fix</u>`.
|
||||
let mut items = vec![
|
||||
make_merge_item("pre", 100.0, 18.0),
|
||||
make_merge_item("fix", 119.0, 18.0),
|
||||
];
|
||||
items[1].is_underline = true;
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(merged[0].text, "pre");
|
||||
assert!(!merged[0].is_underline);
|
||||
assert_eq!(merged[1].text, "fix");
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_stream_order_for_backtracking_heading() {
|
||||
// Some tagged PDFs emit first-letter ActualText fragments, then reset
|
||||
// the text matrix and draw the rest of the word from the line start.
|
||||
let items = vec![
|
||||
with_mcid(make_merge_item("F", 79.4, 4.5)),
|
||||
with_mcid(make_merge_item("r", 83.9, 3.3)),
|
||||
with_mcid(make_merge_item("om tables to data-", 79.4, 89.7)),
|
||||
with_mcid(make_merge_item("", 168.9, 33.9)),
|
||||
with_mcid(make_merge_item("analytics-", 168.9, 75.5)),
|
||||
with_mcid(make_merge_item("ready content", 210.5, 60.8)),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(
|
||||
merged[0].text,
|
||||
"From tables to data-analytics-ready content"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_stream_order_for_reset_word_prefix() {
|
||||
let items = vec![
|
||||
with_mcid(make_merge_item("N", 68.0, 7.0)),
|
||||
with_mcid(make_merge_item("e", 75.1, 4.0)),
|
||||
with_mcid(make_merge_item("w fields created", 68.0, 82.0)),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(merged[0].text, "New fields created");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_uses_x_order_for_untagged_backtracking_text() {
|
||||
let items = vec![
|
||||
make_merge_item("N", 68.0, 7.0),
|
||||
make_merge_item("e", 75.1, 4.0),
|
||||
make_merge_item("w fields created", 68.2, 82.0),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
let texts: Vec<_> = merged.iter().map(|item| item.text.as_str()).collect();
|
||||
assert_eq!(texts, vec!["N", "w fields created", "e"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_bullet_stream_order_with_backtracking() {
|
||||
let items = vec![
|
||||
with_mcid(make_merge_item("•", 79.4, 5.0)),
|
||||
with_mcid(make_merge_item("The MS", 91.0, 32.6)),
|
||||
with_mcid(make_merge_item("A LoS project", 84.4, 70.0)),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(merged[0].text, "• The MSA LoS project");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_keeps_normal_bullet_gap_limit_without_stream_order() {
|
||||
let items = vec![
|
||||
make_merge_item("•", 79.4, 5.0),
|
||||
make_merge_item("Distant item", 91.0, 60.0),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
let texts: Vec<_> = merged.iter().map(|item| item.text.as_str()).collect();
|
||||
assert_eq!(texts, vec!["•", "Distant item"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn suppress_table_underlines_clears_line_detected_table_items() {
|
||||
let mut items = vec![
|
||||
make_merge_item("H1", 125.0, 20.0),
|
||||
make_merge_item("H2", 225.0, 20.0),
|
||||
make_merge_item("A", 125.0, 20.0),
|
||||
make_merge_item("B", 225.0, 20.0),
|
||||
];
|
||||
items[0].y = 490.0;
|
||||
items[1].y = 490.0;
|
||||
items[2].y = 470.0;
|
||||
items[3].y = 470.0;
|
||||
for item in &mut items {
|
||||
item.is_underline = true;
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
let lines = vec![
|
||||
make_line(100.0, 500.0, 300.0, 500.0),
|
||||
make_line(100.0, 480.0, 300.0, 480.0),
|
||||
make_line(100.0, 460.0, 300.0, 460.0),
|
||||
make_line(100.0, 460.0, 100.0, 500.0),
|
||||
make_line(200.0, 460.0, 200.0, 500.0),
|
||||
make_line(300.0, 460.0, 300.0, 500.0),
|
||||
];
|
||||
|
||||
suppress_table_underlines(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
assert!(items.iter().all(|item| !item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subscript_digit_with_different_marks_is_not_absorbed() {
|
||||
// A struck-out word followed by an unmarked footnote digit: merging
|
||||
// would widen the parent's strikeout claim over the digit (and the
|
||||
// reverse would drop the digit's own mark). Style boundaries break
|
||||
// the merge, as in merge_text_items.
|
||||
let mut word = make_merge_item("word", 100.0, 24.0);
|
||||
word.font_size = 10.0;
|
||||
word.is_strikeout = true;
|
||||
let mut digit = make_merge_item("2", 124.5, 4.0);
|
||||
digit.font_size = 6.0;
|
||||
digit.y = word.y + 3.0;
|
||||
|
||||
let merged = merge_subscript_items(vec![word.clone(), digit.clone()]);
|
||||
assert_eq!(merged.len(), 2);
|
||||
|
||||
// Same marks still merge (footnote ref inside the strike).
|
||||
digit.is_strikeout = true;
|
||||
let merged = merge_subscript_items(vec![word, digit]);
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert!(merged[0].text.starts_with("word"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_group_into_lines() {
|
||||
let items = vec![
|
||||
@@ -587,6 +1064,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -601,6 +1080,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -615,6 +1096,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -669,6 +1152,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -683,6 +1168,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -697,6 +1184,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -722,6 +1211,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -736,6 +1227,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -750,6 +1243,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -777,6 +1272,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: true,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -811,6 +1308,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -846,6 +1345,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -860,6 +1361,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -874,6 +1377,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -896,6 +1401,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1008,6 +1515,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1022,6 +1531,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1046,6 +1557,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1060,6 +1573,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1100,6 +1615,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}],
|
||||
@@ -1144,6 +1661,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}],
|
||||
@@ -1188,6 +1707,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}],
|
||||
@@ -1225,6 +1746,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1240,7 +1763,8 @@ mod tests {
|
||||
];
|
||||
let merged = merge_subscript_items(items);
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(merged[0].text, "NH3");
|
||||
// Lowered baseline → Unicode subscript form (NFKC folds back to "NH3")
|
||||
assert_eq!(merged[0].text, "NH₃");
|
||||
assert_eq!(merged[1].text, "Cl");
|
||||
}
|
||||
|
||||
@@ -1254,10 +1778,21 @@ mod tests {
|
||||
];
|
||||
let merged = merge_subscript_items(items);
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(merged[0].text, "H2");
|
||||
assert_eq!(merged[0].text, "H₂");
|
||||
assert_eq!(merged[1].text, "O");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_subscript_items_raised_marker_becomes_superscript() {
|
||||
// Footnote reference: "word" followed by a RAISED small "2" → word²
|
||||
let mut marker = make_item_fs("2", 90.0, 502.5, 2.3, 4.7);
|
||||
marker.y = 502.5; // raised above the 499.0 parent baseline
|
||||
let items = vec![make_item_fs("word", 78.0, 499.0, 12.0, 8.0), marker];
|
||||
let merged = merge_subscript_items(items);
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(merged[0].text, "word²");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_subscript_items_no_merge_far_gap() {
|
||||
// Subscript-sized item that's far from the parent should NOT merge
|
||||
|
||||
@@ -0,0 +1,563 @@
|
||||
//! Geometric underline detection.
|
||||
//!
|
||||
//! PDFs have no underline font flag — underlines are drawn as separate
|
||||
//! graphics: stroked horizontal lines (`l`/`S` operators) or thin filled
|
||||
//! rectangles (`re`/`f`). This pass correlates those graphics with text
|
||||
//! items after extraction: an item is underlined when a horizontal
|
||||
//! line/thin rect sits just below its baseline and covers most of its
|
||||
//! horizontal extent.
|
||||
//!
|
||||
//! Repeated same-span rules are treated as table/form rulings rather than
|
||||
//! underlines, which avoids marking every cell in ruled tables.
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::types::{ItemType, PdfRect, TextItem};
|
||||
|
||||
/// Max thickness (pt) for a stroked line / filled rect to count as an
|
||||
/// underline rule rather than a border or decorative band.
|
||||
const MAX_RULE_THICKNESS: f32 = 2.0;
|
||||
|
||||
/// Fraction of the item's width that the rule must cover horizontally.
|
||||
const MIN_X_OVERLAP: f32 = 0.6;
|
||||
|
||||
/// Same-span rules repeated at this many y-levels are usually table/form
|
||||
/// rulings, not semantic underlines.
|
||||
const MIN_REPEATED_RULE_LEVELS: usize = 3;
|
||||
|
||||
/// Vertical tolerance for considering two rules to be on the same row edge.
|
||||
const RULE_Y_DEDUP_EPS: f32 = 2.0;
|
||||
|
||||
/// Horizontal span similarity required when clustering repeated rulings.
|
||||
const RULE_SPAN_OVERLAP_RATIO: f32 = 0.8;
|
||||
const RULE_SPAN_WIDTH_RATIO: f32 = 1.5;
|
||||
|
||||
/// Multiple separated rule segments on one row are usually per-column table
|
||||
/// header/body separators.
|
||||
const MIN_SEGMENTED_ROW_RULES: usize = 3;
|
||||
const MIN_SEGMENTED_ROW_GAPS: usize = 2;
|
||||
const SEGMENTED_ROW_GAP_MIN: f32 = 12.0;
|
||||
|
||||
/// A single rule under several widely separated items is usually a table
|
||||
/// header/body separator, not a sentence underline.
|
||||
const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
pub(crate) y1: f32,
|
||||
pub(crate) x2: f32,
|
||||
pub(crate) y2: f32,
|
||||
pub(crate) stroke_width: f32,
|
||||
pub(crate) page: u32,
|
||||
}
|
||||
|
||||
/// A horizontal rule candidate in page coordinates (PDF y-up).
|
||||
#[derive(Clone)]
|
||||
struct Rule {
|
||||
x1: f32,
|
||||
x2: f32,
|
||||
y: f32,
|
||||
}
|
||||
|
||||
impl Rule {
|
||||
fn width(&self) -> f32 {
|
||||
self.x2 - self.x1
|
||||
}
|
||||
}
|
||||
|
||||
fn rules_from_graphics(rects: &[PdfRect], lines: &[UnderlineLine], page: u32) -> Vec<Rule> {
|
||||
let mut rules: Vec<Rule> = Vec::new();
|
||||
for l in lines {
|
||||
if l.page != page {
|
||||
continue;
|
||||
}
|
||||
// Horizontal stroked line (tolerate slight skew).
|
||||
if l.stroke_width <= MAX_RULE_THICKNESS && (l.y1 - l.y2).abs() <= MAX_RULE_THICKNESS {
|
||||
let (x1, x2) = if l.x1 <= l.x2 {
|
||||
(l.x1, l.x2)
|
||||
} else {
|
||||
(l.x2, l.x1)
|
||||
};
|
||||
if x2 - x1 > 1.0 {
|
||||
rules.push(Rule {
|
||||
x1,
|
||||
x2,
|
||||
y: (l.y1 + l.y2) / 2.0,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
for r in rects {
|
||||
if r.page != page {
|
||||
continue;
|
||||
}
|
||||
// Thin filled rect used as an underline rule. Extents are
|
||||
// normalized first: `re` operands pass through the CTM, so
|
||||
// width/height can be negative (flipped axes / negative scale) —
|
||||
// without normalization negative-width rules are missed and
|
||||
// negative-height bands sneak past the thickness check.
|
||||
let (x1, x2) = if r.width >= 0.0 {
|
||||
(r.x, r.x + r.width)
|
||||
} else {
|
||||
(r.x + r.width, r.x)
|
||||
};
|
||||
if r.height.abs() <= MAX_RULE_THICKNESS && x2 - x1 > 1.0 {
|
||||
rules.push(Rule {
|
||||
x1,
|
||||
x2,
|
||||
y: r.y + r.height / 2.0,
|
||||
});
|
||||
}
|
||||
}
|
||||
rules
|
||||
}
|
||||
|
||||
fn discard_repeated_ruling_rules(rules: Vec<Rule>) -> Vec<Rule> {
|
||||
if rules.len() < MIN_REPEATED_RULE_LEVELS {
|
||||
return rules;
|
||||
}
|
||||
|
||||
rules
|
||||
.iter()
|
||||
.filter(|rule| {
|
||||
!is_repeated_ruling_rule(rule, &rules) && !is_segmented_row_ruling_rule(rule, &rules)
|
||||
})
|
||||
.cloned()
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn is_repeated_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut y_levels: Vec<f32> = rules
|
||||
.iter()
|
||||
.filter(|other| has_similar_span(rule, other))
|
||||
.map(|other| other.y)
|
||||
.collect();
|
||||
|
||||
y_levels.sort_by(|a, b| a.total_cmp(b));
|
||||
y_levels.dedup_by(|a, b| (*a - *b).abs() <= RULE_Y_DEDUP_EPS);
|
||||
y_levels.len() >= MIN_REPEATED_RULE_LEVELS
|
||||
}
|
||||
|
||||
fn is_segmented_row_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut row_rules: Vec<&Rule> = rules
|
||||
.iter()
|
||||
.filter(|other| (other.y - rule.y).abs() <= RULE_Y_DEDUP_EPS)
|
||||
.collect();
|
||||
|
||||
if row_rules.len() < MIN_SEGMENTED_ROW_RULES {
|
||||
return false;
|
||||
}
|
||||
|
||||
row_rules.sort_by(|a, b| a.x1.total_cmp(&b.x1));
|
||||
let large_gaps = row_rules
|
||||
.windows(2)
|
||||
.filter(|pair| pair[1].x1 - pair[0].x2 > SEGMENTED_ROW_GAP_MIN)
|
||||
.count();
|
||||
|
||||
large_gaps >= MIN_SEGMENTED_ROW_GAPS
|
||||
}
|
||||
|
||||
fn has_similar_span(a: &Rule, b: &Rule) -> bool {
|
||||
let a_width = a.width();
|
||||
let b_width = b.width();
|
||||
if a_width <= 1.0 || b_width <= 1.0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let width_ratio = a_width.max(b_width) / a_width.min(b_width);
|
||||
if width_ratio > RULE_SPAN_WIDTH_RATIO {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = a.x2.min(b.x2) - a.x1.max(b.x1);
|
||||
overlap >= a_width.min(b_width) * RULE_SPAN_OVERLAP_RATIO
|
||||
}
|
||||
|
||||
fn tabular_row_separator_rule_indices(rules: &[Rule], items: &[TextItem]) -> HashSet<usize> {
|
||||
let mut tabular_rules = HashSet::new();
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
let mut matched_items: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
|
||||
.collect();
|
||||
|
||||
if matched_items.len() < MIN_TABULAR_RULE_ITEMS {
|
||||
continue;
|
||||
}
|
||||
|
||||
matched_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let large_gaps = matched_items
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let left = pair[0];
|
||||
let right = pair[1];
|
||||
let gap = right.x - (left.x + left.width);
|
||||
let font_size = left.font_size.max(right.font_size).max(1.0);
|
||||
gap > font_size * TABULAR_RULE_GAP_EM
|
||||
})
|
||||
.count();
|
||||
|
||||
if large_gaps >= MIN_TABULAR_RULE_GAPS {
|
||||
tabular_rules.insert(rule_idx);
|
||||
}
|
||||
}
|
||||
|
||||
tabular_rules
|
||||
}
|
||||
|
||||
fn is_underline_candidate(item: &TextItem) -> bool {
|
||||
matches!(item.item_type, ItemType::Text) && !item.text.trim().is_empty() && item.width > 0.0
|
||||
}
|
||||
|
||||
fn rule_matches_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
// Vertical window: underlines sit at or slightly below the baseline.
|
||||
// Fonts draw them at roughly 5-15% of the em below; allow up to 35%
|
||||
// (min 3pt) below and 1pt above for rounding.
|
||||
let below = (item.font_size * 0.35).max(3.0);
|
||||
let y_min = item.y - below;
|
||||
let y_max = item.y + 1.0;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let ix1 = item.x;
|
||||
let ix2 = item.x + item.width;
|
||||
let min_overlap = item.width * MIN_X_OVERLAP;
|
||||
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
/// Strikeout window: a rule crossing the glyphs. Strikethroughs sit at
|
||||
/// roughly 20-35% of the em above the baseline (about half the x-height);
|
||||
/// accept a band well inside the glyph body so baseline underlines and
|
||||
/// overlines never qualify.
|
||||
fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
let y_min = item.y + item.font_size * 0.12;
|
||||
let y_max = item.y + item.font_size * 0.55;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let ix1 = item.x;
|
||||
let ix2 = item.x + item.width;
|
||||
let min_overlap = item.width * MIN_X_OVERLAP;
|
||||
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
/// page's extraction output (all in PDF coordinates, y-up, where
|
||||
/// `TextItem::y` is the text baseline).
|
||||
pub(crate) fn mark_underlined_items(
|
||||
items: &mut [TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) {
|
||||
let rules = discard_repeated_ruling_rules(rules_from_graphics(rects, lines, page));
|
||||
if rules.is_empty() {
|
||||
return;
|
||||
}
|
||||
let tabular_rules = tabular_row_separator_rule_indices(&rules, items);
|
||||
|
||||
for item in items.iter_mut() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
if rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32, font_size: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: font_size,
|
||||
font: "F1".to_string(),
|
||||
font_size,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn hline(x1: f32, x2: f32, y: f32) -> UnderlineLine {
|
||||
UnderlineLine {
|
||||
x1,
|
||||
y1: y,
|
||||
x2,
|
||||
y2: y,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
fn thin_rect(x: f32, y: f32, width: f32) -> PdfRect {
|
||||
PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 0.8,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stroked_line_under_baseline_marks_underline() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thin_filled_rect_under_baseline_marks_underline() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![thin_rect(100.0, 497.8, 60.0)];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_rule_under_multiple_items_marks_each() {
|
||||
// One underline drawn under a whole sentence: every overlapped
|
||||
// item gets the flag.
|
||||
let mut items = vec![
|
||||
item("first", 100.0, 500.0, 40.0, 10.0),
|
||||
item("second", 145.0, 500.0, 50.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(98.0, 200.0, 498.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
assert!(items[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_far_below_baseline_is_not_an_underline() {
|
||||
// A horizontal rule 30pt below (section divider) must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(90.0, 300.0, 470.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_stroked_line_is_not_an_underline() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let mut line = hline(99.0, 161.0, 498.5);
|
||||
line.stroke_width = 4.0;
|
||||
|
||||
mark_underlined_items(&mut items, &[], &[line], 1);
|
||||
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mid_glyph_rule_marks_strikeout_not_underline() {
|
||||
// Rule at ~30% of the em above the baseline crosses the glyphs.
|
||||
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 503.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_strikeout);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn overline_is_neither_underline_nor_strikeout() {
|
||||
// Rule just above the cap height (overline / next line's rule).
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 507.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thin_filled_rect_at_mid_glyph_marks_strikeout() {
|
||||
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![thin_rect(100.0, 502.6, 60.0)];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_strikeout);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_above_baseline_is_not_an_underline() {
|
||||
// Strikethrough / overline geometry must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(90.0, 300.0, 505.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn insufficient_horizontal_overlap_is_not_an_underline() {
|
||||
// Rule under only a quarter of the item (e.g. neighboring cell
|
||||
// border) must not mark.
|
||||
let mut items = vec![item("wide text item", 100.0, 500.0, 100.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 125.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_width_rect_is_normalized_and_marks_underline() {
|
||||
// A CTM with negative x-scale (or negative `re` operands) produces
|
||||
// rects whose width is negative; the rule extents must normalize.
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 160.0,
|
||||
y: 497.8,
|
||||
width: -60.0,
|
||||
height: 0.8,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_height_band_is_not_an_underline() {
|
||||
// A 14pt band expressed with negative height must not pass the
|
||||
// thickness check via sign trickery.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 95.0,
|
||||
y: 509.0,
|
||||
width: 80.0,
|
||||
height: -14.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_band_is_not_an_underline() {
|
||||
// A highlight bar / filled cell background (tall rect) must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 95.0,
|
||||
y: 495.0,
|
||||
width: 80.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vertical_line_is_not_an_underline() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![UnderlineLine {
|
||||
x1: 120.0,
|
||||
y1: 498.0,
|
||||
x2: 120.0,
|
||||
y2: 400.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn other_pages_graphics_do_not_mark() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let mut line = hline(99.0, 161.0, 498.5);
|
||||
line.page = 2;
|
||||
mark_underlined_items(&mut items, &[], &[line], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_table_row_rules_do_not_mark_cell_text() {
|
||||
let mut items = vec![
|
||||
item("A", 110.0, 500.0, 20.0, 10.0),
|
||||
item("B", 110.0, 480.0, 20.0, 10.0),
|
||||
item("C", 110.0, 460.0, 20.0, 10.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(100.0, 150.0, 498.0),
|
||||
hline(100.0, 150.0, 478.0),
|
||||
hline(100.0, 150.0, 458.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn row_separator_under_spaced_column_labels_is_not_an_underline() {
|
||||
let mut items = vec![
|
||||
item("Date", 100.0, 500.0, 25.0, 10.0),
|
||||
item("Rate", 200.0, 500.0, 25.0, 10.0),
|
||||
item("Yield", 300.0, 500.0, 30.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(90.0, 340.0, 498.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_row_spaced_rule_segments_do_not_mark_column_labels() {
|
||||
let mut items = vec![
|
||||
item("Date", 100.0, 500.0, 25.0, 10.0),
|
||||
item("Rate", 200.0, 500.0, 25.0, 10.0),
|
||||
item("Yield", 300.0, 500.0, 30.0, 10.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(98.0, 128.0, 498.0),
|
||||
hline(198.0, 228.0, 498.0),
|
||||
hline(298.0, 333.0, 498.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
}
|
||||
+69
-18
@@ -1,5 +1,6 @@
|
||||
//! Form XObject and image XObject extraction.
|
||||
|
||||
use super::fonts::descriptor_style_flags;
|
||||
use crate::text_utils::{effective_font_size, expand_ligatures, is_bold_font, is_italic_font};
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{ItemType, TextItem};
|
||||
@@ -8,9 +9,9 @@ use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache, FontStyleCache,
|
||||
};
|
||||
use super::{get_number, multiply_matrices};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
const MAX_FORM_XOBJECT_DEPTH: u8 = 5;
|
||||
|
||||
@@ -114,6 +115,7 @@ pub(crate) fn extract_form_xobject_text(
|
||||
font_cmaps: &FontCMaps,
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> Vec<TextItem> {
|
||||
extract_form_xobject_text_inner(
|
||||
doc,
|
||||
@@ -122,10 +124,12 @@ pub(crate) fn extract_form_xobject_text(
|
||||
font_cmaps,
|
||||
parent_ctm,
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
0,
|
||||
)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn extract_form_xobject_text_inner(
|
||||
doc: &Document,
|
||||
form_id: ObjectId,
|
||||
@@ -133,6 +137,7 @@ fn extract_form_xobject_text_inner(
|
||||
font_cmaps: &FontCMaps,
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
depth: u8,
|
||||
) -> Vec<TextItem> {
|
||||
use lopdf::content::Content;
|
||||
@@ -167,6 +172,7 @@ fn extract_form_xobject_text_inner(
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
let mut inline_cmaps: HashMap<String, crate::tounicode::CMapEntry> = HashMap::new();
|
||||
|
||||
let mut font_style_flags: HashMap<String, (bool, bool)> = HashMap::new();
|
||||
for (font_name, font_dict) in &form_fonts {
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
if let Ok(base_font) = font_dict.get(b"BaseFont") {
|
||||
@@ -175,6 +181,10 @@ fn extract_form_xobject_text_inner(
|
||||
font_base_names.insert(resource_name.clone(), base_name);
|
||||
}
|
||||
}
|
||||
let style = descriptor_style_flags(doc, font_dict, style_cache);
|
||||
if style != (false, false) {
|
||||
font_style_flags.insert(resource_name.clone(), style);
|
||||
}
|
||||
match font_dict.get(b"ToUnicode") {
|
||||
Ok(tounicode) => {
|
||||
if let Ok(obj_ref) = tounicode.as_reference() {
|
||||
@@ -262,19 +272,46 @@ fn extract_form_xobject_text_inner(
|
||||
if !op.operands.is_empty() {
|
||||
if let Ok(name) = op.operands[0].as_name() {
|
||||
let xobj_name = String::from_utf8_lossy(name).to_string();
|
||||
if let Some(XObjectType::Form(nested_id)) = form_xobjects.get(&xobj_name) {
|
||||
if depth < MAX_FORM_XOBJECT_DEPTH {
|
||||
let nested_items = extract_form_xobject_text_inner(
|
||||
doc,
|
||||
*nested_id,
|
||||
page_num,
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
cmap_decisions,
|
||||
depth + 1,
|
||||
);
|
||||
items.extend(nested_items);
|
||||
match form_xobjects.get(&xobj_name) {
|
||||
Some(XObjectType::Form(nested_id)) => {
|
||||
if depth < MAX_FORM_XOBJECT_DEPTH {
|
||||
let nested_items = extract_form_xobject_text_inner(
|
||||
doc,
|
||||
*nested_id,
|
||||
page_num,
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
depth + 1,
|
||||
);
|
||||
items.extend(nested_items);
|
||||
}
|
||||
}
|
||||
Some(XObjectType::Image) => {
|
||||
// Mirror the top-level Image-XObject emission
|
||||
// in content_stream.rs so figures embedded
|
||||
// inside Form XObjects (common in print-to-PDF
|
||||
// workflows) aren't silently dropped.
|
||||
let (x, y, width, height) = image_bbox_from_ctm(&ctm);
|
||||
items.push(TextItem {
|
||||
text: format!("[Image: {}]", xobj_name),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height,
|
||||
font: String::new(),
|
||||
font_size: 0.0,
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Image,
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
None => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -373,6 +410,7 @@ fn extract_form_xobject_text_inner(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
@@ -402,6 +440,10 @@ fn extract_form_xobject_text_inner(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
@@ -411,8 +453,10 @@ fn extract_form_xobject_text_inner(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -517,6 +561,7 @@ fn extract_form_xobject_text_inner(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
current_text.push_str(&text);
|
||||
}
|
||||
@@ -532,6 +577,10 @@ fn extract_form_xobject_text_inner(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
|
||||
for (text, start_w, end_w) in &sub_items {
|
||||
let offset_tm = [
|
||||
@@ -558,8 +607,10 @@ fn extract_form_xobject_text_inner(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
+5879
-85
File diff suppressed because it is too large
Load Diff
@@ -8,6 +8,24 @@ use log::debug;
|
||||
/// Font statistics for a document
|
||||
pub(crate) struct FontStats {
|
||||
pub(crate) most_common_size: f32,
|
||||
/// Font size frequency distribution (size_key → line count).
|
||||
/// Used for rarity-based heading detection.
|
||||
pub(crate) size_counts: HashMap<i32, usize>,
|
||||
/// Total number of lines counted.
|
||||
pub(crate) total_lines: usize,
|
||||
}
|
||||
|
||||
/// Compute how rare a font size is in the document (0.0 = most common, 1.0 = unique).
|
||||
/// Mirrors opendataloader's font rarity boosting approach: heading fonts appear on
|
||||
/// far fewer lines than body text, so their percentile rank is high.
|
||||
pub(crate) fn font_size_rarity(font_size: f32, stats: &FontStats) -> f32 {
|
||||
if stats.total_lines == 0 {
|
||||
return 0.0;
|
||||
}
|
||||
let key = (font_size * 10.0) as i32;
|
||||
let count = stats.size_counts.get(&key).copied().unwrap_or(0);
|
||||
// Rarity = 1 - (frequency ratio). A size used on 1/100 lines has rarity ~0.99.
|
||||
1.0 - (count as f32 / stats.total_lines as f32)
|
||||
}
|
||||
|
||||
/// Calculate font stats directly from items (before grouping into lines)
|
||||
@@ -21,6 +39,8 @@ pub(crate) fn calculate_font_stats_from_items(items: &[TextItem]) -> FontStats {
|
||||
}
|
||||
}
|
||||
|
||||
let total_lines = size_counts.values().sum();
|
||||
|
||||
// Break ties by preferring the smaller font size for deterministic output
|
||||
let most_common_size = size_counts
|
||||
.iter()
|
||||
@@ -30,7 +50,11 @@ pub(crate) fn calculate_font_stats_from_items(items: &[TextItem]) -> FontStats {
|
||||
.map(|(size, _)| *size as f32 / 10.0)
|
||||
.unwrap_or(12.0);
|
||||
|
||||
FontStats { most_common_size }
|
||||
FontStats {
|
||||
most_common_size,
|
||||
size_counts,
|
||||
total_lines,
|
||||
}
|
||||
}
|
||||
|
||||
/// Calculate font stats from grouped lines
|
||||
@@ -48,6 +72,8 @@ pub(crate) fn calculate_font_stats(lines: &[TextLine]) -> FontStats {
|
||||
}
|
||||
}
|
||||
|
||||
let total_lines = size_counts.values().sum();
|
||||
|
||||
// Break ties by preferring the smaller font size for deterministic output
|
||||
let most_common_size = size_counts
|
||||
.iter()
|
||||
@@ -57,7 +83,23 @@ pub(crate) fn calculate_font_stats(lines: &[TextLine]) -> FontStats {
|
||||
.map(|(size, _)| *size as f32 / 10.0)
|
||||
.unwrap_or(12.0);
|
||||
|
||||
FontStats { most_common_size }
|
||||
FontStats {
|
||||
most_common_size,
|
||||
size_counts,
|
||||
total_lines,
|
||||
}
|
||||
}
|
||||
|
||||
/// Determine the heading level for a bold-only line that didn't meet the font-size
|
||||
/// threshold. These are common in academic papers where section headings are bold
|
||||
/// at the same size as body text.
|
||||
///
|
||||
/// Returns a level below the lowest font-size tier (or H2 when no tiers exist).
|
||||
pub(crate) fn bold_heading_level(heading_tiers: &[f32]) -> usize {
|
||||
let level = heading_tiers.len() + 1;
|
||||
// Clamp to 1..=6 — if no font-size tiers, bold headings become H2
|
||||
// (H1 is reserved for titles which are typically larger)
|
||||
level.clamp(2, 6)
|
||||
}
|
||||
|
||||
/// Detect TOC-style lines that contain dot leaders (e.g., "Section Name .... 42").
|
||||
@@ -88,6 +130,29 @@ pub(crate) fn has_dot_leaders(text: &str) -> bool {
|
||||
dot_groups >= 2
|
||||
}
|
||||
|
||||
/// Detect a table-of-contents entry: a line ending in a page number preceded by
|
||||
/// a dot-leader group (e.g. "Measurement Lab worksheet ... 3"). `has_dot_leaders`
|
||||
/// misses single-group leaders ("..."), but a trailing "<dots> <number>" is a
|
||||
/// strong TOC signal on its own. Such lines must never be promoted to headings.
|
||||
pub(crate) fn is_toc_entry_line(text: &str) -> bool {
|
||||
let trimmed = text.trim_end();
|
||||
let digits = trimmed
|
||||
.chars()
|
||||
.rev()
|
||||
.take_while(|c| c.is_ascii_digit())
|
||||
.count();
|
||||
if digits == 0 || digits > 4 {
|
||||
return false;
|
||||
}
|
||||
let before_number = trimmed[..trimmed.len() - digits].trim_end();
|
||||
let dots = before_number
|
||||
.chars()
|
||||
.rev()
|
||||
.take_while(|c| *c == '.')
|
||||
.count();
|
||||
dots >= 3
|
||||
}
|
||||
|
||||
/// Compute the Y-gap threshold for paragraph break detection.
|
||||
///
|
||||
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
|
||||
@@ -278,3 +343,28 @@ pub(crate) fn detect_header_level(
|
||||
Some(4)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn toc_entry_with_single_dot_group() {
|
||||
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
|
||||
assert!(is_toc_entry_line("Results ........ 12"));
|
||||
assert!(is_toc_entry_line("Appendix B...42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_toc_lines_pass() {
|
||||
assert!(!is_toc_entry_line(
|
||||
"6.2. Expectations for Re-Hiring Employees"
|
||||
));
|
||||
assert!(!is_toc_entry_line("What happened in 2020"));
|
||||
assert!(!is_toc_entry_line("IMPLEMENTATION"));
|
||||
// Ellipsis without a trailing page number
|
||||
assert!(!is_toc_entry_line("and so it goes ..."));
|
||||
// Long numbers are data, not page refs
|
||||
assert!(!is_toc_entry_line("ISBN ... 97814"));
|
||||
}
|
||||
}
|
||||
|
||||
+102
-9
@@ -4,13 +4,11 @@
|
||||
pub(crate) fn is_caption_line(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
|
||||
// Common caption prefixes in multiple languages
|
||||
let caption_prefixes = [
|
||||
"Figure ",
|
||||
// Caption prefixes that always match (always followed by identifiers)
|
||||
let always_prefixes = [
|
||||
"Figura ",
|
||||
"Fig. ",
|
||||
"Fig ",
|
||||
"Table ",
|
||||
"Tabela ",
|
||||
"Source:",
|
||||
"Fonte:",
|
||||
@@ -27,23 +25,61 @@ pub(crate) fn is_caption_line(text: &str) -> bool {
|
||||
"Photo ",
|
||||
"Foto ",
|
||||
];
|
||||
|
||||
// Check if line starts with a caption prefix
|
||||
for prefix in &caption_prefixes {
|
||||
for prefix in &always_prefixes {
|
||||
if trimmed.starts_with(prefix) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check case-insensitive patterns
|
||||
// "Figure" and "Table" need a digit/reference after them to distinguish
|
||||
// captions ("Table 1", "Figure 3.2") from headings ("Table of Contents")
|
||||
for prefix in ["Figure ", "Table "] {
|
||||
if let Some(rest) = trimmed.strip_prefix(prefix) {
|
||||
if rest
|
||||
.trim_start()
|
||||
.starts_with(|c: char| c.is_ascii_digit() || c == '(' || c == '#')
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check case-insensitive patterns — require digit or punctuation after
|
||||
// prefix to avoid matching "Table of Contents" or "Figure drawing" etc.
|
||||
let lower = trimmed.to_lowercase();
|
||||
if lower.starts_with("figure ") || lower.starts_with("table ") || lower.starts_with("source:") {
|
||||
for pfx in ["figure ", "table "] {
|
||||
if let Some(rest) = lower.strip_prefix(pfx) {
|
||||
if rest
|
||||
.trim_start()
|
||||
.starts_with(|c: char| c.is_ascii_digit() || c == '(' || c == '#')
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if lower.starts_with("source:") {
|
||||
return true;
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if text starts with an unambiguous bullet marker (●, •, ○, ◦).
|
||||
///
|
||||
/// Narrower than [`is_list_item`]: it excludes numbered/lettered patterns
|
||||
/// like `1.` or `a)`, which legitimately appear as section headings in many
|
||||
/// documents. Used by the heading classifier to reject bullet lines without
|
||||
/// also demoting numbered headings.
|
||||
pub(crate) fn starts_with_bullet_marker(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
trimmed.starts_with("• ")
|
||||
|| trimmed.starts_with("● ")
|
||||
|| trimmed.starts_with("○ ")
|
||||
|| trimmed.starts_with("◦ ")
|
||||
|| trimmed.starts_with("- ")
|
||||
|| trimmed.starts_with("* ")
|
||||
}
|
||||
|
||||
/// Check if text looks like a list item
|
||||
pub(crate) fn is_list_item(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
@@ -95,6 +131,17 @@ pub(crate) fn format_list_item(text: &str) -> String {
|
||||
if let Some(rest) = trimmed.strip_prefix(*bullet) {
|
||||
return format!("- {}", rest.trim_start());
|
||||
}
|
||||
// Bullet inside a leading style run (e.g. "**● Label:** rest" or
|
||||
// "<u>● Label</u>"). The run wraps both the marker and the following
|
||||
// label because both carry the style in the PDF. The marker must move
|
||||
// outside the wrapper so markdown still sees a list item.
|
||||
for wrapper in ["**", "*", "<u>"] {
|
||||
if let Some(after_open) = trimmed.strip_prefix(wrapper) {
|
||||
if let Some(rest) = after_open.strip_prefix(*bullet) {
|
||||
return format!("- {}{}", wrapper, rest.trim_start());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if trimmed.starts_with("- ") || trimmed.starts_with("* ") {
|
||||
@@ -178,3 +225,49 @@ pub(crate) fn is_monospace_font(font_name: &str) -> bool {
|
||||
|
||||
patterns.iter().any(|p| lower.contains(p))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn format_list_item_plain_bullet() {
|
||||
assert_eq!(format_list_item("● Item"), "- Item");
|
||||
assert_eq!(format_list_item("• Item"), "- Item");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_underline() {
|
||||
// Fully-underlined bullet line: the marker must move outside the
|
||||
// <u> wrapper so markdown still renders a list item.
|
||||
assert_eq!(format_list_item("<u>● Item text</u>"), "- <u>Item text</u>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_bold() {
|
||||
// PDF that uses bold font for both the marker and the label produces
|
||||
// a single bold run like "**● Label:** rest"; the bullet must still
|
||||
// be stripped and the bold wrapper preserved on the label.
|
||||
assert_eq!(
|
||||
format_list_item("**● Fraud: Willing cooperation;**"),
|
||||
"- **Fraud: Willing cooperation;**"
|
||||
);
|
||||
assert_eq!(
|
||||
format_list_item("**● Label:** rest of line"),
|
||||
"- **Label:** rest of line"
|
||||
);
|
||||
assert_eq!(format_list_item("*● Italic:* rest"), "- *Italic:* rest");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_already_dash() {
|
||||
assert_eq!(format_list_item("- existing"), "- existing");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_list_item_with_bullet_space() {
|
||||
assert!(is_list_item("● Item"));
|
||||
assert!(is_list_item("• Item"));
|
||||
assert!(is_list_item("- Item"));
|
||||
}
|
||||
}
|
||||
|
||||
+759
-19
File diff suppressed because it is too large
Load Diff
+85
-6
@@ -400,6 +400,8 @@ pub struct MarkdownOptions {
|
||||
pub detect_bold: bool,
|
||||
/// Detect and format italic text from font names
|
||||
pub detect_italic: bool,
|
||||
/// Emit `<u>` runs for text with a geometrically-detected underline
|
||||
pub detect_underline: bool,
|
||||
/// Include image placeholders in output
|
||||
pub include_images: bool,
|
||||
/// Include extracted hyperlinks
|
||||
@@ -422,7 +424,17 @@ impl Default for MarkdownOptions {
|
||||
fix_hyphenation: true,
|
||||
detect_bold: true,
|
||||
detect_italic: true,
|
||||
include_images: true,
|
||||
detect_underline: true,
|
||||
// `include_images: false` is intentional. The content-stream walker
|
||||
// now emits `ItemType::Image` `TextItem`s for every Image XObject
|
||||
// it encounters (see `extractor/content_stream.rs`). If we rendered
|
||||
// those into markdown by default, every existing caller would
|
||||
// suddenly see `` placeholders inserted
|
||||
// throughout their output — a silent regression for anyone who
|
||||
// upgrades. Image bboxes are still available via
|
||||
// `extract_text_with_positions` for callers (e.g. layout-aware
|
||||
// pipelines) that want to crop + caption figures themselves.
|
||||
include_images: false,
|
||||
include_links: true,
|
||||
include_page_numbers: false,
|
||||
strip_headers_footers: true,
|
||||
@@ -601,6 +613,15 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
let group = page_groups.get(&page).unwrap();
|
||||
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
|
||||
|
||||
// Detect columns early — on multi-column pages, the merged-band retry
|
||||
// should skip body-font heuristic table detection (which mistakes column
|
||||
// text for tables). Individual band heuristic detection is left enabled
|
||||
// because bands are scoped to single columns.
|
||||
let page_has_columns = {
|
||||
let cols = crate::extractor::detect_columns(&page_items, page, false);
|
||||
cols.len() >= 2
|
||||
};
|
||||
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// Fallback: use rect hint regions to detect side-by-side layout
|
||||
@@ -873,10 +894,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
run_heuristic(&unclaimed_items, &unclaimed_map, 6);
|
||||
}
|
||||
|
||||
// 4. Column-based table detection: last resort for borderless tabular
|
||||
// layouts (e.g. exam/reference grids) when ALL structural methods
|
||||
// found nothing. Only runs when no rects/lines exist (truly borderless)
|
||||
// and no other detection method found tables in this band.
|
||||
// 4. Column-based table detection for borderless tabular layouts.
|
||||
let band_has_tables = band_items.iter().enumerate().any(|(idx, _)| {
|
||||
band_index_map
|
||||
.get(idx)
|
||||
@@ -903,6 +921,65 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Thin-rect border synthesis: last resort for PDFs that draw table
|
||||
// borders as thin filled rectangles (common in spreadsheet exports).
|
||||
// Only runs when ALL other methods found nothing on this page.
|
||||
if !page_tables.contains_key(&page) {
|
||||
let page_rects: Vec<&crate::types::PdfRect> =
|
||||
rects.iter().filter(|r| r.page == page).collect();
|
||||
let mut synth_lines: Vec<crate::types::PdfLine> = Vec::new();
|
||||
for r in &page_rects {
|
||||
let (mut w, mut h) = (r.width, r.height);
|
||||
let (mut x, mut y) = (r.x, r.y);
|
||||
if w < 0.0 {
|
||||
x += w;
|
||||
w = -w;
|
||||
}
|
||||
if h < 0.0 {
|
||||
y += h;
|
||||
h = -h;
|
||||
}
|
||||
if h < 2.0 && w >= 10.0 {
|
||||
let mid_y = y + h / 2.0;
|
||||
synth_lines.push(crate::types::PdfLine {
|
||||
x1: x,
|
||||
y1: mid_y,
|
||||
x2: x + w,
|
||||
y2: mid_y,
|
||||
page,
|
||||
});
|
||||
} else if w < 2.0 && h >= 10.0 {
|
||||
let mid_x = x + w / 2.0;
|
||||
synth_lines.push(crate::types::PdfLine {
|
||||
x1: mid_x,
|
||||
y1: y,
|
||||
x2: mid_x,
|
||||
y2: y + h,
|
||||
page,
|
||||
});
|
||||
}
|
||||
}
|
||||
if synth_lines.len() >= 10 {
|
||||
let page_text: Vec<TextItem> = text_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
let line_tables = detect_tables_from_lines(&page_text, &synth_lines, page);
|
||||
for table in &line_tables {
|
||||
for &idx in &table.item_indices {
|
||||
table_items.insert(idx);
|
||||
}
|
||||
let table_y = table.rows.first().copied().unwrap_or(0.0);
|
||||
let table_md = table_to_markdown(table);
|
||||
page_tables
|
||||
.entry(page)
|
||||
.or_default()
|
||||
.push((table_y, table_md));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Merged-band retry: if we split into bands but found no tables in
|
||||
// any band, retry heuristic detection with all items as a single band.
|
||||
// This catches borderless tables whose text-column alignment was
|
||||
@@ -915,7 +992,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
band_items.len(),
|
||||
was_split
|
||||
);
|
||||
let heuristic_tables = detect_tables(band_items, base_size, false);
|
||||
let heuristic_tables = detect_tables(band_items, base_size, page_has_columns);
|
||||
for table in &heuristic_tables {
|
||||
for &idx in &table.item_indices {
|
||||
if let Some(&page_idx) = band_index_map.get(idx) {
|
||||
@@ -1143,6 +1220,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
@@ -29,6 +29,8 @@ pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> Str
|
||||
// text item, which combine with gap-based space insertion to produce
|
||||
// double spaces ("Vice President" instead of "Vice President").
|
||||
collapse_consecutive_spaces(&mut text);
|
||||
remove_spaces_before_closing_brackets(&mut text);
|
||||
remove_spaces_before_sentence_punctuation(&mut text);
|
||||
|
||||
// Remove excessive newlines (more than 2 in a row)
|
||||
while text.contains("\n\n\n") {
|
||||
@@ -71,6 +73,46 @@ fn collapse_consecutive_spaces(text: &mut String) {
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Remove spaces before closing square brackets.
|
||||
/// Unit markers and markdown links occasionally pick up a gap-inserted space
|
||||
/// before `]` (e.g. `[kg/m3 ]`), which is cosmetic padding.
|
||||
fn remove_spaces_before_closing_brackets(text: &mut String) {
|
||||
let mut result = String::with_capacity(text.len());
|
||||
for ch in text.chars() {
|
||||
if ch == ']' && result.ends_with(' ') {
|
||||
result.pop();
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Remove a stray space before sentence punctuation ("word ." → "word.").
|
||||
/// Style-boundary item splits (bold/italic/underline runs) can strand a
|
||||
/// trailing period or comma in its own fragment, and several assembly paths
|
||||
/// join fragments with spaces. Only fires when the punctuation ends the
|
||||
/// token (followed by whitespace or end of text), so decimals ("3 .14" stays
|
||||
/// untouched — no such input exists, but the guard is cheap) and dot leaders
|
||||
/// (" ... ") are unaffected.
|
||||
fn remove_spaces_before_sentence_punctuation(text: &mut String) {
|
||||
let chars: Vec<char> = text.chars().collect();
|
||||
let mut result = String::with_capacity(text.len());
|
||||
for (i, &ch) in chars.iter().enumerate() {
|
||||
if matches!(ch, '.' | ',' | ';') && result.ends_with(' ') {
|
||||
let next = chars.get(i + 1);
|
||||
// `|` counts as a token end so table cells get the same fix.
|
||||
let token_ends = next.is_none_or(|c| c.is_whitespace() || *c == '|');
|
||||
// Never touch runs of dots (ellipsis / dot leaders).
|
||||
let in_dot_run = ch == '.' && next == Some(&'.');
|
||||
if token_ends && !in_dot_run {
|
||||
result.pop();
|
||||
}
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Collapse dot leaders (runs of 4+ dots) into " ... "
|
||||
/// Common in tables of contents: "Introduction...............................1" -> "Introduction ... 1"
|
||||
fn collapse_dot_leaders(text: &str) -> String {
|
||||
@@ -342,6 +384,48 @@ mod tests {
|
||||
assert!(result.contains("Chapter 2 ... 20"));
|
||||
}
|
||||
|
||||
// --- remove_spaces_before_closing_brackets ---
|
||||
|
||||
#[test]
|
||||
fn test_remove_spaces_before_closing_brackets() {
|
||||
let mut input = "Density [kg/m3 ] and [linked text ](https://example.com)".to_string();
|
||||
remove_spaces_before_closing_brackets(&mut input);
|
||||
assert_eq!(
|
||||
input,
|
||||
"Density [kg/m3] and [linked text](https://example.com)"
|
||||
);
|
||||
}
|
||||
|
||||
// --- remove_spaces_before_sentence_punctuation ---
|
||||
|
||||
#[test]
|
||||
fn strips_space_before_trailing_period() {
|
||||
let mut t = "Foreign insurance companies . The provisions".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "Foreign insurance companies. The provisions");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_space_before_period_at_cell_boundary() {
|
||||
let mut t = "|Applicability date .|This section|".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "|Applicability date.|This section|");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_dot_leaders_and_ellipses() {
|
||||
let mut t = "Introduction ... 1".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "Introduction ... 1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_mid_token_periods() {
|
||||
let mut t = "version 3 .14 released".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "version 3 .14 released");
|
||||
}
|
||||
|
||||
// --- fix_hyphenation ---
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -542,6 +542,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
|
||||
+202
-13
@@ -30,6 +30,9 @@ pub struct PyPdfResult {
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
/// Title from PDF metadata.
|
||||
#[pyo3(get)]
|
||||
pub title: Option<String>,
|
||||
@@ -60,6 +63,28 @@ impl PyPdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reasons for a single 1-indexed page.
|
||||
#[pyclass(name = "PageOcrReasons")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageOcrReasons {
|
||||
/// 1-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Machine-readable OCR reason identifiers.
|
||||
#[pyo3(get)]
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageOcrReasons {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageOcrReasons(page={}, reasons={:?})",
|
||||
self.page, self.reasons
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classification wrapper (lightweight)
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -106,6 +131,9 @@ pub struct PyRegionText {
|
||||
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -146,6 +174,73 @@ impl PyPageRegionTexts {
|
||||
// Text item wrapper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Per-page markdown extraction result.
|
||||
#[pyclass(name = "PageMarkdown")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageMarkdown {
|
||||
/// 0-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Formatted markdown for this page.
|
||||
#[pyo3(get)]
|
||||
pub markdown: String,
|
||||
/// True when text on this page is unreliable (GID-encoded fonts,
|
||||
/// encoding issues, garbage text, or empty extraction).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageMarkdown {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageMarkdown(page={}, markdown='{}', needs_ocr={})",
|
||||
self.page,
|
||||
self.markdown.chars().take(40).collect::<String>(),
|
||||
self.needs_ocr
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// Combined per-page markdown extraction and layout classification result.
|
||||
#[pyclass(name = "PagesExtractionResult")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPagesExtractionResult {
|
||||
/// Per-page markdown results, in the order requested.
|
||||
#[pyo3(get)]
|
||||
pub pages: Vec<PyPageMarkdown>,
|
||||
/// 1-indexed pages where tables were detected.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// 1-indexed pages where multi-column layout was detected.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// 1-indexed pages that need OCR (scanned/image-based or unreliable text).
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
/// True if any page has tables or columns.
|
||||
#[pyo3(get)]
|
||||
pub is_complex: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPagesExtractionResult {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PagesExtractionResult(pages={}, pages_with_tables={:?}, is_complex={})",
|
||||
self.pages.len(),
|
||||
self.pages_with_tables,
|
||||
self.is_complex
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// A positioned text item extracted from a PDF.
|
||||
#[pyclass(name = "TextItem")]
|
||||
#[derive(Clone)]
|
||||
@@ -171,6 +266,10 @@ pub struct PyTextItem {
|
||||
#[pyo3(get)]
|
||||
pub is_italic: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_underline: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_strikeout: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
}
|
||||
|
||||
@@ -207,6 +306,7 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
title: r.title,
|
||||
confidence: r.confidence,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
@@ -216,6 +316,16 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_page_ocr_reasons(reasons: Vec<crate::PageOcrReasons>) -> Vec<PyPageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PyPageOcrReasons {
|
||||
page: reason.page,
|
||||
reasons: reason.reasons,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_py_err(e: crate::PdfError) -> PyErr {
|
||||
PyValueError::new_err(e.to_string())
|
||||
}
|
||||
@@ -243,30 +353,65 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn parse_page_regions(page_regions: Vec<(u32, Vec<Vec<f64>>)>) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||
fn parse_page_regions(
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<(u32, Vec<[f32; 4]>)>> {
|
||||
page_regions
|
||||
.into_iter()
|
||||
.map(|(page, regions)| {
|
||||
let bboxes: Vec<[f32; 4]> = regions
|
||||
.iter()
|
||||
.map(|r| {
|
||||
if r.len() != 4 {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
} else {
|
||||
[r[0] as f32, r[1] as f32, r[2] as f32, r[3] as f32]
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(page, bboxes)
|
||||
let mut bboxes: Vec<[f32; 4]> = Vec::with_capacity(regions.len());
|
||||
for (idx, region) in regions.into_iter().enumerate() {
|
||||
if region.len() != 4 {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"Invalid region at page {page}, index {idx}: expected [x1, y1, x2, y2], got {} values",
|
||||
region.len()
|
||||
)));
|
||||
}
|
||||
let [x1, y1, x2, y2] = [region[0], region[1], region[2], region[3]];
|
||||
if !(x1.is_finite() && y1.is_finite() && x2.is_finite() && y2.is_finite()) {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"Invalid region at page {page}, index {idx}: coordinates must be finite numbers"
|
||||
)));
|
||||
}
|
||||
if x2 < x1 || y2 < y1 {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"Invalid region at page {page}, index {idx}: expected x2>=x1 and y2>=y1, got [{x1}, {y1}, {x2}, {y2}]"
|
||||
)));
|
||||
}
|
||||
bboxes.push([x1 as f32, y1 as f32, x2 as f32, y2 as f32]);
|
||||
}
|
||||
Ok((page, bboxes))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_py_pages_result(r: crate::PagesExtractionResult) -> PyPagesExtractionResult {
|
||||
PyPagesExtractionResult {
|
||||
pages: r
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|p| PyPageMarkdown {
|
||||
page: p.page,
|
||||
markdown: p.markdown,
|
||||
needs_ocr: p.needs_ocr,
|
||||
ocr_reason: p.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: r.pages_with_tables,
|
||||
pages_with_columns: r.pages_with_columns,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
is_complex: r.is_complex,
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRegionTexts> {
|
||||
results
|
||||
.into_iter()
|
||||
@@ -278,6 +423,7 @@ fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRe
|
||||
.map(|r| PyRegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
@@ -424,19 +570,60 @@ fn extract_text_in_regions_bytes(
|
||||
data: &[u8],
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<PyPageRegionTexts>> {
|
||||
let regions = parse_page_regions(page_regions);
|
||||
let regions = parse_page_regions(page_regions)?;
|
||||
let results = crate::extract_text_in_regions_mem(data, ®ions).map_err(to_py_err)?;
|
||||
Ok(convert_region_results(results))
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF file, with layout
|
||||
/// classification metadata.
|
||||
///
|
||||
/// Returns per-page markdown and classification data (tables, columns,
|
||||
/// OCR needs) from a single parse. Font statistics are computed from the
|
||||
/// full document so header detection is consistent across pages.
|
||||
///
|
||||
/// Args:
|
||||
/// path: Path to the PDF file.
|
||||
/// pages: Optional list of 0-indexed pages. When None (default), every
|
||||
/// page is returned in document order. When provided, output
|
||||
/// matches the caller-supplied order.
|
||||
///
|
||||
/// Returns:
|
||||
/// PagesExtractionResult with per-page markdown and classification data.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn extract_pages_markdown(
|
||||
path: &str,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<PyPagesExtractionResult> {
|
||||
let result = crate::extract_pages_markdown(path, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(to_py_pages_result(result))
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF from bytes.
|
||||
///
|
||||
/// See [`extract_pages_markdown`] for details.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn extract_pages_markdown_bytes(
|
||||
data: &[u8],
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<PyPagesExtractionResult> {
|
||||
let result = crate::extract_pages_markdown_mem(data, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(to_py_pages_result(result))
|
||||
}
|
||||
|
||||
/// Python module definition.
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPdfResult>()?;
|
||||
m.add_class::<PyPageOcrReasons>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
m.add_class::<PyPageRegionTexts>()?;
|
||||
m.add_class::<PyPageMarkdown>()?;
|
||||
m.add_class::<PyPagesExtractionResult>()?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
||||
@@ -449,5 +636,7 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_pages_markdown_bytes, m)?)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
+573
-67
@@ -104,6 +104,8 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
page: first_item.page,
|
||||
is_bold: first_item.is_bold,
|
||||
is_italic: first_item.is_italic,
|
||||
is_underline: first_item.is_underline,
|
||||
is_strikeout: first_item.is_strikeout,
|
||||
item_type: first_item.item_type.clone(),
|
||||
mcid: first_item.mcid,
|
||||
});
|
||||
@@ -219,7 +221,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
body_font_low,
|
||||
body_font_high,
|
||||
);
|
||||
if body_candidates.len() >= 9 {
|
||||
if body_candidates.len() >= 6 {
|
||||
let regions = find_table_regions_strict(&body_candidates);
|
||||
log::debug!("body-font: {} strict regions found", regions.len());
|
||||
|
||||
@@ -241,7 +243,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
body_candidates.len()
|
||||
);
|
||||
|
||||
if region_items.len() < 9 {
|
||||
if region_items.len() < 6 {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -581,8 +583,14 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
// Validation 1: some rows should have content in first column.
|
||||
// Use a lower threshold (25%) for tables with wrapped cells where
|
||||
// continuation lines leave the first column empty.
|
||||
// Skip when cells form a narrow TOC pattern: hierarchical entries indented
|
||||
// across multiple X levels leave the leftmost column sparse (only top-level
|
||||
// chapters land there) but the structure is still a valid TOC. Narrow only
|
||||
// (<=5 cols) — wide multi-column TOCs (e.g. 2-up indices) would render
|
||||
// poorly through format_toc_as_list, which assumes one entry per row.
|
||||
let rows_with_first_col = cells.iter().filter(|row| !row[0].is_empty()).count();
|
||||
if rows_with_first_col < rows.len() / 4 {
|
||||
let is_narrow_toc = columns.len() <= 5 && is_table_of_contents(&cells);
|
||||
if rows_with_first_col < rows.len() / 4 && !is_narrow_toc {
|
||||
log::debug!(
|
||||
" validation 1 fail: {}/{} rows have first col",
|
||||
rows_with_first_col,
|
||||
@@ -653,15 +661,24 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
return None;
|
||||
}
|
||||
|
||||
// Validation 8: Check for Table of Contents pattern
|
||||
if is_table_of_contents(&cells) {
|
||||
log::debug!(" validation 8 fail: table of contents");
|
||||
// Validation 8: Reject paragraph-like content falsely detected as tables.
|
||||
// TOC pages with deep indentation (top-level chapters in col 0, subsections
|
||||
// in cols 1-3, page numbers in last col) leave most cells empty and trip
|
||||
// the paragraph heuristic; TOC shape is a safer signal here. Narrow only
|
||||
// — see narrow-TOC rationale at validation 1.
|
||||
if is_paragraph_content(&cells) && !is_narrow_toc {
|
||||
log::debug!(" validation 9 fail: paragraph content");
|
||||
return None;
|
||||
}
|
||||
|
||||
// Validation 9: Reject paragraph-like content falsely detected as tables
|
||||
if is_paragraph_content(&cells) {
|
||||
log::debug!(" validation 9 fail: paragraph content");
|
||||
// Validation 9: Reject wide "index" layouts where every cell carries a
|
||||
// full "label ... page" fragment (back-of-book IRS-style indices).
|
||||
// These render poorly in any structured form; text flow is the best
|
||||
// fallback. Narrow dot-leader TOCs (2-3 cols) are kept so format.rs
|
||||
// can emit them as a per-row flat list with titles tab-joined to page
|
||||
// numbers.
|
||||
if is_inline_leader_index(&cells) {
|
||||
log::debug!(" validation 9 fail: inline-leader index");
|
||||
return None;
|
||||
}
|
||||
|
||||
@@ -672,12 +689,7 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
item_indices.len()
|
||||
);
|
||||
|
||||
Some(Table {
|
||||
columns,
|
||||
rows,
|
||||
cells,
|
||||
item_indices,
|
||||
})
|
||||
Some(Table::new(columns, rows, cells, item_indices))
|
||||
}
|
||||
|
||||
/// Check if this looks like a key-value pair layout rather than a table
|
||||
@@ -801,7 +813,25 @@ fn has_table_like_content(cells: &[Vec<String>], mode: TableDetectionMode) -> bo
|
||||
// Bypass content check for wide tables (3+ columns) — text-only tables
|
||||
// (category lists, program descriptions) are legitimate if they passed
|
||||
// all structural validations (alignment, consistency, not key-value).
|
||||
pct_data > min_pct || num_cols >= 3
|
||||
// Also bypass for 2-column body-font tables with short cells (avg ≤40 chars),
|
||||
// which are likely definition/category lists, not paragraph text.
|
||||
if pct_data > min_pct || num_cols >= 3 {
|
||||
return true;
|
||||
}
|
||||
if num_cols == 2 && matches!(mode, TableDetectionMode::BodyFont) {
|
||||
let non_empty: Vec<usize> = cells
|
||||
.iter()
|
||||
.skip(1)
|
||||
.flat_map(|row| row.iter())
|
||||
.filter(|c| !c.trim().is_empty())
|
||||
.map(|c| c.trim().len())
|
||||
.collect();
|
||||
if !non_empty.is_empty() {
|
||||
let avg_len = non_empty.iter().sum::<usize>() / non_empty.len();
|
||||
return avg_len <= 25;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if a cell value looks like table data
|
||||
@@ -880,77 +910,247 @@ fn looks_like_number(s: &str) -> bool {
|
||||
&& s.chars().any(|c| c.is_ascii_digit())
|
||||
}
|
||||
|
||||
/// Check if this looks like a Table of Contents
|
||||
/// TOCs have characteristic patterns: leader dots, page numbers, section names
|
||||
fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
|
||||
/// Check if this looks like a Table of Contents (either style).
|
||||
///
|
||||
/// Used by format.rs to render TOCs as flat lists instead of markdown tables.
|
||||
pub fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells)
|
||||
}
|
||||
|
||||
/// Dot-leader TOC: any "Chapter 1 ........ 42" style with explicit leader
|
||||
/// dots. Covers both narrow 2-3 col TOCs (where the leader is a dedicated
|
||||
/// cell) and wide indices (where each cell encodes a full "label ... page"
|
||||
/// fragment). Used by format.rs to render as a flat list.
|
||||
pub(super) fn is_dot_leader_toc(cells: &[Vec<String>]) -> bool {
|
||||
has_structural_dot_leader(cells) || is_inline_leader_index(cells)
|
||||
}
|
||||
|
||||
/// Rows with a dedicated dots-only cell flanked by label + number (2-3 col
|
||||
/// TOC layout). Format.rs handles these well via per-row flat-list
|
||||
/// rendering; they should NOT be rejected at detect time.
|
||||
fn has_structural_dot_leader(cells: &[Vec<String>]) -> bool {
|
||||
if cells.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let structural_rows = cells.iter().filter(|row| row_has_dot_leader(row)).count();
|
||||
structural_rows as f32 / cells.len() as f32 >= 0.3
|
||||
}
|
||||
|
||||
let num_cols = cells[0].len();
|
||||
let mut dot_cells = 0;
|
||||
let mut page_number_cells = 0;
|
||||
let mut total_cells = 0;
|
||||
// Track which columns contain dots vs numbers to distinguish
|
||||
// TOC (dots span middle, page number at end) from data tables
|
||||
// (dots only in label column, many number columns).
|
||||
let mut dot_cols = vec![0u32; num_cols];
|
||||
let mut numeric_cols = vec![0u32; num_cols];
|
||||
|
||||
/// Wide index layout: each cell holds a full "label ... page" fragment
|
||||
/// because the column detector kept multi-column indices as single cells.
|
||||
/// These render poorly both as markdown tables (column boundaries are
|
||||
/// arbitrary) and as flat lists (each row holds 3+ separate index
|
||||
/// entries). Reject these at detect time so they fall back to the page's
|
||||
/// normal text flow.
|
||||
pub(super) fn is_inline_leader_index(cells: &[Vec<String>]) -> bool {
|
||||
let mut inline_cells = 0;
|
||||
let mut total_nonempty = 0;
|
||||
for row in cells {
|
||||
for (ci, cell) in row.iter().enumerate() {
|
||||
for cell in row {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.is_empty() {
|
||||
continue;
|
||||
}
|
||||
total_cells += 1;
|
||||
|
||||
// Check for leader dots (sequences of periods)
|
||||
// TOCs often have "........" or ". . . ." patterns
|
||||
let dot_count = trimmed.chars().filter(|&c| c == '.').count();
|
||||
let is_mostly_dots = dot_count > trimmed.len() / 2 && dot_count >= 3;
|
||||
if is_mostly_dots {
|
||||
dot_cells += 1;
|
||||
if ci < num_cols {
|
||||
dot_cols[ci] += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for standalone page numbers (1-4 digits, possibly with spaces)
|
||||
let digits_only: String = trimmed.chars().filter(|c| !c.is_whitespace()).collect();
|
||||
if digits_only.len() <= 4
|
||||
&& !digits_only.is_empty()
|
||||
&& digits_only.chars().all(|c| c.is_ascii_digit())
|
||||
{
|
||||
page_number_cells += 1;
|
||||
if ci < num_cols {
|
||||
numeric_cols[ci] += 1;
|
||||
}
|
||||
total_nonempty += 1;
|
||||
if cell_is_inline_leader(trimmed) {
|
||||
inline_cells += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
total_nonempty >= 4 && inline_cells as f32 / total_nonempty as f32 >= 0.25
|
||||
}
|
||||
|
||||
if total_cells == 0 {
|
||||
/// A row with a dot-leader. Accepts two layouts:
|
||||
/// 1. A dedicated dots-only cell ("....") with a text label somewhere
|
||||
/// to its left and a page number somewhere to its right.
|
||||
/// 2. A "title ... " cell (trailing leader dots glued to the title)
|
||||
/// with a page number elsewhere in the same row.
|
||||
fn row_has_dot_leader(row: &[String]) -> bool {
|
||||
let has_page_number = row.iter().any(|c| row_cell_is_page_number(c));
|
||||
|
||||
for (ci, cell) in row.iter().enumerate() {
|
||||
let trimmed = cell.trim();
|
||||
|
||||
// Pattern 1: dedicated dots-only cell.
|
||||
let dot_count = trimmed.chars().filter(|&c| c == '.').count();
|
||||
let is_mostly_dots = dot_count >= 3
|
||||
&& dot_count > trimmed.len() / 2
|
||||
&& trimmed.chars().all(|c| c == '.' || c.is_whitespace());
|
||||
if is_mostly_dots {
|
||||
let has_label_left = row[..ci].iter().any(|c| {
|
||||
let t = c.trim();
|
||||
!t.is_empty() && t.chars().any(|ch| ch.is_alphabetic())
|
||||
});
|
||||
if has_label_left && has_page_number {
|
||||
return true;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Pattern 2: cell ends with a trailing " ... " run after a label.
|
||||
if has_page_number && cell_has_trailing_leader(trimmed) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Cell ends with a run of ≥3 dots preceded by alphabetic text and a
|
||||
/// space — the "Title ... " layout where the leader is glued to the name.
|
||||
/// Alphabetic (not alphanumeric) so that data-table row labels like
|
||||
/// "1973 ... " do not register as titles.
|
||||
fn cell_has_trailing_leader(cell: &str) -> bool {
|
||||
let trimmed = cell.trim_end();
|
||||
if !trimmed.ends_with('.') {
|
||||
return false;
|
||||
}
|
||||
let without_dots = trimmed.trim_end_matches('.');
|
||||
let dot_run = trimmed.len() - without_dots.len();
|
||||
if dot_run < 3 {
|
||||
return false;
|
||||
}
|
||||
// Require a space before the dot run (rules out "etc..." / "Mr...") and
|
||||
// at least one alphabetic char (rules out "1973 ... " data-row labels).
|
||||
without_dots.ends_with(' ') && without_dots.trim().chars().any(|c| c.is_alphabetic())
|
||||
}
|
||||
|
||||
/// Page-number shape: single ≤4-digit integer, a ", "-separated list of
|
||||
/// ≤4-digit integers ("18, 36, 107"), or a dashed section-page ID
|
||||
/// ("A-1", "5-21"). Rejects decimal cells ("4. 0"), thousands-separated
|
||||
/// values ("189,164"), and other long numeric data that appears in
|
||||
/// statistical tables.
|
||||
fn row_cell_is_page_number(cell: &str) -> bool {
|
||||
let t = cell.trim();
|
||||
if t.is_empty() {
|
||||
return false;
|
||||
}
|
||||
if looks_like_section_page_id(t) {
|
||||
return true;
|
||||
}
|
||||
// Page list: ", " separator (with space) distinguishes real page lists
|
||||
// from thousands-separated numbers like "189,164".
|
||||
let parts: Vec<&str> = t.split(", ").collect();
|
||||
parts
|
||||
.iter()
|
||||
.all(|p| !p.is_empty() && p.len() <= 4 && p.chars().all(|c| c.is_ascii_digit()))
|
||||
}
|
||||
|
||||
/// A cell shaped like an index leader fragment. Accepts two forms:
|
||||
/// - "text ... number" — label + dots + page number in one cell
|
||||
/// - "... number" — bare leader + number (row where the label
|
||||
/// landed in a separate column)
|
||||
///
|
||||
/// Both only count if followed by pure numeric content (optionally
|
||||
/// comma-separated page lists like "127, 213").
|
||||
fn cell_is_inline_leader(cell: &str) -> bool {
|
||||
let cell = cell.trim();
|
||||
|
||||
// Find the first "..." run. Surrounding-whitespace checks below
|
||||
// reject intra-word ellipses ("etc...").
|
||||
let idx = match cell.match_indices("...").next() {
|
||||
Some((i, _)) => i,
|
||||
None => return false,
|
||||
};
|
||||
|
||||
let before = &cell[..idx];
|
||||
let after_dots = &cell[idx + 3..];
|
||||
// Allow extra dots (e.g. "....") by skipping any additional '.'
|
||||
let after = after_dots.trim_start_matches('.');
|
||||
|
||||
// Require space (or start-of-cell) before the dots and space/digit
|
||||
// after — blocks intra-word ellipses.
|
||||
let before_ok = before.is_empty() || before.ends_with(' ');
|
||||
let after_ok = after.starts_with(' ') || after.is_empty();
|
||||
if !before_ok || !after_ok {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Data tables with dot leaders (e.g. "1973....") have dots concentrated
|
||||
// in one column (the label column) while many other columns contain numbers.
|
||||
// True TOCs have dots spanning the middle and one page-number column at the end.
|
||||
// If dots are confined to ≤1 column AND there are ≥3 columns with numbers,
|
||||
// this is a data table, not a TOC.
|
||||
let cols_with_dots = dot_cols.iter().filter(|&&c| c >= 2).count();
|
||||
let cols_with_numbers = numeric_cols.iter().filter(|&&c| c >= 2).count();
|
||||
if cols_with_dots <= 1 && cols_with_numbers >= 3 {
|
||||
let after_trim = after.trim();
|
||||
if after_trim.is_empty() {
|
||||
return false;
|
||||
}
|
||||
// Tail must be purely numeric/page-list content.
|
||||
let tail_numeric = after_trim
|
||||
.chars()
|
||||
.all(|c| c.is_ascii_digit() || matches!(c, ',' | ' ' | '.' | '-' | '$'))
|
||||
&& after_trim.chars().any(|c| c.is_ascii_digit());
|
||||
if !tail_numeric {
|
||||
return false;
|
||||
}
|
||||
|
||||
// If a significant portion of cells are dots or page numbers, it's likely a TOC
|
||||
let dot_ratio = dot_cells as f32 / total_cells as f32;
|
||||
let page_num_ratio = page_number_cells as f32 / total_cells as f32;
|
||||
// Either we have a label before, or the leader is bare (starts the cell)
|
||||
// — both are legitimate index fragments.
|
||||
before.chars().any(|c| c.is_alphabetic()) || before.trim().is_empty()
|
||||
}
|
||||
|
||||
// TOC typically has >15% dot cells and >10% page number cells
|
||||
dot_ratio > 0.15 || (dot_ratio > 0.05 && page_num_ratio > 0.15)
|
||||
/// Dot-less tabular TOC: tagged PDFs emit entries as rows where the first
|
||||
/// column starts with a dotted section number (e.g. "4.3.1 Something") and
|
||||
/// the last column is one or more page numbers. These have no leader dots
|
||||
/// and benefit from flat-list formatting (page numbers aligned to titles).
|
||||
pub(super) fn is_tabular_toc(cells: &[Vec<String>]) -> bool {
|
||||
if cells.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let num_cols = cells[0].len();
|
||||
if num_cols < 2 || cells.len() < 4 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let section_rows = cells
|
||||
.iter()
|
||||
.filter(|row| {
|
||||
row.iter()
|
||||
.find(|c| !c.trim().is_empty())
|
||||
.is_some_and(|c| starts_with_section_number(c.trim()))
|
||||
})
|
||||
.count();
|
||||
|
||||
let last_col = num_cols - 1;
|
||||
let (last_filled, last_page_num) = cells.iter().fold((0u32, 0u32), |(f, n), row| {
|
||||
let cell = row.get(last_col).map(|s| s.trim()).unwrap_or("");
|
||||
if cell.is_empty() {
|
||||
return (f, n);
|
||||
}
|
||||
let is_page_nums = cell
|
||||
.split_whitespace()
|
||||
.all(|tok| !tok.is_empty() && tok.chars().all(|c| c.is_ascii_digit()));
|
||||
(f + 1, n + if is_page_nums { 1 } else { 0 })
|
||||
});
|
||||
|
||||
let section_ratio = section_rows as f32 / cells.len() as f32;
|
||||
let page_num_last_ratio = if last_filled > 0 {
|
||||
last_page_num as f32 / last_filled as f32
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
|
||||
section_ratio >= 0.6 && last_filled >= 3 && page_num_last_ratio >= 0.7
|
||||
}
|
||||
|
||||
/// Matches dashed section-page identifiers used in technical manuals:
|
||||
/// "5-21", "A-1", "B--3", "TC-2". At least one ASCII digit is required.
|
||||
fn looks_like_section_page_id(s: &str) -> bool {
|
||||
let ok = s
|
||||
.chars()
|
||||
.all(|c| c.is_ascii_digit() || c.is_ascii_uppercase() || c == '-');
|
||||
ok && s.chars().any(|c| c.is_ascii_digit())
|
||||
}
|
||||
|
||||
/// Returns true when the leading token looks like a dotted section number:
|
||||
/// "1", "1.2", "1.2.3", "4.3.1.2" — integer components joined by dots,
|
||||
/// with at least one dot (single-number prefixes are too ambiguous).
|
||||
fn starts_with_section_number(s: &str) -> bool {
|
||||
let Some(first) = s.split_whitespace().next() else {
|
||||
return false;
|
||||
};
|
||||
let first = first.trim_end_matches('.');
|
||||
let parts: Vec<&str> = first.split('.').collect();
|
||||
if parts.len() < 2 || parts.len() > 6 {
|
||||
return false;
|
||||
}
|
||||
parts
|
||||
.iter()
|
||||
.all(|p| !p.is_empty() && p.len() <= 3 && p.chars().all(|c| c.is_ascii_digit()))
|
||||
}
|
||||
|
||||
/// Check if detected "table" cells are actually paragraph text fragments.
|
||||
@@ -1126,6 +1326,33 @@ pub(crate) fn find_first_table_row(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip rows that have duplicate non-empty cells. These are spanning
|
||||
// super-headers (e.g., "First Degree | First Degree | Higher Degree")
|
||||
// that sit above the real column header row. Using them as the markdown
|
||||
// header produces duplicate column names that downstream validation
|
||||
// rejects. Only skip if a subsequent row looks like a better header
|
||||
// (denser fill or has data).
|
||||
if filled_count >= 2 && !has_data {
|
||||
let mut text_counts: std::collections::HashMap<&str, usize> =
|
||||
std::collections::HashMap::new();
|
||||
for cell in &filled_cells {
|
||||
*text_counts.entry(cell.trim()).or_insert(0) += 1;
|
||||
}
|
||||
let has_duplicates = text_counts.values().any(|&count| count >= 2);
|
||||
if has_duplicates {
|
||||
// Check if a later row is a better header candidate
|
||||
let has_better_below = cells.iter().skip(row_idx + 1).take(3).any(|r| {
|
||||
let next_filled = r.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
let next_fill = next_filled as f32 / total_cols as f32;
|
||||
let next_numeric = r.iter().filter(|c| looks_like_number(c.trim())).count();
|
||||
next_fill >= 0.4 || next_numeric >= 2
|
||||
});
|
||||
if has_better_below {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Data rows are definitely table content
|
||||
if has_data {
|
||||
first_table_row = row_idx;
|
||||
@@ -1380,4 +1607,283 @@ mod tests {
|
||||
"data table with dot-leader labels should not be rejected as TOC"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_accepts_hierarchical_indented_toc() {
|
||||
// Mythos system card pages 4-5: top-level chapters indent at col 0,
|
||||
// subsections at cols 1-2, leaving col 0 mostly empty (only ~10% of
|
||||
// rows). Validation 1 was rejecting these even though the structure
|
||||
// is unambiguously a TOC.
|
||||
let cells = vec![
|
||||
vec!["Abstract".to_string(), String::new(), "3".to_string()],
|
||||
vec![
|
||||
"1 Introduction".to_string(),
|
||||
String::new(),
|
||||
"10".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.1 Model training".to_string(),
|
||||
"11".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.1.1 Training data".to_string(),
|
||||
"11".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.1.2 Crowd workers".to_string(),
|
||||
"12".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.2 Release decision".to_string(),
|
||||
"13".to_string(),
|
||||
],
|
||||
vec![
|
||||
"2 RSP evaluations".to_string(),
|
||||
String::new(),
|
||||
"16".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"2.1 RSP risk assessment".to_string(),
|
||||
"16".to_string(),
|
||||
],
|
||||
vec![String::new(), "2.1.1 Context".to_string(), "16".to_string()],
|
||||
vec![
|
||||
String::new(),
|
||||
"2.2 CB evaluations".to_string(),
|
||||
"20".to_string(),
|
||||
],
|
||||
];
|
||||
assert!(
|
||||
is_table_of_contents(&cells),
|
||||
"hierarchical TOC with sparse col 0 should still be detected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_rejects_dotless_toc() {
|
||||
// Tabular TOC without leader dots: first column starts with dotted
|
||||
// section numbers, last column is page numbers. Pattern from
|
||||
// Mythos system card pages 6-8.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"4.3 Case studies and targeted evaluations".to_string(),
|
||||
String::new(),
|
||||
"86".to_string(),
|
||||
],
|
||||
vec![
|
||||
"4.3.1 Destructive or reckless actions".to_string(),
|
||||
"4.3.1.1 Synthetic-backend evaluation".to_string(),
|
||||
"86 86".to_string(),
|
||||
],
|
||||
vec![
|
||||
"4.3.2 Adherence to constitution".to_string(),
|
||||
"4.3.2.1 Overview".to_string(),
|
||||
"89 89".to_string(),
|
||||
],
|
||||
vec![
|
||||
"4.3.3 Honesty and hallucinations".to_string(),
|
||||
"4.3.3.1 Factual hallucinations".to_string(),
|
||||
"93 94".to_string(),
|
||||
],
|
||||
vec![
|
||||
"4.4 Capability evaluations".to_string(),
|
||||
String::new(),
|
||||
"101".to_string(),
|
||||
],
|
||||
];
|
||||
assert!(
|
||||
is_table_of_contents(&cells),
|
||||
"dot-less TOC with section numbers + page numbers should be rejected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dot_leader_toc_accepts_short_inline_leaders() {
|
||||
// Index-style cells where the full "label ... number" pattern is
|
||||
// preserved in a single cell (IRS Publication 17 back-of-book index).
|
||||
let cells = vec![
|
||||
vec!["Child tax credit ... 235".to_string(), String::new()],
|
||||
vec!["Church employee ... 252".to_string(), String::new()],
|
||||
vec!["Citizens outside the U.S ... 6".to_string(), String::new()],
|
||||
vec![
|
||||
"Claim for refund ... 18, 36, 107".to_string(),
|
||||
String::new(),
|
||||
],
|
||||
vec!["Clergy ... 7, 52".to_string(), String::new()],
|
||||
];
|
||||
assert!(is_dot_leader_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dot_leader_toc_allows_ellipsis_data_table() {
|
||||
// Data tables using "..." as a row-omission marker must not be
|
||||
// mistaken for dot-leader TOCs. Based on MCF5235RM QSPI RAM layout.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"0x00".to_string(),
|
||||
"QTR0".to_string(),
|
||||
"Transmit RAM".to_string(),
|
||||
],
|
||||
vec!["0x01".to_string(), "QTR1".to_string(), String::new()],
|
||||
vec![
|
||||
"...".to_string(),
|
||||
"...".to_string(),
|
||||
"16 bits wide".to_string(),
|
||||
],
|
||||
vec!["0x0F".to_string(), "QTR15".to_string(), String::new()],
|
||||
vec![
|
||||
"0x10".to_string(),
|
||||
"QRR0".to_string(),
|
||||
"Receive RAM".to_string(),
|
||||
],
|
||||
vec!["0x11".to_string(), "QRR1".to_string(), String::new()],
|
||||
vec![
|
||||
"...".to_string(),
|
||||
"...".to_string(),
|
||||
"16 bits wide".to_string(),
|
||||
],
|
||||
vec!["0x1F".to_string(), "QRR15".to_string(), String::new()],
|
||||
];
|
||||
assert!(
|
||||
!is_dot_leader_toc(&cells),
|
||||
"ellipsis markers in a data table should not match TOC detection"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dot_leader_toc_rejects_year_row_data_table() {
|
||||
// ERP-2025 economic data tables: year labels with trailing " ... ",
|
||||
// a final " ... " column, and decimal-looking numeric cells. The
|
||||
// detection previously classified these as dot-leader TOCs and
|
||||
// routed them through flat-list formatting, destroying the grid.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"1973 ... ".to_string(),
|
||||
"4. 0".to_string(),
|
||||
"1. 8".to_string(),
|
||||
"0. 4".to_string(),
|
||||
"3. 2".to_string(),
|
||||
" ... ".to_string(),
|
||||
],
|
||||
vec![
|
||||
"1974 ... ".to_string(),
|
||||
"–1. 9".to_string(),
|
||||
"–1. 6".to_string(),
|
||||
"–5. 6".to_string(),
|
||||
"2. 4".to_string(),
|
||||
" ... ".to_string(),
|
||||
],
|
||||
vec![
|
||||
"1975 ... ".to_string(),
|
||||
"2. 6".to_string(),
|
||||
"5. 1".to_string(),
|
||||
"6. 1".to_string(),
|
||||
"4. 1".to_string(),
|
||||
" ... ".to_string(),
|
||||
],
|
||||
vec![
|
||||
"1976 ... ".to_string(),
|
||||
"4. 3".to_string(),
|
||||
"5. 4".to_string(),
|
||||
"6. 4".to_string(),
|
||||
"4. 5".to_string(),
|
||||
" ... ".to_string(),
|
||||
],
|
||||
];
|
||||
assert!(
|
||||
!is_dot_leader_toc(&cells),
|
||||
"year-indexed data tables with decimal cells must not match TOC detection"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dot_leader_toc_rejects_monthly_data_table() {
|
||||
// ERP-2025 Table B-22: monthly labor-force rows with "Jan ... ",
|
||||
// "Feb ... " labels and thousands-separated cells ("189,164").
|
||||
// Previously matched TOC detection because "Jan ..." has alphabetic
|
||||
// text and "189,164" passed the page-number shape check.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"2023: Jan ... ".to_string(),
|
||||
"265,962".to_string(),
|
||||
"165,871".to_string(),
|
||||
"160,152".to_string(),
|
||||
"62. 4".to_string(),
|
||||
],
|
||||
vec![
|
||||
"Feb ... ".to_string(),
|
||||
"266,112".to_string(),
|
||||
"166,263".to_string(),
|
||||
"160,301".to_string(),
|
||||
"62. 5".to_string(),
|
||||
],
|
||||
vec![
|
||||
"Mar ... ".to_string(),
|
||||
"266,272".to_string(),
|
||||
"166,690".to_string(),
|
||||
"160,824".to_string(),
|
||||
"62. 6".to_string(),
|
||||
],
|
||||
vec![
|
||||
"Apr ... ".to_string(),
|
||||
"266,443".to_string(),
|
||||
"166,678".to_string(),
|
||||
"160,962".to_string(),
|
||||
"62. 6".to_string(),
|
||||
],
|
||||
];
|
||||
assert!(
|
||||
!is_dot_leader_toc(&cells),
|
||||
"monthly labor-force rows with thousands-separated data must not match TOC detection"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tabular_toc_requires_section_numbers_and_pages() {
|
||||
// Dot-less tabular TOC matches is_tabular_toc but not dot-leader.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"4.3 Case studies".to_string(),
|
||||
String::new(),
|
||||
"86".to_string(),
|
||||
],
|
||||
vec![
|
||||
"4.3.1 Destructive actions".to_string(),
|
||||
String::new(),
|
||||
"86".to_string(),
|
||||
],
|
||||
vec![
|
||||
"4.3.2 Adherence".to_string(),
|
||||
String::new(),
|
||||
"89".to_string(),
|
||||
],
|
||||
vec!["4.3.3 Honesty".to_string(), String::new(), "93".to_string()],
|
||||
];
|
||||
assert!(is_tabular_toc(&cells));
|
||||
assert!(!is_dot_leader_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn starts_with_section_number_matches_dotted() {
|
||||
assert!(starts_with_section_number("1.2"));
|
||||
assert!(starts_with_section_number("4.3.1"));
|
||||
assert!(starts_with_section_number("4.3.1.2"));
|
||||
assert!(starts_with_section_number("4.3 Case studies"));
|
||||
assert!(starts_with_section_number("2.2.5.1 Expert red teaming"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn starts_with_section_number_rejects_non_sections() {
|
||||
assert!(!starts_with_section_number("Chapter 1"));
|
||||
assert!(!starts_with_section_number("1973"));
|
||||
assert!(!starts_with_section_number("1.5M"));
|
||||
assert!(!starts_with_section_number("10.0%"));
|
||||
assert!(!starts_with_section_number(""));
|
||||
assert!(!starts_with_section_number("Hello world"));
|
||||
}
|
||||
}
|
||||
|
||||
+272
-36
@@ -4,11 +4,74 @@
|
||||
//! gridlines. Many IRS forms and government PDFs use these instead of
|
||||
//! `re` (rectangle) operators.
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::tables::Table;
|
||||
use crate::types::{PdfLine, TextItem};
|
||||
|
||||
use super::detect_rects::{assign_items_to_grid, snap_edges};
|
||||
|
||||
/// Derive column edges from the x-endpoints of horizontal-rule
|
||||
/// segments when no vertical lines were drawn.
|
||||
///
|
||||
/// Catalog and archival-finding-aid tables are commonly drawn with
|
||||
/// per-row horizontal rules broken into N segments (one segment per
|
||||
/// cell), with no vertical dividers at all. The segment break points
|
||||
/// (e.g. `[50, 127], [127, 485], [485, 562]` per row) implicitly
|
||||
/// encode the column boundaries.
|
||||
///
|
||||
/// Returns column edges if ≥3 distinct x-positions each show up as a
|
||||
/// segment endpoint on ≥50% of the unique horizontal-line rows.
|
||||
/// Returns `None` otherwise — decorative rules with varying widths
|
||||
/// shouldn't be mistaken for a table.
|
||||
fn derive_columns_from_horizontal_segments(horizontals: &[(f32, f32, f32)]) -> Option<Vec<f32>> {
|
||||
if horizontals.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut endpoints: Vec<f32> = Vec::with_capacity(horizontals.len() * 2);
|
||||
for &(_, x_min, x_max) in horizontals {
|
||||
endpoints.push(x_min);
|
||||
endpoints.push(x_max);
|
||||
}
|
||||
let clusters = snap_edges(&endpoints, 5.0);
|
||||
if clusters.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Bucket y-values to count unique rows. Tolerance ~0.1pt (×10
|
||||
// rounding) tolerates the snap_edges 3pt clustering used later
|
||||
// for row edges.
|
||||
let unique_rows: HashSet<i32> = horizontals
|
||||
.iter()
|
||||
.map(|&(y, _, _)| (y * 10.0).round() as i32)
|
||||
.collect();
|
||||
if unique_rows.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let min_rows = (unique_rows.len() as f32 * 0.5).ceil() as usize;
|
||||
|
||||
let qualifying: Vec<f32> = clusters
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&cluster_x| {
|
||||
let rows_touched: HashSet<i32> = horizontals
|
||||
.iter()
|
||||
.filter(|&&(_, x_min, x_max)| {
|
||||
(x_min - cluster_x).abs() < 5.0 || (x_max - cluster_x).abs() < 5.0
|
||||
})
|
||||
.map(|&(y, _, _)| (y * 10.0).round() as i32)
|
||||
.collect();
|
||||
rows_touched.len() >= min_rows
|
||||
})
|
||||
.collect();
|
||||
|
||||
if qualifying.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
Some(qualifying)
|
||||
}
|
||||
|
||||
/// Detect tables from line segments on a given page.
|
||||
///
|
||||
/// Lines are classified as horizontal or vertical, snapped into grid edges,
|
||||
@@ -52,25 +115,50 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
// Diagonal lines are ignored
|
||||
}
|
||||
|
||||
if horizontals.len() < 3 || verticals.len() < 2 {
|
||||
if horizontals.len() < 3 {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// If no/very-few vertical lines are drawn, try to derive column edges
|
||||
// from the x-endpoints of the horizontal-rule segments. Catalog and
|
||||
// archival-finding-aid layouts commonly draw each row's horizontal
|
||||
// rule as N segments (one per cell), with no vertical dividers at
|
||||
// all — the segment break points encode the column boundaries.
|
||||
let implicit_col_edges: Option<Vec<f32>> = if verticals.len() < 2 {
|
||||
derive_columns_from_horizontal_segments(&horizontals)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if verticals.len() < 2 && implicit_col_edges.is_none() {
|
||||
return Vec::new();
|
||||
}
|
||||
let cols_from_segments = implicit_col_edges.is_some();
|
||||
|
||||
log::debug!(
|
||||
"detect_lines p{}: {} horiz, {} vert lines (of {} total on page)",
|
||||
"detect_lines p{}: {} horiz, {} vert lines (of {} total on page){}",
|
||||
page,
|
||||
horizontals.len(),
|
||||
verticals.len(),
|
||||
page_lines.len()
|
||||
page_lines.len(),
|
||||
if cols_from_segments {
|
||||
" — columns from horizontal segments"
|
||||
} else {
|
||||
""
|
||||
}
|
||||
);
|
||||
|
||||
// Snap Y-values of horizontal lines → row edges
|
||||
let h_ys: Vec<f32> = horizontals.iter().map(|(y, _, _)| *y).collect();
|
||||
let row_edges = snap_edges(&h_ys, 3.0);
|
||||
|
||||
// Snap X-values of vertical lines → column edges
|
||||
let v_xs: Vec<f32> = verticals.iter().map(|(x, _, _)| *x).collect();
|
||||
let col_edges = snap_edges(&v_xs, 3.0);
|
||||
// Column edges from drawn verticals when present, else from the
|
||||
// horizontal-segment endpoints derived above.
|
||||
let col_edges = if let Some(c) = implicit_col_edges {
|
||||
c
|
||||
} else {
|
||||
let v_xs: Vec<f32> = verticals.iter().map(|(x, _, _)| *x).collect();
|
||||
snap_edges(&v_xs, 3.0)
|
||||
};
|
||||
|
||||
log::debug!(
|
||||
"detect_lines p{}: {} row edges, {} col edges after snap",
|
||||
@@ -110,15 +198,21 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Reject page-spanning frames: if the grid covers >90% of a standard page
|
||||
// dimension in both axes, it's a border frame, not a table.
|
||||
// Reject page-spanning frames: a decorative outer border has just 4
|
||||
// edges (top/bottom/left/right). Real full-page tables — common in
|
||||
// governmental ledgers, financial reports, etc. — span the same A4 /
|
||||
// Letter dimensions but have many internal row/column rules. Only
|
||||
// reject when the line set looks like a bare frame, not a grid.
|
||||
// Standard pages are ~595×842 (A4) or ~612×792 (Letter).
|
||||
if table_width > 500.0 && table_height > 700.0 {
|
||||
if table_width > 500.0 && table_height > 700.0 && horizontals.len() <= 4 && verticals.len() <= 4
|
||||
{
|
||||
log::debug!(
|
||||
"detect_lines p{}: rejected — page-spanning frame ({:.0}×{:.0})",
|
||||
"detect_lines p{}: rejected — page-spanning frame ({:.0}×{:.0}, {} h + {} v)",
|
||||
page,
|
||||
table_width,
|
||||
table_height
|
||||
table_height,
|
||||
horizontals.len(),
|
||||
verticals.len()
|
||||
);
|
||||
return Vec::new();
|
||||
}
|
||||
@@ -146,24 +240,33 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
|
||||
// Validate vertical lines: at least 2 should span a meaningful height.
|
||||
// Full spanning (>30%) is ideal, but accept many shorter lines (>10%)
|
||||
// for tables with partial column separators.
|
||||
let spanning_v = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.3)
|
||||
.count();
|
||||
let partial_v = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.10)
|
||||
.count();
|
||||
if spanning_v < 2 && partial_v < 4 {
|
||||
log::debug!(
|
||||
"detect_lines p{}: rejected — {} spanning + {} partial V lines",
|
||||
page,
|
||||
spanning_v,
|
||||
partial_v
|
||||
);
|
||||
return Vec::new();
|
||||
}
|
||||
// for tables with partial column separators. Skipped entirely when
|
||||
// columns came from horizontal-segment endpoints — there are no
|
||||
// vertical lines to validate against, and the segment-endpoint
|
||||
// consistency check in `derive_columns_from_horizontal_segments`
|
||||
// is the equivalent guard.
|
||||
let spanning_v = if cols_from_segments {
|
||||
0
|
||||
} else {
|
||||
let s = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.3)
|
||||
.count();
|
||||
let p = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.10)
|
||||
.count();
|
||||
if s < 2 && p < 4 {
|
||||
log::debug!(
|
||||
"detect_lines p{}: rejected — {} spanning + {} partial V lines",
|
||||
page,
|
||||
s,
|
||||
p
|
||||
);
|
||||
return Vec::new();
|
||||
}
|
||||
s
|
||||
};
|
||||
|
||||
// Row edges need to be in descending order (top of page = higher Y first)
|
||||
let mut row_edges_desc = row_edges;
|
||||
@@ -243,9 +346,11 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
.map(|s| (s - mean_spacing).powi(2))
|
||||
.sum::<f32>()
|
||||
/ spacings.len() as f32;
|
||||
let cv = variance.sqrt() / mean_spacing; // coefficient of variation
|
||||
// CV < 0.05 means nearly identical spacing — chart grid
|
||||
if cv < 0.05 {
|
||||
let cv = variance.sqrt() / mean_spacing;
|
||||
// CV < 0.02 means nearly identical spacing — likely chart grid.
|
||||
// Spreadsheet-exported tables often have uniform rows (CV 0.03-0.05),
|
||||
// so we use a tighter threshold to avoid false negatives.
|
||||
if cv < 0.02 {
|
||||
return Vec::new();
|
||||
}
|
||||
}
|
||||
@@ -263,12 +368,12 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
page, num_rows, num_cols, item_indices.len(), page_item_count, non_empty_rows, cols_with_content
|
||||
);
|
||||
|
||||
vec![Table {
|
||||
columns: col_edges,
|
||||
rows: row_edges_desc[..num_rows].to_vec(),
|
||||
vec![Table::new(
|
||||
col_edges,
|
||||
row_edges_desc[..num_rows].to_vec(),
|
||||
cells,
|
||||
item_indices,
|
||||
}]
|
||||
)]
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -288,6 +393,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -408,6 +515,135 @@ mod tests {
|
||||
assert!(tables.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_horizontal_segments_only_implicit_columns_accepted() {
|
||||
// Catalog/finding-aid pattern: each row's horizontal rule is
|
||||
// drawn as 3 segments at consistent x-endpoints (50, 127, 485,
|
||||
// 562), with no vertical lines anywhere. The segment break
|
||||
// points must be inferred as column edges.
|
||||
let mut lines = Vec::new();
|
||||
// Slightly uneven row spacing so the chart-gridline rejector
|
||||
// (CV < 0.02) doesn't fire.
|
||||
let row_ys = [80.0_f32, 145.0, 215.0, 280.0, 350.0, 415.0, 485.0];
|
||||
for &y in &row_ys {
|
||||
lines.push(make_hline(y, 50.0, 127.0, 1));
|
||||
lines.push(make_hline(y, 127.0, 485.0, 1));
|
||||
lines.push(make_hline(y, 485.0, 562.0, 1));
|
||||
}
|
||||
// Populate every cell so capture / density checks pass.
|
||||
let mut items = Vec::new();
|
||||
for w in row_ys.windows(2) {
|
||||
let row_y = (w[0] + w[1]) / 2.0;
|
||||
items.push(make_item("id", 80.0, row_y, 1));
|
||||
items.push(make_item("description here", 200.0, row_y, 1));
|
||||
items.push(make_item("date", 510.0, row_y, 1));
|
||||
}
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert_eq!(
|
||||
tables.len(),
|
||||
1,
|
||||
"horizontal-segment-only grid should be accepted"
|
||||
);
|
||||
let t = &tables[0];
|
||||
assert!(
|
||||
t.cells.len() >= 4,
|
||||
"expected ≥4 rows, got {}",
|
||||
t.cells.len()
|
||||
);
|
||||
assert_eq!(t.cells[0].len(), 3, "expected 3 columns");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_horizontal_segments_with_inconsistent_endpoints_rejected() {
|
||||
// Decorative rules of varying widths shouldn't be detected as a
|
||||
// table — each line has its own x-endpoints, no consistent
|
||||
// column boundary survives the 50%-of-rows threshold.
|
||||
let lines = vec![
|
||||
make_hline(100.0, 50.0, 150.0, 1),
|
||||
make_hline(200.0, 50.0, 220.0, 1),
|
||||
make_hline(300.0, 50.0, 310.0, 1),
|
||||
make_hline(400.0, 50.0, 470.0, 1),
|
||||
];
|
||||
let items = vec![
|
||||
make_item("decorative", 100.0, 150.0, 1),
|
||||
make_item("text", 100.0, 250.0, 1),
|
||||
];
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"varying-width decorative rules should not be detected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_page_spanning_bare_frame_rejected() {
|
||||
// Just an outer A4-sized rectangle: 2 horizontals + 2 verticals.
|
||||
// No internal structure → decorative border, not a table.
|
||||
let lines = vec![
|
||||
make_hline(20.0, 20.0, 575.0, 1), // top
|
||||
make_hline(820.0, 20.0, 575.0, 1), // bottom
|
||||
make_vline(20.0, 20.0, 820.0, 1), // left
|
||||
make_vline(575.0, 20.0, 820.0, 1), // right
|
||||
];
|
||||
let items = vec![
|
||||
make_item("title", 100.0, 100.0, 1),
|
||||
make_item("body", 100.0, 200.0, 1),
|
||||
];
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"Page-sized 4-edge frame should be rejected as decoration"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_page_spanning_grid_with_internal_lines_accepted() {
|
||||
// Full-page table (governmental-ledger pattern): A4-sized grid
|
||||
// that previously hit the "page-spanning frame" early reject
|
||||
// before downstream validation could even look at it.
|
||||
// Verticals span the full table height so we isolate the
|
||||
// frame-vs-grid decision under test.
|
||||
let mut lines = Vec::new();
|
||||
// 13 horizontal rules: header + 12 row separators
|
||||
let h_ys = [
|
||||
22.5, 37.9, 95.5, 144.5, 184.9, 233.9, 291.7, 340.7, 415.8, 499.6, 574.7, 623.7, 698.8,
|
||||
];
|
||||
for &y in &h_ys {
|
||||
lines.push(make_hline(y, 22.6, 566.6, 1));
|
||||
}
|
||||
// 7 column dividers spanning full table height.
|
||||
let v_xs = [22.6, 66.3, 116.3, 186.6, 263.1, 493.5, 566.5];
|
||||
for &x in &v_xs {
|
||||
lines.push(make_vline(x, 22.5, 698.8, 1));
|
||||
}
|
||||
// Populate every cell so the capture-ratio + density checks pass.
|
||||
let mut items = Vec::new();
|
||||
for r in 0..(h_ys.len() - 1) {
|
||||
let row_y = (h_ys[r] + h_ys[r + 1]) / 2.0;
|
||||
for c in 0..(v_xs.len() - 1) {
|
||||
let col_x = (v_xs[c] + v_xs[c + 1]) / 2.0;
|
||||
items.push(make_item("x", col_x, row_y, 1));
|
||||
}
|
||||
}
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert_eq!(
|
||||
tables.len(),
|
||||
1,
|
||||
"Full-page table with internal grid should be accepted"
|
||||
);
|
||||
let t = &tables[0];
|
||||
assert!(
|
||||
t.cells.len() >= 6,
|
||||
"expected ≥6 rows, got {}",
|
||||
t.cells.len()
|
||||
);
|
||||
assert!(
|
||||
t.cells[0].len() >= 3,
|
||||
"expected ≥3 columns, got {}",
|
||||
t.cells[0].len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_single_column_rejected() {
|
||||
// Only 2 col edges (1 column) — not a table even with verticals
|
||||
|
||||
+809
-61
File diff suppressed because it is too large
Load Diff
+848
-62
@@ -4,15 +4,385 @@
|
||||
//! elements linked to MCIDs, this module builds `Table` structs directly from
|
||||
//! the semantic hierarchy — no geometry heuristics needed.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use log::debug;
|
||||
|
||||
use crate::structure_tree::StructTable;
|
||||
use crate::structure_tree::{StructTable, StructTableRow};
|
||||
use crate::types::TextItem;
|
||||
|
||||
use super::Table;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct MatchedCell {
|
||||
text: String,
|
||||
item_indices: Vec<usize>,
|
||||
x: Option<f32>,
|
||||
y: Option<f32>,
|
||||
}
|
||||
|
||||
fn legacy_column_positions(
|
||||
page_rows: &[&StructTableRow],
|
||||
mcid_to_items: &HashMap<i64, Vec<usize>>,
|
||||
items: &[TextItem],
|
||||
page: u32,
|
||||
num_cols: usize,
|
||||
) -> Vec<f32> {
|
||||
let mut col_positions: Vec<f32> = vec![0.0; num_cols];
|
||||
for (col, col_pos) in col_positions.iter_mut().enumerate() {
|
||||
for row in page_rows {
|
||||
if col < row.cells.len() {
|
||||
if let Some(x) = row.cells[col]
|
||||
.mcids
|
||||
.iter()
|
||||
.filter(|(_, p)| *p == page)
|
||||
.filter_map(|(mcid, _)| mcid_to_items.get(mcid))
|
||||
.flatten()
|
||||
.map(|&idx| items[idx].x)
|
||||
.reduce(f32::min)
|
||||
{
|
||||
*col_pos = x;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
col_positions
|
||||
}
|
||||
|
||||
fn infer_column_positions(
|
||||
raw_rows: &[Vec<MatchedCell>],
|
||||
fallback_positions: &[f32],
|
||||
num_cols: usize,
|
||||
) -> Vec<f32> {
|
||||
const SAME_COLUMN_TOLERANCE: f32 = 18.0;
|
||||
|
||||
let mut anchors = raw_rows
|
||||
.iter()
|
||||
.max_by_key(|row| row.iter().filter(|cell| cell.x.is_some()).count())
|
||||
.map(|row| row.iter().filter_map(|cell| cell.x).collect::<Vec<_>>())
|
||||
.unwrap_or_default();
|
||||
|
||||
if anchors.len() > num_cols {
|
||||
anchors.truncate(num_cols);
|
||||
}
|
||||
|
||||
let mut additional_positions: Vec<f32> = raw_rows
|
||||
.iter()
|
||||
.flat_map(|row| row.iter().filter_map(|cell| cell.x))
|
||||
.collect();
|
||||
additional_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
for x in additional_positions {
|
||||
if anchors.len() >= num_cols {
|
||||
break;
|
||||
}
|
||||
if anchors
|
||||
.iter()
|
||||
.all(|existing| (x - *existing).abs() > SAME_COLUMN_TOLERANCE)
|
||||
{
|
||||
anchors.push(x);
|
||||
anchors.sort_by(|a, b| a.total_cmp(b));
|
||||
}
|
||||
}
|
||||
|
||||
if anchors.len() < num_cols {
|
||||
for &x in fallback_positions {
|
||||
if anchors.len() >= num_cols {
|
||||
break;
|
||||
}
|
||||
if anchors
|
||||
.iter()
|
||||
.all(|existing| (x - *existing).abs() > SAME_COLUMN_TOLERANCE)
|
||||
{
|
||||
anchors.push(x);
|
||||
anchors.sort_by(|a, b| a.total_cmp(b));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if anchors.is_empty() {
|
||||
return fallback_positions.to_vec();
|
||||
}
|
||||
|
||||
while anchors.len() < num_cols {
|
||||
anchors.push(*anchors.last().unwrap());
|
||||
}
|
||||
|
||||
anchors
|
||||
}
|
||||
|
||||
fn align_positions_to_columns(cell_xs: &[f32], columns: &[f32]) -> Vec<usize> {
|
||||
if cell_xs.is_empty() || columns.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
if cell_xs.len() >= columns.len() {
|
||||
return (0..cell_xs.len().min(columns.len())).collect();
|
||||
}
|
||||
|
||||
let mut dp = vec![vec![f32::INFINITY; columns.len() + 1]; cell_xs.len() + 1];
|
||||
let mut take = vec![vec![false; columns.len() + 1]; cell_xs.len() + 1];
|
||||
|
||||
for value in &mut dp[0] {
|
||||
*value = 0.0;
|
||||
}
|
||||
|
||||
for i in 1..=cell_xs.len() {
|
||||
for j in 1..=columns.len() {
|
||||
let skip_cost = dp[i][j - 1];
|
||||
let take_cost = dp[i - 1][j - 1] + (cell_xs[i - 1] - columns[j - 1]).abs();
|
||||
if take_cost <= skip_cost {
|
||||
dp[i][j] = take_cost;
|
||||
take[i][j] = true;
|
||||
} else {
|
||||
dp[i][j] = skip_cost;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let mut assignments_rev = Vec::with_capacity(cell_xs.len());
|
||||
let mut i = cell_xs.len();
|
||||
let mut j = columns.len();
|
||||
while i > 0 && j > 0 {
|
||||
if take[i][j] {
|
||||
assignments_rev.push(j - 1);
|
||||
i -= 1;
|
||||
j -= 1;
|
||||
} else {
|
||||
j -= 1;
|
||||
}
|
||||
}
|
||||
|
||||
assignments_rev.reverse();
|
||||
assignments_rev
|
||||
}
|
||||
|
||||
fn align_struct_rows(
|
||||
raw_rows: &[Vec<MatchedCell>],
|
||||
col_positions: &[f32],
|
||||
) -> (Vec<Vec<String>>, Vec<f32>, Vec<usize>) {
|
||||
let mut cells: Vec<Vec<String>> = Vec::with_capacity(raw_rows.len());
|
||||
let mut row_positions: Vec<f32> = Vec::with_capacity(raw_rows.len());
|
||||
let mut all_item_indices: Vec<usize> = Vec::new();
|
||||
|
||||
for row in raw_rows {
|
||||
let present_cells: Vec<&MatchedCell> = row
|
||||
.iter()
|
||||
.filter(|cell| {
|
||||
!cell.item_indices.is_empty() || !cell.text.is_empty() || cell.x.is_some()
|
||||
})
|
||||
.collect();
|
||||
let cell_xs: Vec<f32> = present_cells.iter().filter_map(|cell| cell.x).collect();
|
||||
let assignments = if cell_xs.len() == present_cells.len() {
|
||||
align_positions_to_columns(&cell_xs, col_positions)
|
||||
} else {
|
||||
(0..present_cells.len().min(col_positions.len())).collect()
|
||||
};
|
||||
|
||||
let mut row_cells = vec![String::new(); col_positions.len()];
|
||||
for (cell, &col_idx) in present_cells.iter().zip(assignments.iter()) {
|
||||
if !cell.text.is_empty() {
|
||||
if !row_cells[col_idx].is_empty() {
|
||||
row_cells[col_idx].push(' ');
|
||||
}
|
||||
row_cells[col_idx].push_str(&cell.text);
|
||||
}
|
||||
all_item_indices.extend(cell.item_indices.iter().copied());
|
||||
}
|
||||
|
||||
let row_y = row
|
||||
.iter()
|
||||
.filter_map(|cell| cell.y)
|
||||
.reduce(f32::max)
|
||||
.unwrap_or(0.0);
|
||||
cells.push(row_cells);
|
||||
row_positions.push(row_y);
|
||||
}
|
||||
|
||||
(cells, row_positions, all_item_indices)
|
||||
}
|
||||
|
||||
fn left_align_struct_rows(
|
||||
raw_rows: &[Vec<MatchedCell>],
|
||||
num_cols: usize,
|
||||
) -> (Vec<Vec<String>>, Vec<f32>, Vec<usize>) {
|
||||
let mut cells: Vec<Vec<String>> = Vec::with_capacity(raw_rows.len());
|
||||
let mut row_positions: Vec<f32> = Vec::with_capacity(raw_rows.len());
|
||||
let mut all_item_indices: Vec<usize> = Vec::new();
|
||||
|
||||
for row in raw_rows {
|
||||
let mut row_cells: Vec<String> = row.iter().map(|cell| cell.text.clone()).collect();
|
||||
row_cells.truncate(num_cols);
|
||||
while row_cells.len() < num_cols {
|
||||
row_cells.push(String::new());
|
||||
}
|
||||
cells.push(row_cells);
|
||||
|
||||
all_item_indices.extend(
|
||||
row.iter()
|
||||
.flat_map(|cell| cell.item_indices.iter().copied()),
|
||||
);
|
||||
row_positions.push(
|
||||
row.iter()
|
||||
.filter_map(|cell| cell.y)
|
||||
.reduce(f32::max)
|
||||
.unwrap_or(0.0),
|
||||
);
|
||||
}
|
||||
|
||||
(cells, row_positions, all_item_indices)
|
||||
}
|
||||
|
||||
fn recover_unclaimed_header_row(table: &mut Table, items: &[TextItem], has_ragged_rows: bool) {
|
||||
if !has_ragged_rows || table.rows.is_empty() || table.columns.len() < 3 {
|
||||
return;
|
||||
}
|
||||
|
||||
const MAX_HEADER_DISTANCE: f32 = 90.0;
|
||||
const MAX_GAP_TO_TABLE: f32 = 35.0;
|
||||
const MAX_INTER_HEADER_GAP: f32 = 25.0;
|
||||
const MAX_HEADER_ROWS: usize = 3;
|
||||
const Y_TOLERANCE: f32 = 5.0;
|
||||
|
||||
let top_row_y = table.rows[0];
|
||||
let x_min = table.columns.first().copied().unwrap_or(0.0) - 25.0;
|
||||
let x_max = table.columns.last().copied().unwrap_or(0.0) + 120.0;
|
||||
let claimed: HashSet<usize> = table.item_indices.iter().copied().collect();
|
||||
|
||||
let mut candidate_rows: Vec<(f32, Vec<(usize, &TextItem)>)> = Vec::new();
|
||||
for (idx, item) in items.iter().enumerate() {
|
||||
if claimed.contains(&idx)
|
||||
|| item.text.trim().is_empty()
|
||||
|| item.y <= top_row_y
|
||||
|| item.y - top_row_y > MAX_HEADER_DISTANCE
|
||||
|| item.x < x_min
|
||||
|| item.x > x_max
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if let Some((_, row_items)) = candidate_rows
|
||||
.iter_mut()
|
||||
.find(|(row_y, _)| (item.y - *row_y).abs() < Y_TOLERANCE)
|
||||
{
|
||||
row_items.push((idx, item));
|
||||
} else {
|
||||
candidate_rows.push((item.y, vec![(idx, item)]));
|
||||
}
|
||||
}
|
||||
|
||||
if candidate_rows.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
for (_, row_items) in &mut candidate_rows {
|
||||
row_items.sort_by(|a, b| a.1.x.total_cmp(&b.1.x));
|
||||
}
|
||||
candidate_rows.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
if candidate_rows[0].0 - top_row_y > MAX_GAP_TO_TABLE {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut candidate_iter = candidate_rows.into_iter();
|
||||
let Some(first_row) = candidate_iter.next() else {
|
||||
return;
|
||||
};
|
||||
let mut selected_rows: Vec<(f32, Vec<(usize, &TextItem)>)> = vec![first_row];
|
||||
let mut prev_y = selected_rows[0].0;
|
||||
for (row_y, row_items) in candidate_iter {
|
||||
if selected_rows.len() >= MAX_HEADER_ROWS {
|
||||
break;
|
||||
}
|
||||
if row_y - prev_y > MAX_INTER_HEADER_GAP {
|
||||
break;
|
||||
}
|
||||
prev_y = row_y;
|
||||
selected_rows.push((row_y, row_items));
|
||||
}
|
||||
|
||||
if selected_rows.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut assigned_rows: Vec<(f32, Vec<String>, Vec<usize>)> = Vec::new();
|
||||
let mut closest_row_populated = 0usize;
|
||||
let mut combined_cols: HashSet<usize> = HashSet::new();
|
||||
|
||||
for (row_idx, (row_y, row_items)) in selected_rows.iter().enumerate() {
|
||||
if row_items.len() > table.columns.len() {
|
||||
return;
|
||||
}
|
||||
|
||||
let row_xs: Vec<f32> = row_items.iter().map(|(_, item)| item.x).collect();
|
||||
let assignments = align_positions_to_columns(&row_xs, &table.columns);
|
||||
if assignments.len() != row_items.len() {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut row_cells = vec![String::new(); table.columns.len()];
|
||||
let mut row_indices = Vec::with_capacity(row_items.len());
|
||||
let mut populated_cols: HashSet<usize> = HashSet::new();
|
||||
|
||||
for ((idx, item), &col_idx) in row_items.iter().zip(assignments.iter()) {
|
||||
let text = item.text.trim();
|
||||
if text.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if !row_cells[col_idx].is_empty() {
|
||||
row_cells[col_idx].push(' ');
|
||||
}
|
||||
row_cells[col_idx].push_str(text);
|
||||
row_indices.push(*idx);
|
||||
populated_cols.insert(col_idx);
|
||||
}
|
||||
|
||||
if row_idx == 0 {
|
||||
closest_row_populated = populated_cols.len();
|
||||
}
|
||||
|
||||
combined_cols.extend(populated_cols.iter().copied());
|
||||
assigned_rows.push((*row_y, row_cells, row_indices));
|
||||
}
|
||||
|
||||
let required_cols = if table.columns.len() <= 4 {
|
||||
table.columns.len()
|
||||
} else {
|
||||
table.columns.len() - 1
|
||||
};
|
||||
if closest_row_populated < 2 || combined_cols.len() < required_cols {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut header_cells = vec![String::new(); table.columns.len()];
|
||||
let mut header_indices = Vec::new();
|
||||
for (_, row_cells, row_indices) in assigned_rows.iter().rev() {
|
||||
for (col_idx, cell_text) in row_cells.iter().enumerate() {
|
||||
if cell_text.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if !header_cells[col_idx].is_empty() {
|
||||
header_cells[col_idx].push(' ');
|
||||
}
|
||||
header_cells[col_idx].push_str(cell_text);
|
||||
}
|
||||
header_indices.extend(row_indices.iter().copied());
|
||||
}
|
||||
|
||||
table.rows.insert(
|
||||
0,
|
||||
assigned_rows
|
||||
.iter()
|
||||
.map(|(row_y, _, _)| *row_y)
|
||||
.reduce(f32::max)
|
||||
.unwrap_or(top_row_y),
|
||||
);
|
||||
table.cells.insert(0, header_cells);
|
||||
table.item_indices.extend(header_indices);
|
||||
table.item_indices.sort_unstable();
|
||||
table.item_indices.dedup();
|
||||
}
|
||||
|
||||
/// Build tables from structure-tree table descriptors by matching MCIDs to TextItems.
|
||||
///
|
||||
/// Returns tables for the given page. Tables where fewer than 50% of cells
|
||||
@@ -67,18 +437,14 @@ pub fn detect_tables_from_struct_tree(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Build cell text and collect item indices
|
||||
let mut cells: Vec<Vec<String>> = Vec::new();
|
||||
let mut all_item_indices: Vec<usize> = Vec::new();
|
||||
// Build cell text and geometry for alignment and header recovery.
|
||||
let mut raw_rows: Vec<Vec<MatchedCell>> = Vec::new();
|
||||
let mut total_cells = 0u32;
|
||||
let mut matched_cells = 0u32;
|
||||
|
||||
for row in &page_rows {
|
||||
let mut row_cells = Vec::with_capacity(num_cols);
|
||||
for (col_idx, cell) in row.cells.iter().enumerate() {
|
||||
if col_idx >= num_cols {
|
||||
break;
|
||||
}
|
||||
let mut row_cells = Vec::with_capacity(row.cells.len());
|
||||
for cell in &row.cells {
|
||||
total_cells += 1;
|
||||
|
||||
// Collect all items for this cell's MCIDs
|
||||
@@ -115,18 +481,18 @@ pub fn detect_tables_from_struct_tree(
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
|
||||
for (idx, _) in &cell_items {
|
||||
all_item_indices.push(*idx);
|
||||
}
|
||||
let item_indices = cell_items.iter().map(|(idx, _)| *idx).collect::<Vec<_>>();
|
||||
let x = cell_items.iter().map(|(_, item)| item.x).reduce(f32::min);
|
||||
let y = cell_items.iter().map(|(_, item)| item.y).reduce(f32::max);
|
||||
|
||||
row_cells.push(text);
|
||||
row_cells.push(MatchedCell {
|
||||
text,
|
||||
item_indices,
|
||||
x,
|
||||
y,
|
||||
});
|
||||
}
|
||||
|
||||
// Pad to num_cols
|
||||
while row_cells.len() < num_cols {
|
||||
row_cells.push(String::new());
|
||||
}
|
||||
cells.push(row_cells);
|
||||
raw_rows.push(row_cells);
|
||||
}
|
||||
|
||||
// Reject if too few cells matched (stale structure tree)
|
||||
@@ -148,51 +514,54 @@ pub fn detect_tables_from_struct_tree(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Derive row/column positions from item geometry
|
||||
let mut row_positions: Vec<f32> = Vec::new();
|
||||
for row in &page_rows {
|
||||
let y = row
|
||||
.cells
|
||||
.iter()
|
||||
.flat_map(|c| c.mcids.iter())
|
||||
.filter(|(_, p)| *p == page)
|
||||
.filter_map(|(mcid, _)| mcid_to_items.get(mcid))
|
||||
.flatten()
|
||||
.map(|&idx| items[idx].y)
|
||||
.reduce(f32::max)
|
||||
.unwrap_or(0.0);
|
||||
row_positions.push(y);
|
||||
}
|
||||
let has_ragged_rows = raw_rows
|
||||
.iter()
|
||||
.any(|row| row.iter().filter(|cell| cell.x.is_some()).count() < num_cols);
|
||||
let first_row_has_tagged_header = page_rows.first().is_some_and(|row| {
|
||||
let header_cells = row.cells.iter().filter(|cell| cell.is_header).count();
|
||||
header_cells * 2 >= row.cells.len()
|
||||
});
|
||||
let fallback_col_positions =
|
||||
legacy_column_positions(&page_rows, &mcid_to_items, items, page, num_cols);
|
||||
let (legacy_cells, legacy_row_positions, mut legacy_item_indices) =
|
||||
left_align_struct_rows(&raw_rows, num_cols);
|
||||
legacy_item_indices.sort_unstable();
|
||||
legacy_item_indices.dedup();
|
||||
let legacy_table = Table::new(
|
||||
fallback_col_positions.clone(),
|
||||
legacy_row_positions,
|
||||
legacy_cells,
|
||||
legacy_item_indices,
|
||||
);
|
||||
|
||||
// Column positions: use X positions of first non-empty cell in each column
|
||||
let mut col_positions: Vec<f32> = vec![0.0; num_cols];
|
||||
for (col, col_pos) in col_positions.iter_mut().enumerate() {
|
||||
for row in &page_rows {
|
||||
if col < row.cells.len() {
|
||||
if let Some(x) = row.cells[col]
|
||||
.mcids
|
||||
.iter()
|
||||
.filter(|(_, p)| *p == page)
|
||||
.filter_map(|(mcid, _)| mcid_to_items.get(mcid))
|
||||
.flatten()
|
||||
.map(|&idx| items[idx].x)
|
||||
.reduce(f32::min)
|
||||
{
|
||||
*col_pos = x;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let col_positions = infer_column_positions(&raw_rows, &fallback_col_positions, num_cols);
|
||||
let (aligned_cells, aligned_row_positions, mut aligned_item_indices) =
|
||||
align_struct_rows(&raw_rows, &col_positions);
|
||||
aligned_item_indices.sort_unstable();
|
||||
aligned_item_indices.dedup();
|
||||
|
||||
all_item_indices.sort_unstable();
|
||||
all_item_indices.dedup();
|
||||
let mut aligned_table = Table::new(
|
||||
col_positions,
|
||||
aligned_row_positions,
|
||||
aligned_cells,
|
||||
aligned_item_indices,
|
||||
);
|
||||
let item_count_before_header = aligned_table.item_indices.len();
|
||||
let row_count_before_header = aligned_table.cells.len();
|
||||
recover_unclaimed_header_row(
|
||||
&mut aligned_table,
|
||||
items,
|
||||
has_ragged_rows && !first_row_has_tagged_header,
|
||||
);
|
||||
|
||||
tables.push(Table {
|
||||
columns: col_positions,
|
||||
rows: row_positions,
|
||||
cells,
|
||||
item_indices: all_item_indices,
|
||||
let recovered_header = aligned_table.item_indices.len() > item_count_before_header
|
||||
|| aligned_table.cells.len() > row_count_before_header;
|
||||
let prefer_aligned = recovered_header;
|
||||
|
||||
tables.push(if prefer_aligned {
|
||||
aligned_table
|
||||
} else {
|
||||
legacy_table
|
||||
});
|
||||
}
|
||||
|
||||
@@ -217,6 +586,8 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
@@ -374,4 +745,419 @@ mod tests {
|
||||
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 2);
|
||||
assert_eq!(tables.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn realigns_ragged_rows_and_recovers_untagged_header() {
|
||||
let items = vec![
|
||||
make_item("Category", 50.0, 120.0, 1, None),
|
||||
make_item("Potentially", 150.0, 120.0, 1, None),
|
||||
make_item("Summary", 250.0, 120.0, 1, None),
|
||||
make_item("Most commonly", 350.0, 120.0, 1, None),
|
||||
make_item("concerning aspect", 150.0, 110.0, 1, None),
|
||||
make_item("suggested", 350.0, 110.0, 1, None),
|
||||
make_item("of circumstances", 150.0, 100.0, 1, None),
|
||||
make_item("intervention", 350.0, 100.0, 1, None),
|
||||
make_item("Existence of red-teaming", 150.0, 80.0, 1, Some(10)),
|
||||
make_item("Important for safety", 250.0, 80.0, 1, Some(11)),
|
||||
make_item("Ensure welfare interviews", 350.0, 80.0, 1, Some(12)),
|
||||
make_item("Identity & self-knowledge", 50.0, 60.0, 1, Some(20)),
|
||||
make_item("Lack of knowledge", 150.0, 60.0, 1, Some(21)),
|
||||
make_item("Overall negative", 250.0, 60.0, 1, Some(22)),
|
||||
make_item("Describe training process", 350.0, 60.0, 1, Some(23)),
|
||||
make_item("Uncertainty around other copies", 150.0, 40.0, 1, Some(30)),
|
||||
make_item("High uncertainty", 250.0, 40.0, 1, Some(31)),
|
||||
make_item("No intervention suggested", 350.0, 40.0, 1, Some(32)),
|
||||
];
|
||||
|
||||
let struct_tables = vec![StructTable {
|
||||
rows: vec![
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(10, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(11, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(12, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(20, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(21, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(22, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(23, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(30, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(31, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(32, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
}];
|
||||
|
||||
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
|
||||
assert_eq!(tables.len(), 1);
|
||||
let table = &tables[0];
|
||||
assert_eq!(table.cells.len(), 4);
|
||||
assert_eq!(
|
||||
table.cells[0],
|
||||
vec![
|
||||
"Category",
|
||||
"Potentially concerning aspect of circumstances",
|
||||
"Summary",
|
||||
"Most commonly suggested intervention",
|
||||
]
|
||||
);
|
||||
assert_eq!(table.cells[1][0], "");
|
||||
assert_eq!(table.cells[1][1], "Existence of red-teaming");
|
||||
assert_eq!(table.cells[2][0], "Identity & self-knowledge");
|
||||
assert_eq!(table.cells[3][0], "");
|
||||
assert_eq!(table.columns.len(), 4);
|
||||
assert!(table.columns.windows(2).all(|w| w[0] < w[1]));
|
||||
assert_eq!(table.item_indices.len(), items.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn does_not_absorb_caption_above_ragged_struct_table() {
|
||||
let items = vec![
|
||||
make_item("Table 5-7: Summary of responses", 50.0, 120.0, 1, None),
|
||||
make_item("Aspect one", 150.0, 80.0, 1, Some(10)),
|
||||
make_item("Summary one", 250.0, 80.0, 1, Some(11)),
|
||||
make_item("Category", 50.0, 60.0, 1, Some(20)),
|
||||
make_item("Aspect two", 150.0, 60.0, 1, Some(21)),
|
||||
make_item("Summary two", 250.0, 60.0, 1, Some(22)),
|
||||
make_item("Aspect three", 150.0, 40.0, 1, Some(30)),
|
||||
make_item("Summary three", 250.0, 40.0, 1, Some(31)),
|
||||
];
|
||||
|
||||
let struct_tables = vec![StructTable {
|
||||
rows: vec![
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(10, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(11, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(20, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(21, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(22, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(30, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(31, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
}];
|
||||
|
||||
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
|
||||
assert_eq!(tables.len(), 1);
|
||||
let table = &tables[0];
|
||||
assert_eq!(table.cells.len(), 3);
|
||||
assert!(
|
||||
table
|
||||
.cells
|
||||
.iter()
|
||||
.flatten()
|
||||
.all(|cell| !cell.contains("Table 5-7")),
|
||||
"caption must stay outside the table"
|
||||
);
|
||||
assert!(!table.item_indices.contains(&0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_existing_tagged_header_without_absorbing_intro_or_caption() {
|
||||
let items = vec![
|
||||
make_item(
|
||||
"Eighteen people left other comments regarding the Project.",
|
||||
50.0,
|
||||
130.0,
|
||||
1,
|
||||
None,
|
||||
),
|
||||
make_item("Table 5-1:", 220.0, 130.0, 1, None),
|
||||
make_item("Other Comments", 350.0, 130.0, 1, None),
|
||||
make_item("Theme", 50.0, 110.0, 1, Some(10)),
|
||||
make_item("Specific Concern/Inquiry", 200.0, 110.0, 1, Some(11)),
|
||||
make_item("Response", 420.0, 110.0, 1, Some(12)),
|
||||
make_item("Traffic", 50.0, 90.0, 1, Some(20)),
|
||||
make_item("Road conditions", 200.0, 90.0, 1, Some(21)),
|
||||
make_item("Maintenance response", 420.0, 90.0, 1, Some(22)),
|
||||
make_item("Noise", 50.0, 70.0, 1, Some(30)),
|
||||
make_item("Dust concerns", 200.0, 70.0, 1, Some(31)),
|
||||
make_item("Mitigation response", 420.0, 70.0, 1, Some(32)),
|
||||
make_item("Resource Use", 50.0, 50.0, 1, Some(40)),
|
||||
make_item("Snowmobile trails", 200.0, 50.0, 1, Some(41)),
|
||||
make_item("Access response", 420.0, 50.0, 1, Some(42)),
|
||||
];
|
||||
|
||||
let struct_tables = vec![StructTable {
|
||||
rows: vec![
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(10, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(11, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(12, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(20, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(21, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(22, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(30, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(31, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(32, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(40, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(41, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(42, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
}];
|
||||
|
||||
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
|
||||
assert_eq!(tables.len(), 1);
|
||||
let table = &tables[0];
|
||||
assert_eq!(table.cells.len(), 4);
|
||||
assert_eq!(
|
||||
table.cells[0],
|
||||
vec!["Theme", "Specific Concern/Inquiry", "Response"]
|
||||
);
|
||||
assert!(!table.item_indices.contains(&0));
|
||||
assert!(!table.item_indices.contains(&1));
|
||||
assert!(!table.item_indices.contains(&2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn does_not_recover_header_for_narrow_two_column_table() {
|
||||
let items = vec![
|
||||
make_item("Alpha", 50.0, 120.0, 1, None),
|
||||
make_item("Beta", 200.0, 120.0, 1, None),
|
||||
make_item("First value", 200.0, 80.0, 1, Some(10)),
|
||||
make_item("Only labeled row", 50.0, 60.0, 1, Some(20)),
|
||||
make_item("Second value", 200.0, 60.0, 1, Some(21)),
|
||||
make_item("Third value", 200.0, 40.0, 1, Some(30)),
|
||||
];
|
||||
|
||||
let struct_tables = vec![StructTable {
|
||||
rows: vec![
|
||||
StructTableRow {
|
||||
cells: vec![StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(10, 1)],
|
||||
}],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(20, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(21, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(30, 1)],
|
||||
}],
|
||||
},
|
||||
],
|
||||
}];
|
||||
|
||||
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
|
||||
assert_eq!(tables.len(), 1);
|
||||
let table = &tables[0];
|
||||
assert_eq!(table.cells.len(), 3);
|
||||
assert_eq!(table.cells[0], vec!["First value", ""]);
|
||||
assert_eq!(table.cells[1], vec!["Only labeled row", "Second value"]);
|
||||
assert_eq!(table.cells[2], vec!["Third value", ""]);
|
||||
assert!(!table.item_indices.contains(&0));
|
||||
assert!(!table.item_indices.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ragged_rows_without_recovered_header_keep_legacy_alignment() {
|
||||
let items = vec![
|
||||
make_item("Date", 150.0, 120.0, 1, Some(10)),
|
||||
make_item("Title", 250.0, 120.0, 1, Some(11)),
|
||||
make_item("PE", 350.0, 120.0, 1, Some(12)),
|
||||
make_item("Bidder", 450.0, 120.0, 1, Some(13)),
|
||||
make_item("Amount", 550.0, 120.0, 1, Some(14)),
|
||||
make_item("1", 50.0, 100.0, 1, Some(20)),
|
||||
make_item("8/1", 150.0, 100.0, 1, Some(21)),
|
||||
make_item("Procurement", 250.0, 100.0, 1, Some(22)),
|
||||
make_item("PUC", 350.0, 100.0, 1, Some(23)),
|
||||
make_item("Vendor", 450.0, 100.0, 1, Some(24)),
|
||||
make_item("SR1", 550.0, 100.0, 1, Some(25)),
|
||||
];
|
||||
|
||||
let struct_tables = vec![StructTable {
|
||||
rows: vec![
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(10, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(11, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(12, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(13, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: true,
|
||||
mcids: vec![(14, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
StructTableRow {
|
||||
cells: vec![
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(20, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(21, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(22, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(23, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(24, 1)],
|
||||
},
|
||||
StructTableCell {
|
||||
is_header: false,
|
||||
mcids: vec![(25, 1)],
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
}];
|
||||
|
||||
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
|
||||
assert_eq!(tables.len(), 1);
|
||||
let table = &tables[0];
|
||||
assert_eq!(table.cells[0][0], "Date");
|
||||
assert_eq!(table.cells[0][4], "Amount");
|
||||
assert_eq!(table.cells[0][5], "");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -108,6 +108,8 @@ pub(crate) fn try_split_financial_item(item: &TextItem) -> Option<Vec<TextItem>>
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item.item_type.clone(),
|
||||
mcid: item.mcid,
|
||||
});
|
||||
|
||||
+495
-8
@@ -1,12 +1,22 @@
|
||||
//! Table-to-markdown formatting and cell cleanup.
|
||||
|
||||
use super::Table;
|
||||
use super::{Table, TableKind};
|
||||
|
||||
pub fn table_to_markdown(table: &Table) -> String {
|
||||
if table.cells.is_empty() || table.cells[0].is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// TOCs render poorly as markdown tables — emit a flat per-row text list
|
||||
// instead so the page numbers stay aligned with their section titles
|
||||
// rather than drifting to a separate column. Format from raw cells
|
||||
// because continuation-row merging in clean_table_cells collapses
|
||||
// separate TOC entries (e.g. "6.2 Contamination" + "6.2.1 SWE-bench")
|
||||
// into one line where sub-entries leave column 0 empty.
|
||||
if table.kind == TableKind::Toc {
|
||||
return format_toc_as_list(&table.cells, &[]);
|
||||
}
|
||||
|
||||
// Clean up the table: merge continuation rows, extract footnotes, remove empty rows
|
||||
let (cleaned_cells, footnotes) = clean_table_cells(&table.cells);
|
||||
|
||||
@@ -49,6 +59,182 @@ pub fn table_to_markdown(table: &Table) -> String {
|
||||
output
|
||||
}
|
||||
|
||||
/// Render a table-of-contents as a flat per-row text block.
|
||||
///
|
||||
/// Each row becomes one line: non-empty cells joined with spaces, and the
|
||||
/// last cell (typically a page number) is separated by a tab so the page
|
||||
/// numbers stay aligned with their titles instead of being pulled into a
|
||||
/// separate column by the column-aware reader.
|
||||
fn format_toc_as_list(cells: &[Vec<String>], footnotes: &[String]) -> String {
|
||||
let mut output = String::new();
|
||||
|
||||
for row in cells {
|
||||
let trimmed: Vec<&str> = row.iter().map(|c| c.trim()).collect();
|
||||
let last_idx = trimmed.iter().rposition(|c| !c.is_empty());
|
||||
let Some(last_idx) = last_idx else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let last_cell = trimmed[last_idx];
|
||||
let last_is_page = is_page_number_cell(last_cell);
|
||||
|
||||
let (title_cells, trailing) = if last_is_page && last_idx > 0 {
|
||||
(&trimmed[..last_idx], Some(last_cell))
|
||||
} else {
|
||||
(&trimmed[..=last_idx], None)
|
||||
};
|
||||
|
||||
// Skip dots-only cells when joining the title — in a detected TOC
|
||||
// layout, a "...." cell is a leader separator, not part of the
|
||||
// entry name.
|
||||
let title = title_cells
|
||||
.iter()
|
||||
.filter(|c| !c.is_empty() && !is_dots_only(c))
|
||||
.copied()
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
|
||||
if title.is_empty() && trailing.is_none() {
|
||||
continue;
|
||||
}
|
||||
|
||||
if !title.is_empty() {
|
||||
output.push_str(&title);
|
||||
}
|
||||
if let Some(page) = trailing {
|
||||
if !title.is_empty() {
|
||||
output.push('\t');
|
||||
}
|
||||
output.push_str(page);
|
||||
}
|
||||
output.push('\n');
|
||||
}
|
||||
|
||||
if !footnotes.is_empty() {
|
||||
output.push('\n');
|
||||
for footnote in footnotes {
|
||||
output.push_str(footnote);
|
||||
output.push('\n');
|
||||
}
|
||||
}
|
||||
|
||||
output
|
||||
}
|
||||
|
||||
/// True when the cell looks like a page number. Accepts:
|
||||
/// - plain digit tokens: "42", "86 86"
|
||||
/// - dashed section-page IDs: "5-21", "A-1", "B--3", "TC-2" (common in
|
||||
/// technical manuals)
|
||||
fn is_page_number_cell(cell: &str) -> bool {
|
||||
let tokens: Vec<&str> = cell.split_whitespace().collect();
|
||||
if tokens.is_empty() {
|
||||
return false;
|
||||
}
|
||||
tokens.iter().all(|t| {
|
||||
if t.is_empty() || t.len() > 8 {
|
||||
return false;
|
||||
}
|
||||
let all_digits = t.chars().all(|c| c.is_ascii_digit());
|
||||
if all_digits {
|
||||
return t.len() <= 4;
|
||||
}
|
||||
// Section-page form: uppercase letters, digits, dashes; at least
|
||||
// one digit present.
|
||||
t.chars()
|
||||
.all(|c| c.is_ascii_digit() || c.is_ascii_uppercase() || c == '-')
|
||||
&& t.chars().any(|c| c.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
/// True when the cell is purely leader dots (any length ≥ 3) with optional
|
||||
/// whitespace.
|
||||
fn is_dots_only(cell: &str) -> bool {
|
||||
let t = cell.trim();
|
||||
let dots = t.chars().filter(|&c| c == '.').count();
|
||||
dots >= 3 && t.chars().all(|c| c == '.' || c.is_whitespace())
|
||||
}
|
||||
|
||||
fn starts_with_uppercase_word(cell: &str) -> bool {
|
||||
cell.chars()
|
||||
.find(|c| c.is_alphanumeric())
|
||||
.is_some_and(|c| c.is_uppercase())
|
||||
}
|
||||
|
||||
fn starts_with_uppercase_alpha(cell: &str) -> bool {
|
||||
cell.chars()
|
||||
.find(|c| c.is_alphabetic())
|
||||
.is_some_and(|c| c.is_uppercase())
|
||||
}
|
||||
|
||||
fn starts_with_lowercase_alpha(cell: &str) -> bool {
|
||||
cell.chars()
|
||||
.find(|c| c.is_alphabetic())
|
||||
.is_some_and(|c| c.is_lowercase())
|
||||
}
|
||||
|
||||
fn starts_with_numbered_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim_start();
|
||||
let digit_count = trimmed.chars().take_while(|c| c.is_ascii_digit()).count();
|
||||
|
||||
digit_count > 0
|
||||
&& digit_count <= 3
|
||||
&& trimmed
|
||||
.chars()
|
||||
.nth(digit_count)
|
||||
.is_some_and(|c| matches!(c, '.' | ')' | '-' | ':'))
|
||||
}
|
||||
|
||||
fn alpha_word_count(cell: &str) -> usize {
|
||||
cell.split_whitespace()
|
||||
.filter(|word| word.chars().any(|c| c.is_alphabetic()))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn looks_like_compact_entry_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 3 || trimmed.len() > 80 {
|
||||
return false;
|
||||
}
|
||||
|
||||
if !starts_with_uppercase_alpha(trimmed) && !starts_with_numbered_label(trimmed) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if trimmed.ends_with(['.', ',', ';', ':']) {
|
||||
return false;
|
||||
}
|
||||
|
||||
let words = alpha_word_count(trimmed);
|
||||
(1..=6).contains(&words)
|
||||
}
|
||||
|
||||
fn looks_like_plain_section_label(cell: &str) -> bool {
|
||||
let trimmed = cell.trim();
|
||||
if trimmed.len() < 4 || trimmed.len() > 40 {
|
||||
return false;
|
||||
}
|
||||
if trimmed.ends_with(['.', ',', ';', ':']) || trimmed.contains(|ch: char| ch.is_ascii_digit()) {
|
||||
return false;
|
||||
}
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|ch| !ch.is_lowercase()) {
|
||||
return false;
|
||||
}
|
||||
trimmed
|
||||
.chars()
|
||||
.all(|ch| ch.is_alphabetic() || ch.is_whitespace() || matches!(ch, '&' | '/' | '-'))
|
||||
&& starts_with_uppercase_alpha(trimmed)
|
||||
&& (1..=4).contains(&alpha_word_count(trimmed))
|
||||
}
|
||||
|
||||
fn ends_like_incomplete_phrase(cell: &str) -> bool {
|
||||
let lower = cell.trim_end().to_ascii_lowercase();
|
||||
lower.ends_with(" and")
|
||||
|| lower.ends_with(" or")
|
||||
|| lower.ends_with(',')
|
||||
|| lower.ends_with('-')
|
||||
|| lower.ends_with('/')
|
||||
}
|
||||
|
||||
/// Clean up table cells: merge continuation rows, extract footnotes, remove empty rows
|
||||
fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
let mut cleaned: Vec<Vec<String>> = Vec::new();
|
||||
@@ -74,6 +260,9 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let num_cols = row.len();
|
||||
let filled_cells = row.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
|
||||
// Check if this is a continuation row (first column is empty but others have content).
|
||||
// A row with only 1 short non-empty cell (besides the first) is more likely a
|
||||
// section sub-header (e.g. "JAN", "FEB") than overflow text — don't merge it.
|
||||
@@ -107,26 +296,80 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
let looks_like_data_row = non_first_cells.len() >= 2
|
||||
&& avg_cell_len <= 10.0
|
||||
&& numeric_cells > non_first_cells.len() / 2;
|
||||
let uppercase_leading_cells = non_first_cells
|
||||
.iter()
|
||||
.filter(|cell| starts_with_uppercase_word(cell))
|
||||
.count();
|
||||
let first_non_empty_col = row.iter().position(|c| !c.trim().is_empty());
|
||||
let first_non_empty_cell = first_non_empty_col
|
||||
.and_then(|idx| row.get(idx))
|
||||
.map(|c| c.trim())
|
||||
.unwrap_or("");
|
||||
let title_like_later_cells = first_non_empty_col
|
||||
.map(|idx| {
|
||||
row.iter()
|
||||
.skip(idx + 1)
|
||||
.map(|c| c.trim())
|
||||
.filter(|c| !c.is_empty() && starts_with_uppercase_alpha(c))
|
||||
.count()
|
||||
})
|
||||
.unwrap_or(0);
|
||||
let prev_first_cell_empty = cleaned
|
||||
.last()
|
||||
.and_then(|r| r.first())
|
||||
.is_some_and(|c| c.trim().is_empty());
|
||||
let prev_first_cell = cleaned
|
||||
.last()
|
||||
.and_then(|r| r.first())
|
||||
.map(|c| c.trim())
|
||||
.unwrap_or("");
|
||||
let header_filled = cleaned
|
||||
.first()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(num_cols);
|
||||
let looks_like_spanning_first_column_row = first_cell.is_empty()
|
||||
&& row.len() >= 4
|
||||
&& non_first_cells.len() == row.len().saturating_sub(1)
|
||||
&& uppercase_leading_cells >= non_first_cells.len().saturating_sub(1);
|
||||
// Hierarchical tables often use a row-spanned first column: sub-rows
|
||||
// leave column 0 blank, then start a compact title-like label in
|
||||
// column 1. Wrapped continuations in the existing fixtures start
|
||||
// mid-sentence/lowercase ("continued text here", "with 3.5%...") or
|
||||
// carry lowercase fragments in the later cells, so keep those mergeable.
|
||||
let looks_like_hierarchical_subrow = first_cell.is_empty()
|
||||
&& row.len() >= 3
|
||||
&& first_non_empty_col == Some(1)
|
||||
&& looks_like_compact_entry_label(first_non_empty_cell)
|
||||
&& ((non_first_cells.len() >= 2 && title_like_later_cells > 0)
|
||||
|| (non_first_cells.len() == 1
|
||||
&& prev_first_cell_empty
|
||||
&& alpha_word_count(first_non_empty_cell) >= 2));
|
||||
let looks_like_new_first_column_entry = !first_cell.is_empty()
|
||||
&& (starts_with_numbered_label(first_cell) || starts_with_uppercase_alpha(first_cell))
|
||||
&& filled_cells >= 2
|
||||
&& non_first_cells
|
||||
.iter()
|
||||
.any(|cell| looks_like_compact_entry_label(cell));
|
||||
let looks_like_section_label_row = !first_cell.is_empty()
|
||||
&& filled_cells == 1
|
||||
&& header_filled >= 3
|
||||
&& looks_like_plain_section_label(first_cell);
|
||||
// Classic continuation: first cell empty, content in other cells
|
||||
let is_classic_continuation = first_cell.is_empty()
|
||||
&& !non_first_cells.is_empty()
|
||||
&& !is_short_subheader
|
||||
&& !looks_like_data_row
|
||||
&& !looks_like_spanning_first_column_row
|
||||
&& !looks_like_hierarchical_subrow
|
||||
&& cleaned.len() > 1;
|
||||
|
||||
// Wrapped-cell continuation: row has fewer filled cells than the header
|
||||
// row, suggesting it's overflow text from the previous row's cells.
|
||||
// Only trigger when the previous row has significantly more filled cells.
|
||||
let num_cols = row.len();
|
||||
let filled_cells = row.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
let prev_filled = cleaned
|
||||
.last()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(0);
|
||||
let header_filled = cleaned
|
||||
.first()
|
||||
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
|
||||
.unwrap_or(num_cols);
|
||||
// Merge when the row has significantly fewer filled cells than header.
|
||||
// For wide tables (5+ cols), require ≤50% of header cells.
|
||||
// For narrow tables (2-4 cols), require fewer than header cells.
|
||||
@@ -137,10 +380,18 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
} else {
|
||||
header_filled.saturating_sub(1)
|
||||
};
|
||||
let continues_wrapped_first_column_label = !first_cell.is_empty()
|
||||
&& starts_with_lowercase_alpha(first_cell)
|
||||
&& ends_like_incomplete_phrase(prev_first_cell);
|
||||
let is_wrapped_continuation = cleaned.len() > 1
|
||||
&& filled_cells <= max_filled_for_merge
|
||||
&& prev_filled > filled_cells
|
||||
&& (prev_filled > filled_cells
|
||||
|| (continues_wrapped_first_column_label && prev_filled >= filled_cells))
|
||||
&& !looks_like_data_row
|
||||
&& !looks_like_spanning_first_column_row
|
||||
&& !looks_like_hierarchical_subrow
|
||||
&& !looks_like_new_first_column_entry
|
||||
&& !looks_like_section_label_row
|
||||
&& !is_short_subheader;
|
||||
|
||||
let is_continuation = is_classic_continuation || is_wrapped_continuation;
|
||||
@@ -286,6 +537,44 @@ mod tests {
|
||||
assert!(cleaned[1][1].contains("continued text here"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_first_column_section_label_not_merged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Properties".into(),
|
||||
"Conditions".into(),
|
||||
"Method".into(),
|
||||
"Typical values".into(),
|
||||
"Units".into(),
|
||||
],
|
||||
vec![
|
||||
"Melt Flow Rate".into(),
|
||||
"230 C/2.16 kg".into(),
|
||||
"ASTM D1238".into(),
|
||||
"3.0".into(),
|
||||
"g/10 min".into(),
|
||||
],
|
||||
vec![
|
||||
"Mechanical".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"Tensile Stress at Yield".into(),
|
||||
"50 mm/min".into(),
|
||||
"ASTM D638".into(),
|
||||
"31".into(),
|
||||
"MPa".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 4);
|
||||
assert_eq!(cleaned[2][0], "Mechanical");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_short_subheader_not_merged() {
|
||||
let cells = vec![
|
||||
@@ -310,6 +599,159 @@ mod tests {
|
||||
assert_eq!(cleaned.len(), 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_spanning_first_column_row_not_merged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Category".into(),
|
||||
"Potentially concerning aspect".into(),
|
||||
"Summary".into(),
|
||||
"Intervention".into(),
|
||||
],
|
||||
vec![
|
||||
"Identity & self-knowledge".into(),
|
||||
"Lack of knowledge".into(),
|
||||
"Overall negative".into(),
|
||||
"Describe training".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Uncertainty around other copies".into(),
|
||||
"High uncertainty".into(),
|
||||
"No intervention suggested".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
assert_eq!(cleaned.len(), 3);
|
||||
assert_eq!(cleaned[2][0], "");
|
||||
assert_eq!(cleaned[2][1], "Uncertainty around other copies");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_numbered_hierarchy_rows_not_overmerged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Group".into(),
|
||||
"Task".into(),
|
||||
"Detail".into(),
|
||||
"Benefit".into(),
|
||||
],
|
||||
vec![
|
||||
"1. Group alpha".into(),
|
||||
"Task setup and".into(),
|
||||
"Begin setup".into(),
|
||||
"Faster start".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"management".into(),
|
||||
"recommended profile".into(),
|
||||
"with saved defaults".into(),
|
||||
],
|
||||
vec![
|
||||
"2. Group beta and".into(),
|
||||
"Storage setup".into(),
|
||||
"Provides upload tools".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"fine-tuning".into(),
|
||||
"".into(),
|
||||
"for filtered inputs".into(),
|
||||
"service".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Label workspace".into(),
|
||||
"Creates review sets".into(),
|
||||
"Lets teams review".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Model training".into(),
|
||||
"".into(),
|
||||
"Supports custom model".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 5);
|
||||
assert_eq!(cleaned[1][0], "1. Group alpha");
|
||||
assert_eq!(cleaned[1][1], "Task setup and management");
|
||||
assert_eq!(cleaned[2][0], "2. Group beta and fine-tuning");
|
||||
assert_eq!(cleaned[2][1], "Storage setup");
|
||||
assert_eq!(cleaned[3][1], "Label workspace");
|
||||
assert_eq!(cleaned[4][1], "Model training");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_partial_hierarchical_subrow_not_merged() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Group".into(),
|
||||
"Task".into(),
|
||||
"Detail".into(),
|
||||
"Benefit".into(),
|
||||
],
|
||||
vec![
|
||||
"Group A".into(),
|
||||
"Alpha task".into(),
|
||||
"Initial detail".into(),
|
||||
"Initial benefit".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"Beta task".into(),
|
||||
"Parallel detail".into(),
|
||||
"".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"second line".into(),
|
||||
"additional detail".into(),
|
||||
"".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 3);
|
||||
assert_eq!(cleaned[1][1], "Alpha task");
|
||||
assert_eq!(cleaned[2][0], "");
|
||||
assert_eq!(cleaned[2][1], "Beta task second line");
|
||||
assert_eq!(cleaned[2][2], "Parallel detail additional detail");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_full_width_continuation_row_still_merges_when_lowercase() {
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Classification".into(),
|
||||
"Before tax".into(),
|
||||
"After tax".into(),
|
||||
"Standard equipment".into(),
|
||||
"Options".into(),
|
||||
],
|
||||
vec![
|
||||
"Exclusive Special".into(),
|
||||
"83,500,000".into(),
|
||||
"79,275,000".into(),
|
||||
"Standard equipment".into(),
|
||||
"Option A".into(),
|
||||
],
|
||||
vec![
|
||||
"".into(),
|
||||
"with 3.5% individual consumption tax applied".into(),
|
||||
"with 3.5% individual consumption tax applied".into(),
|
||||
"lighting(crash pad)".into(),
|
||||
"sound system".into(),
|
||||
],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
assert_eq!(cleaned.len(), 2);
|
||||
assert!(cleaned[1][1].contains("83,500,000"));
|
||||
assert!(cleaned[1][1].contains("with 3.5%"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_header_row_not_merged() {
|
||||
// Continuation requires cleaned.len() > 1 (don't merge into header)
|
||||
@@ -358,6 +800,7 @@ mod tests {
|
||||
vec!["Bob".into(), "25".into()],
|
||||
],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
let md = table_to_markdown(&table);
|
||||
assert!(md.contains("|Name|"));
|
||||
@@ -373,6 +816,7 @@ mod tests {
|
||||
rows: vec![500.0],
|
||||
cells: vec![vec!["Only".into(), "Row".into()]],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
let md = table_to_markdown(&table);
|
||||
assert!(md.contains("|Only|"));
|
||||
@@ -386,6 +830,7 @@ mod tests {
|
||||
rows: vec![],
|
||||
cells: vec![],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
assert_eq!(table_to_markdown(&table), "");
|
||||
}
|
||||
@@ -401,6 +846,7 @@ mod tests {
|
||||
vec!["(1)".into(), "Footnote text".into()],
|
||||
],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
let md = table_to_markdown(&table);
|
||||
assert!(md.contains("(1) Footnote text"));
|
||||
@@ -416,6 +862,7 @@ mod tests {
|
||||
vec!["太郎".into(), "25".into()],
|
||||
],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
let md = table_to_markdown(&table);
|
||||
assert!(md.contains("名前"));
|
||||
@@ -429,7 +876,47 @@ mod tests {
|
||||
rows: vec![500.0],
|
||||
cells: vec![vec![]],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
assert_eq!(table_to_markdown(&table), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_table_to_markdown_toc_renders_as_flat_list() {
|
||||
// A TOC-shaped table with section numbers in col 0 and page numbers
|
||||
// in the last column should render as a flat list, not a markdown
|
||||
// table, so the page numbers stay on the same line as their titles.
|
||||
let table = Table::new(
|
||||
vec![50.0, 80.0, 300.0],
|
||||
vec![500.0; 5],
|
||||
vec![
|
||||
vec![
|
||||
"4.3".into(),
|
||||
"Case studies and targeted evaluations".into(),
|
||||
"86".into(),
|
||||
],
|
||||
vec![
|
||||
"4.3.1".into(),
|
||||
"Destructive or reckless actions".into(),
|
||||
"86".into(),
|
||||
],
|
||||
vec![
|
||||
"4.3.2".into(),
|
||||
"Adherence to its constitution".into(),
|
||||
"89".into(),
|
||||
],
|
||||
vec!["4.4".into(), "Capability evaluations".into(), "101".into()],
|
||||
vec!["4.5".into(), "White-box analyses".into(), "113".into()],
|
||||
],
|
||||
vec![],
|
||||
);
|
||||
assert_eq!(table.kind, TableKind::Toc);
|
||||
let md = table_to_markdown(&table);
|
||||
assert!(
|
||||
!md.contains("|---|"),
|
||||
"TOC should not render as a markdown table: {md}"
|
||||
);
|
||||
assert!(md.contains("4.3 Case studies and targeted evaluations\t86"));
|
||||
assert!(md.contains("4.5 White-box analyses\t113"));
|
||||
}
|
||||
}
|
||||
|
||||
+162
-12
@@ -82,33 +82,42 @@ pub(crate) fn find_column_boundaries(
|
||||
}
|
||||
}
|
||||
|
||||
let mut columns = Vec::new();
|
||||
let mut cluster_items: Vec<f32> = vec![x_positions[0]];
|
||||
// Track cluster membership: for each cluster, store the list of x positions
|
||||
let mut cluster_xs: Vec<Vec<f32>> = vec![vec![x_positions[0]]];
|
||||
|
||||
for &x in &x_positions[1..] {
|
||||
let last_cluster = cluster_xs.last().unwrap();
|
||||
// For dense columns (gap-histogram triggered), use edge-based clustering:
|
||||
// compare with the last item to avoid center-drift that merges adjacent
|
||||
// narrow columns. For normal tables, use center-based (original behavior).
|
||||
let reference = if use_edge_clustering {
|
||||
*cluster_items.last().unwrap()
|
||||
*last_cluster.last().unwrap()
|
||||
} else {
|
||||
cluster_items.iter().sum::<f32>() / cluster_items.len() as f32
|
||||
last_cluster.iter().sum::<f32>() / last_cluster.len() as f32
|
||||
};
|
||||
|
||||
if x - reference > cluster_threshold {
|
||||
let cluster_center = cluster_items.iter().sum::<f32>() / cluster_items.len() as f32;
|
||||
columns.push(cluster_center);
|
||||
cluster_items = vec![x];
|
||||
cluster_xs.push(vec![x]);
|
||||
} else {
|
||||
cluster_items.push(x);
|
||||
cluster_xs.last_mut().unwrap().push(x);
|
||||
}
|
||||
}
|
||||
|
||||
// Don't forget last cluster
|
||||
if !cluster_items.is_empty() {
|
||||
columns.push(cluster_items.iter().sum::<f32>() / cluster_items.len() as f32);
|
||||
// Numeric column merge pass: when a sparse cluster (few items, typically
|
||||
// header text) is adjacent to a dense numeric cluster and within 1.5×
|
||||
// threshold, merge them. This fixes tables where multi-line wrapped
|
||||
// headers have slightly different X positions than the data columns,
|
||||
// causing the header and data to split into separate clusters.
|
||||
let columns_before_merge = cluster_xs.len();
|
||||
if columns_before_merge >= 3 {
|
||||
cluster_xs = merge_numeric_adjacent_clusters(cluster_xs, items, cluster_threshold);
|
||||
}
|
||||
|
||||
let columns: Vec<f32> = cluster_xs
|
||||
.iter()
|
||||
.map(|xs| xs.iter().sum::<f32>() / xs.len() as f32)
|
||||
.collect();
|
||||
|
||||
// Filter columns - each should have multiple items
|
||||
let min_items_per_col = (items.len() / columns.len().max(1) / 4).max(2);
|
||||
let columns: Vec<f32> = columns
|
||||
@@ -123,8 +132,9 @@ pub(crate) fn find_column_boundaries(
|
||||
.collect();
|
||||
|
||||
log::debug!(
|
||||
" find_column_boundaries: {} columns before filter, threshold={:.1}, {} items",
|
||||
" find_column_boundaries: {} columns (merged from {}), threshold={:.1}, {} items",
|
||||
columns.len(),
|
||||
columns_before_merge,
|
||||
cluster_threshold,
|
||||
items.len()
|
||||
);
|
||||
@@ -148,6 +158,116 @@ pub(crate) fn find_column_boundaries(
|
||||
columns
|
||||
}
|
||||
|
||||
/// Check if a text string looks like a number (digits, decimals, sign, comma).
|
||||
fn is_numeric_text(s: &str) -> bool {
|
||||
let s = s.trim();
|
||||
if s.is_empty() {
|
||||
return false;
|
||||
}
|
||||
// Match patterns like: 8.23, -1.05, 9.99, 7.12, 100, 3,456.78, +5%, ---
|
||||
// But NOT: BIO, Department, Core Courses
|
||||
s.chars()
|
||||
.all(|c| c.is_ascii_digit() || c == '.' || c == ',' || c == '-' || c == '+' || c == '%')
|
||||
&& s.chars().any(|c| c.is_ascii_digit())
|
||||
}
|
||||
|
||||
/// Merge adjacent X-position clusters when one is a sparse header cluster
|
||||
/// and the other is a dense numeric data cluster. This prevents multi-line
|
||||
/// wrapped headers from splitting a logical column into two clusters.
|
||||
fn merge_numeric_adjacent_clusters(
|
||||
mut clusters: Vec<Vec<f32>>,
|
||||
items: &[(usize, &TextItem)],
|
||||
threshold: f32,
|
||||
) -> Vec<Vec<f32>> {
|
||||
// For each cluster, compute: center, item count, numeric fraction
|
||||
struct ClusterInfo {
|
||||
center: f32,
|
||||
count: usize,
|
||||
numeric_frac: f32,
|
||||
}
|
||||
|
||||
let compute_info = |xs: &[f32]| -> ClusterInfo {
|
||||
let center = xs.iter().sum::<f32>() / xs.len() as f32;
|
||||
// Count items and numeric fraction for items near this cluster center
|
||||
let mut total = 0;
|
||||
let mut numeric = 0;
|
||||
for (_, item) in items {
|
||||
if (item.x - center).abs() < threshold {
|
||||
total += 1;
|
||||
if is_numeric_text(&item.text) {
|
||||
numeric += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
ClusterInfo {
|
||||
center,
|
||||
count: total,
|
||||
numeric_frac: if total > 0 {
|
||||
numeric as f32 / total as f32
|
||||
} else {
|
||||
0.0
|
||||
},
|
||||
}
|
||||
};
|
||||
|
||||
// Merge distance: allow merging clusters that are slightly beyond the
|
||||
// original threshold. Use 1.5× threshold to catch header-vs-data splits.
|
||||
let merge_dist = threshold * 1.5;
|
||||
|
||||
// Iterate and merge adjacent pairs. Use a simple left-to-right scan.
|
||||
let mut merged = true;
|
||||
while merged {
|
||||
merged = false;
|
||||
let mut i = 0;
|
||||
while i + 1 < clusters.len() {
|
||||
let info_a = compute_info(&clusters[i]);
|
||||
let info_b = compute_info(&clusters[i + 1]);
|
||||
let dist = (info_b.center - info_a.center).abs();
|
||||
|
||||
if dist > merge_dist {
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Determine if one cluster is sparse (header) and the other
|
||||
// is dense and numeric (data). A cluster is "sparse" if it has
|
||||
// significantly fewer items than the other.
|
||||
let (sparse, dense) = if info_a.count < info_b.count {
|
||||
(&info_a, &info_b)
|
||||
} else {
|
||||
(&info_b, &info_a)
|
||||
};
|
||||
|
||||
// Merge if the dense cluster is predominantly numeric (>50%)
|
||||
// and the sparse cluster has at most 1/3 the items of the dense one.
|
||||
let should_merge =
|
||||
dense.numeric_frac > 0.50 && sparse.count <= dense.count / 2 && sparse.count <= 5;
|
||||
|
||||
if should_merge {
|
||||
log::debug!(
|
||||
" merging column clusters: center {:.1} ({} items, {:.0}% numeric) + {:.1} ({} items, {:.0}% numeric), dist={:.1}",
|
||||
info_a.center,
|
||||
info_a.count,
|
||||
info_a.numeric_frac * 100.0,
|
||||
info_b.center,
|
||||
info_b.count,
|
||||
info_b.numeric_frac * 100.0,
|
||||
dist,
|
||||
);
|
||||
// Merge cluster i+1 into cluster i
|
||||
let next = clusters.remove(i + 1);
|
||||
clusters[i].extend(next);
|
||||
merged = true;
|
||||
// Don't increment i — check if the merged cluster can merge further
|
||||
} else {
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
clusters
|
||||
}
|
||||
|
||||
/// Find row boundaries by clustering Y positions
|
||||
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||
@@ -249,6 +369,10 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|
||||
let prev_ends_with_hyphen = result.ends_with('-');
|
||||
let curr_is_hyphen = text == "-";
|
||||
let curr_starts_with_hyphen = text.starts_with('-');
|
||||
let prev_ends_with_open_delimiter =
|
||||
result.ends_with('(') || result.ends_with('[') || result.ends_with('{');
|
||||
let curr_starts_with_close_delimiter =
|
||||
text.starts_with(')') || text.starts_with(']') || text.starts_with('}');
|
||||
|
||||
// Detect subscript/superscript: smaller font size and/or Y offset
|
||||
let font_ratio = item.font_size / prev_item.font_size;
|
||||
@@ -265,6 +389,8 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|
||||
|| curr_starts_with_hyphen
|
||||
|| is_sub_super
|
||||
|| was_sub_super
|
||||
|| prev_ends_with_open_delimiter
|
||||
|| curr_starts_with_close_delimiter
|
||||
{
|
||||
result.push_str(text);
|
||||
} else {
|
||||
@@ -379,6 +505,7 @@ pub(crate) fn recover_header_row(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::tables::TableKind;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn make_item(text: &str, x: f32, y: f32, font_size: f32) -> TextItem {
|
||||
@@ -393,6 +520,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -606,6 +735,18 @@ mod tests {
|
||||
assert_eq!(join_cell_items(&[&a, &b, &c]), "pre-fix");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_join_cell_items_parenthetical_no_inner_spaces() {
|
||||
let a = make_item("The first sentence", 100.0, 500.0, 10.0);
|
||||
let b = make_item("(", 190.0, 500.0, 10.0);
|
||||
let c = make_item("twice", 195.0, 500.0, 10.0);
|
||||
let d = make_item(")", 220.0, 500.0, 10.0);
|
||||
assert_eq!(
|
||||
join_cell_items(&[&a, &b, &c, &d]),
|
||||
"The first sentence (twice)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_join_cell_items_subscript_no_space() {
|
||||
let a = make_item("H", 100.0, 500.0, 12.0);
|
||||
@@ -636,6 +777,7 @@ mod tests {
|
||||
rows: vec![500.0, 480.0],
|
||||
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||
item_indices: vec![2, 3],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
recover_header_row(&mut table, &all_items, 9.0);
|
||||
@@ -654,6 +796,7 @@ mod tests {
|
||||
rows: vec![500.0],
|
||||
cells: vec![vec!["A".into(), "B".into()]],
|
||||
item_indices: vec![0, 1],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
let rows_before = table.rows.len();
|
||||
@@ -674,6 +817,7 @@ mod tests {
|
||||
rows: vec![500.0, 480.0],
|
||||
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
|
||||
item_indices: vec![2, 3],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
let rows_before = table.rows.len();
|
||||
@@ -694,6 +838,7 @@ mod tests {
|
||||
rows: vec![500.0],
|
||||
cells: vec![vec!["A".into(), "B".into()]],
|
||||
item_indices: vec![1, 2],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
let rows_before = table.rows.len();
|
||||
@@ -709,6 +854,7 @@ mod tests {
|
||||
rows: vec![],
|
||||
cells: vec![],
|
||||
item_indices: vec![],
|
||||
kind: TableKind::Data,
|
||||
};
|
||||
|
||||
recover_header_row(&mut table, &all_items, 9.0);
|
||||
@@ -740,6 +886,8 @@ mod tests {
|
||||
font: String::new(),
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
page: 1,
|
||||
@@ -776,6 +924,8 @@ mod tests {
|
||||
font: String::new(),
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
page: 1,
|
||||
|
||||
+1290
-9
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,972 @@
|
||||
//! Structure-recovery-aware (TSR) table assembly.
|
||||
//!
|
||||
//! Consumes the raw output of an external table-structure recognition model
|
||||
//! (e.g. SLANet on PaddleOCR): a flat list of HTML structure tokens plus a
|
||||
//! parallel list of per-cell bboxes. Pairs each cell open-tag with its bbox
|
||||
//! in document order, tracks row/column position with rowspan/colspan
|
||||
//! awareness, and emits a markdown pipe-table.
|
||||
//!
|
||||
//! No real HTML parser is needed — the token grammar is restricted (see
|
||||
//! [`parse_structure`]), so a small state machine is enough.
|
||||
//!
|
||||
//! Cell text is supplied separately by the caller (typically by overlap-
|
||||
//! testing PDF text items against each cell's page-PDF-pt bbox).
|
||||
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
/// A single resolved cell, with both structural metadata and its bbox in
|
||||
/// page PDF-points (top-left origin).
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct StructuredCell {
|
||||
/// 0-indexed grid row.
|
||||
pub row: usize,
|
||||
/// 0-indexed grid column.
|
||||
pub col: usize,
|
||||
/// 1 for a normal cell.
|
||||
pub rowspan: usize,
|
||||
/// 1 for a normal cell.
|
||||
pub colspan: usize,
|
||||
/// `true` when the cell is a `<th>` or sits inside `<thead>`.
|
||||
pub is_header: bool,
|
||||
/// Cell text (filled in by the caller after overlap-testing PDF items).
|
||||
pub text: String,
|
||||
/// Axis-aligned bbox `[x1, y1, x2, y2]` in page PDF-points, top-left origin.
|
||||
pub page_pt_bbox: [f32; 4],
|
||||
}
|
||||
|
||||
/// Intermediate parse result before the caller fills in text + page coords.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct CellSlot {
|
||||
pub row: usize,
|
||||
pub col: usize,
|
||||
pub rowspan: usize,
|
||||
pub colspan: usize,
|
||||
pub is_header: bool,
|
||||
/// Index into the parallel `cell_bboxes` array.
|
||||
pub bbox_idx: usize,
|
||||
}
|
||||
|
||||
/// Parse a sequence of SLANet structure tokens into ordered cell slots.
|
||||
///
|
||||
/// Token grammar (no real HTML parsing required):
|
||||
/// - Section markers: `<thead>`, `</thead>`, `<tbody>`, `</tbody>` and
|
||||
/// wrapper tokens (`<html>`, `<body>`, `<table>`, plus closing variants)
|
||||
/// are tracked or skipped.
|
||||
/// - Row markers: `<tr>` opens a new row, `</tr>` is informational.
|
||||
/// - Empty cell, single token: `<td></td>` or `<th></th>`.
|
||||
/// - Cell with attributes, multi-token sequence: `<td` (or `<th`), then
|
||||
/// attribute fragments like ` colspan="4"`, then `>`, then later `</td>`
|
||||
/// (or `</th>`). Cells get paired with the next bbox in document order.
|
||||
///
|
||||
/// Cells inside `<thead>` and any `<th>` cells are flagged as headers.
|
||||
/// rowspan/colspan attributes are honoured and prior-row rowspans push
|
||||
/// later-row cells to the right.
|
||||
pub(crate) fn parse_structure(tokens: &[String]) -> Vec<CellSlot> {
|
||||
let mut slots: Vec<CellSlot> = Vec::new();
|
||||
let mut occupied: HashSet<(usize, usize)> = HashSet::new();
|
||||
let mut row: usize = 0;
|
||||
let mut col: usize = 0;
|
||||
let mut bbox_idx: usize = 0;
|
||||
let mut in_thead = false;
|
||||
let mut started_first_row = false;
|
||||
|
||||
let mut i = 0;
|
||||
while i < tokens.len() {
|
||||
let tok = tokens[i].trim();
|
||||
match tok {
|
||||
"<thead>" => {
|
||||
in_thead = true;
|
||||
}
|
||||
"</thead>" => {
|
||||
in_thead = false;
|
||||
}
|
||||
"<tr>" => {
|
||||
if started_first_row {
|
||||
row += 1;
|
||||
}
|
||||
col = 0;
|
||||
started_first_row = true;
|
||||
}
|
||||
"<td></td>" | "<th></th>" => {
|
||||
let is_th = tok == "<th></th>";
|
||||
while occupied.contains(&(row, col)) {
|
||||
col += 1;
|
||||
}
|
||||
slots.push(CellSlot {
|
||||
row,
|
||||
col,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: in_thead || is_th,
|
||||
bbox_idx,
|
||||
});
|
||||
bbox_idx += 1;
|
||||
col += 1;
|
||||
}
|
||||
"<td" | "<th" => {
|
||||
let is_th = tok == "<th";
|
||||
let mut rowspan: usize = 1;
|
||||
let mut colspan: usize = 1;
|
||||
// Consume attribute fragments until we hit ">".
|
||||
i += 1;
|
||||
while i < tokens.len() && tokens[i].trim() != ">" {
|
||||
let attr = tokens[i].as_str();
|
||||
if let Some(v) = parse_int_attr(attr, "rowspan") {
|
||||
rowspan = v.max(1);
|
||||
} else if let Some(v) = parse_int_attr(attr, "colspan") {
|
||||
colspan = v.max(1);
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
// i now points at the `>` token (or off the end if malformed).
|
||||
while occupied.contains(&(row, col)) {
|
||||
col += 1;
|
||||
}
|
||||
slots.push(CellSlot {
|
||||
row,
|
||||
col,
|
||||
rowspan,
|
||||
colspan,
|
||||
is_header: in_thead || is_th,
|
||||
bbox_idx,
|
||||
});
|
||||
for r in row..row + rowspan {
|
||||
for c in col..col + colspan {
|
||||
occupied.insert((r, c));
|
||||
}
|
||||
}
|
||||
bbox_idx += 1;
|
||||
col += colspan;
|
||||
}
|
||||
// Wrapper / informational tokens — no-op.
|
||||
_ => {}
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
|
||||
slots
|
||||
}
|
||||
|
||||
/// Parse an attribute fragment like ` colspan="4"` or `rowspan='2'`.
|
||||
///
|
||||
/// Tolerates leading whitespace and either single or double quotes.
|
||||
fn parse_int_attr(s: &str, name: &str) -> Option<usize> {
|
||||
let trimmed = s.trim();
|
||||
if !trimmed.starts_with(name) {
|
||||
return None;
|
||||
}
|
||||
let rest = trimmed[name.len()..].trim_start();
|
||||
let rest = rest.strip_prefix('=')?.trim_start();
|
||||
let value = rest
|
||||
.trim_start_matches(['"', '\''])
|
||||
.trim_end_matches(['"', '\'']);
|
||||
value.parse().ok()
|
||||
}
|
||||
|
||||
/// Convert a SLANet polygon (4 or 8 elements) into an axis-aligned
|
||||
/// `[x1, y1, x2, y2]` rect.
|
||||
///
|
||||
/// 8-element form: `[x1,y1, x2,y1, x2,y2, x1,y2]` (4 corners). We ignore the
|
||||
/// implicit corner order and just take min/max so rotated polygons collapse
|
||||
/// to a sane bounding box.
|
||||
///
|
||||
/// 4-element form: `[x1, y1, x2, y2]` (axis-aligned, older SLANet variants).
|
||||
pub(crate) fn polygon_to_aabb(coords: &[f32]) -> Option<[f32; 4]> {
|
||||
match coords.len() {
|
||||
4 => {
|
||||
let x1 = coords[0].min(coords[2]);
|
||||
let y1 = coords[1].min(coords[3]);
|
||||
let x2 = coords[0].max(coords[2]);
|
||||
let y2 = coords[1].max(coords[3]);
|
||||
Some([x1, y1, x2, y2])
|
||||
}
|
||||
8 => {
|
||||
let xs = [coords[0], coords[2], coords[4], coords[6]];
|
||||
let ys = [coords[1], coords[3], coords[5], coords[7]];
|
||||
let x1 = xs.iter().copied().fold(f32::INFINITY, f32::min);
|
||||
let y1 = ys.iter().copied().fold(f32::INFINITY, f32::min);
|
||||
let x2 = xs.iter().copied().fold(f32::NEG_INFINITY, f32::max);
|
||||
let y2 = ys.iter().copied().fold(f32::NEG_INFINITY, f32::max);
|
||||
if x1.is_finite() && y1.is_finite() && x2.is_finite() && y2.is_finite() {
|
||||
Some([x1, y1, x2, y2])
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a cell rect from crop image-pixel space to page PDF-points
|
||||
/// (top-left origin), given the crop's PDF-point offset on the page and the
|
||||
/// DPI the crop image was rendered at.
|
||||
pub(crate) fn cell_px_to_page_pt(
|
||||
cell_px: [f32; 4],
|
||||
render_dpi: f32,
|
||||
crop_origin_pt: [f32; 2],
|
||||
) -> [f32; 4] {
|
||||
let pt_per_px = if render_dpi > 0.0 {
|
||||
72.0 / render_dpi
|
||||
} else {
|
||||
1.0
|
||||
};
|
||||
let [x_off, y_off] = crop_origin_pt;
|
||||
[
|
||||
cell_px[0] * pt_per_px + x_off,
|
||||
cell_px[1] * pt_per_px + y_off,
|
||||
cell_px[2] * pt_per_px + x_off,
|
||||
cell_px[3] * pt_per_px + y_off,
|
||||
]
|
||||
}
|
||||
|
||||
/// Refine TSR cell bboxes into non-overlapping row/column bands.
|
||||
///
|
||||
/// SLANet-style bboxes are often plausible but too tall on dense borderless
|
||||
/// tables. Native PDF text assignment is more reliable when each parsed row
|
||||
/// owns the band between neighboring row centers instead of the full model box.
|
||||
pub(crate) fn normalize_cell_bands(cells: &mut [StructuredCell]) {
|
||||
if cells.len() < 2 {
|
||||
return;
|
||||
}
|
||||
|
||||
let row_bands = derive_axis_bands(cells, Axis::Y);
|
||||
let col_bands = derive_axis_bands(cells, Axis::X);
|
||||
|
||||
for cell in cells {
|
||||
let row_end = cell.row + cell.rowspan.max(1).saturating_sub(1);
|
||||
if let (Some(&(y1, _)), Some(&(_, y2))) =
|
||||
(row_bands.get(&cell.row), row_bands.get(&row_end))
|
||||
{
|
||||
let clamped_y1 = cell.page_pt_bbox[1].max(y1);
|
||||
let clamped_y2 = cell.page_pt_bbox[3].min(y2);
|
||||
if clamped_y1 < clamped_y2 {
|
||||
cell.page_pt_bbox[1] = clamped_y1;
|
||||
cell.page_pt_bbox[3] = clamped_y2;
|
||||
}
|
||||
}
|
||||
|
||||
let col_end = cell.col + cell.colspan.max(1).saturating_sub(1);
|
||||
if let (Some(&(x1, _)), Some(&(_, x2))) =
|
||||
(col_bands.get(&cell.col), col_bands.get(&col_end))
|
||||
{
|
||||
let clamped_x1 = cell.page_pt_bbox[0].max(x1);
|
||||
let clamped_x2 = cell.page_pt_bbox[2].min(x2);
|
||||
if clamped_x1 < clamped_x2 {
|
||||
cell.page_pt_bbox[0] = clamped_x1;
|
||||
cell.page_pt_bbox[2] = clamped_x2;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
enum Axis {
|
||||
X,
|
||||
Y,
|
||||
}
|
||||
|
||||
fn derive_axis_bands(cells: &[StructuredCell], axis: Axis) -> HashMap<usize, (f32, f32)> {
|
||||
let mut by_index: HashMap<usize, Vec<(f32, f32)>> = HashMap::new();
|
||||
|
||||
// Prefer non-spanning cells so colspan/rowspan boxes do not skew a single
|
||||
// column/row center. If an axis has no non-spanning examples for an index,
|
||||
// fall back to anchored cells below.
|
||||
for cell in cells {
|
||||
let span = match axis {
|
||||
Axis::X => cell.colspan.max(1),
|
||||
Axis::Y => cell.rowspan.max(1),
|
||||
};
|
||||
if span == 1 {
|
||||
let idx = match axis {
|
||||
Axis::X => cell.col,
|
||||
Axis::Y => cell.row,
|
||||
};
|
||||
by_index
|
||||
.entry(idx)
|
||||
.or_default()
|
||||
.push(axis_bounds(cell.page_pt_bbox, axis));
|
||||
}
|
||||
}
|
||||
|
||||
for cell in cells {
|
||||
let idx = match axis {
|
||||
Axis::X => cell.col,
|
||||
Axis::Y => cell.row,
|
||||
};
|
||||
if !by_index.contains_key(&idx) {
|
||||
by_index
|
||||
.entry(idx)
|
||||
.or_default()
|
||||
.push(axis_bounds(cell.page_pt_bbox, axis));
|
||||
}
|
||||
}
|
||||
|
||||
let mut rows: Vec<(usize, f32, f32, f32)> = by_index
|
||||
.into_iter()
|
||||
.filter_map(|(idx, bounds)| {
|
||||
let mut min_edge = f32::INFINITY;
|
||||
let mut max_edge = f32::NEG_INFINITY;
|
||||
let mut center_sum = 0.0;
|
||||
let mut count = 0usize;
|
||||
for (lo, hi) in bounds {
|
||||
if lo.is_finite() && hi.is_finite() && lo < hi {
|
||||
min_edge = min_edge.min(lo);
|
||||
max_edge = max_edge.max(hi);
|
||||
center_sum += (lo + hi) * 0.5;
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
(count > 0).then_some((idx, center_sum / count as f32, min_edge, max_edge))
|
||||
})
|
||||
.collect();
|
||||
|
||||
if rows.len() < 2 {
|
||||
return rows
|
||||
.into_iter()
|
||||
.map(|(idx, _center, lo, hi)| (idx, (lo, hi)))
|
||||
.collect();
|
||||
}
|
||||
|
||||
rows.sort_by_key(|(idx, _, _, _)| *idx);
|
||||
|
||||
let mut bands = HashMap::new();
|
||||
for i in 0..rows.len() {
|
||||
let (idx, _center, min_edge, max_edge) = rows[i];
|
||||
let lo = if i == 0 {
|
||||
min_edge
|
||||
} else {
|
||||
(rows[i - 1].1 + rows[i].1) * 0.5
|
||||
};
|
||||
let hi = if i + 1 == rows.len() {
|
||||
max_edge
|
||||
} else {
|
||||
(rows[i].1 + rows[i + 1].1) * 0.5
|
||||
};
|
||||
if lo.is_finite() && hi.is_finite() && lo < hi {
|
||||
bands.insert(idx, (lo, hi));
|
||||
}
|
||||
}
|
||||
|
||||
bands
|
||||
}
|
||||
|
||||
fn axis_bounds(bbox: [f32; 4], axis: Axis) -> (f32, f32) {
|
||||
match axis {
|
||||
Axis::X => (bbox[0].min(bbox[2]), bbox[0].max(bbox[2])),
|
||||
Axis::Y => (bbox[1].min(bbox[3]), bbox[1].max(bbox[3])),
|
||||
}
|
||||
}
|
||||
|
||||
/// Sanitize cell text for inclusion in a markdown pipe-table cell:
|
||||
/// collapse whitespace runs, drop newlines/tabs (cells must be one line),
|
||||
/// and escape pipes that would otherwise break the table.
|
||||
fn sanitize_cell(text: &str) -> String {
|
||||
let mut s = String::with_capacity(text.len());
|
||||
let mut prev_space = false;
|
||||
for c in text.chars() {
|
||||
match c {
|
||||
'|' => {
|
||||
s.push_str("\\|");
|
||||
prev_space = false;
|
||||
}
|
||||
'\n' | '\r' | '\t' | ' ' => {
|
||||
if !prev_space {
|
||||
s.push(' ');
|
||||
}
|
||||
prev_space = true;
|
||||
}
|
||||
other => {
|
||||
s.push(other);
|
||||
prev_space = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
s.trim().to_string()
|
||||
}
|
||||
|
||||
/// Render a list of explicitly-positioned cells as a markdown pipe-table.
|
||||
///
|
||||
/// Grid dimensions are inferred from the cells' (row, col, rowspan, colspan)
|
||||
/// extents. A cell with colspan/rowspan > 1 is rendered in its top-left
|
||||
/// position; the absorbed grid positions are emitted as empty cells so the
|
||||
/// markdown stays a valid rectangular grid that downstream readers can
|
||||
/// column-count correctly.
|
||||
///
|
||||
/// The separator row (`|---|...|`) is emitted after the **last** row that
|
||||
/// contains a header cell (`is_header == true`). When no cells are flagged
|
||||
/// as headers — e.g. the upstream TSR model didn't emit `<thead>`/`<th>` —
|
||||
/// the separator falls back to "after row 0" so the output is still a
|
||||
/// valid pipe-table.
|
||||
pub fn cells_to_markdown(cells: &[StructuredCell]) -> String {
|
||||
if cells.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
let num_rows = cells
|
||||
.iter()
|
||||
.map(|c| c.row + c.rowspan.max(1))
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
let num_cols = cells
|
||||
.iter()
|
||||
.map(|c| c.col + c.colspan.max(1))
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
if num_rows == 0 || num_cols == 0 {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Separator goes after the last header row, falling back to row 0 when
|
||||
// no header cells exist. Clamped into range so a malformed cell with
|
||||
// row >= num_rows can't push it past the table.
|
||||
let separator_after_row = cells
|
||||
.iter()
|
||||
.filter(|c| c.is_header)
|
||||
.map(|c| c.row)
|
||||
.max()
|
||||
.unwrap_or(0)
|
||||
.min(num_rows.saturating_sub(1));
|
||||
|
||||
let mut grid: Vec<Vec<String>> = vec![vec![String::new(); num_cols]; num_rows];
|
||||
for cell in cells {
|
||||
if cell.row < num_rows && cell.col < num_cols {
|
||||
grid[cell.row][cell.col] = sanitize_cell(&cell.text);
|
||||
}
|
||||
}
|
||||
|
||||
let mut output = String::new();
|
||||
for (row_idx, row) in grid.iter().enumerate() {
|
||||
output.push('|');
|
||||
for cell in row {
|
||||
output.push_str(cell);
|
||||
output.push('|');
|
||||
}
|
||||
output.push('\n');
|
||||
if row_idx == separator_after_row {
|
||||
output.push('|');
|
||||
for _ in 0..num_cols {
|
||||
output.push_str("---|");
|
||||
}
|
||||
output.push('\n');
|
||||
}
|
||||
}
|
||||
output
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn t(s: &str) -> String {
|
||||
s.to_string()
|
||||
}
|
||||
|
||||
/// Tokens for the synthetic 3×3 grid example (one colspan-4 row + two
|
||||
/// data rows of 4 cells each = 9 cells total, 3 rows × 4 cols).
|
||||
fn synthetic_3x3_tokens() -> Vec<String> {
|
||||
vec![
|
||||
"<html>",
|
||||
"<body>",
|
||||
"<table>",
|
||||
"<tbody>",
|
||||
"<tr>",
|
||||
"<td",
|
||||
" colspan=\"4\"",
|
||||
">",
|
||||
"</td>",
|
||||
"</tr>",
|
||||
"<tr>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"<tr>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"</tbody>",
|
||||
"</table>",
|
||||
"</body>",
|
||||
"</html>",
|
||||
]
|
||||
.into_iter()
|
||||
.map(t)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Bboxes for the synthetic 3×3 grid (8-element polygon form), all
|
||||
/// within a 400×120 px crop.
|
||||
fn synthetic_3x3_bboxes() -> Vec<Vec<f32>> {
|
||||
vec![
|
||||
vec![3.0, 2.0, 395.0, 2.0, 396.0, 59.0, 3.0, 59.0],
|
||||
vec![26.0, 62.0, 140.0, 62.0, 141.0, 120.0, 26.0, 120.0],
|
||||
vec![149.0, 64.0, 248.0, 64.0, 248.0, 119.0, 149.0, 119.0],
|
||||
vec![257.0, 64.0, 350.0, 64.0, 350.0, 119.0, 257.0, 119.0],
|
||||
vec![359.0, 64.0, 395.0, 64.0, 395.0, 119.0, 359.0, 119.0],
|
||||
vec![26.0, 122.0, 140.0, 122.0, 140.0, 178.0, 26.0, 178.0],
|
||||
vec![149.0, 124.0, 248.0, 124.0, 248.0, 179.0, 149.0, 179.0],
|
||||
vec![257.0, 124.0, 350.0, 124.0, 350.0, 179.0, 257.0, 179.0],
|
||||
vec![359.0, 124.0, 395.0, 124.0, 395.0, 179.0, 359.0, 179.0],
|
||||
]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_structure_synthetic_3x3() {
|
||||
let tokens = synthetic_3x3_tokens();
|
||||
let slots = parse_structure(&tokens);
|
||||
|
||||
assert_eq!(slots.len(), 9, "should parse 9 cells");
|
||||
|
||||
// Cell 0: row 0 col 0, colspan 4
|
||||
assert_eq!(slots[0].row, 0);
|
||||
assert_eq!(slots[0].col, 0);
|
||||
assert_eq!(slots[0].colspan, 4);
|
||||
assert_eq!(slots[0].rowspan, 1);
|
||||
|
||||
// Cells 1..5: row 1, cols 0..3
|
||||
for (i, slot) in slots.iter().enumerate().skip(1).take(4) {
|
||||
assert_eq!(slot.row, 1, "cell {i}: row should be 1");
|
||||
assert_eq!(slot.col, i - 1, "cell {i}: col should be {}", i - 1);
|
||||
assert_eq!(slot.colspan, 1);
|
||||
assert_eq!(slot.rowspan, 1);
|
||||
}
|
||||
|
||||
// Cells 5..9: row 2, cols 0..3
|
||||
for (i, slot) in slots.iter().enumerate().skip(5).take(4) {
|
||||
assert_eq!(slot.row, 2, "cell {i}: row should be 2");
|
||||
assert_eq!(slot.col, i - 5);
|
||||
assert_eq!(slot.colspan, 1);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn polygon_to_aabb_8elt() {
|
||||
// Synthetic cell bbox 0
|
||||
let coords = vec![3.0, 2.0, 395.0, 2.0, 396.0, 59.0, 3.0, 59.0];
|
||||
let aabb = polygon_to_aabb(&coords).unwrap();
|
||||
assert_eq!(aabb, [3.0, 2.0, 396.0, 59.0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn polygon_to_aabb_4elt() {
|
||||
let coords = vec![5.0, 10.0, 50.0, 60.0];
|
||||
let aabb = polygon_to_aabb(&coords).unwrap();
|
||||
assert_eq!(aabb, [5.0, 10.0, 50.0, 60.0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn polygon_to_aabb_4elt_unordered() {
|
||||
// Caller may pass corners in any order; min/max should normalise.
|
||||
let coords = vec![50.0, 60.0, 5.0, 10.0];
|
||||
let aabb = polygon_to_aabb(&coords).unwrap();
|
||||
assert_eq!(aabb, [5.0, 10.0, 50.0, 60.0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn polygon_to_aabb_invalid_len() {
|
||||
assert!(polygon_to_aabb(&[1.0, 2.0, 3.0]).is_none());
|
||||
assert!(polygon_to_aabb(&[1.0; 6]).is_none());
|
||||
assert!(polygon_to_aabb(&[]).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn synthetic_3x3_aabbs_inside_crop() {
|
||||
// All 9 bboxes should produce valid (x1<x2, y1<y2) rects within the
|
||||
// crop bounds (400 wide, ~180 tall by inspection of the fixture).
|
||||
let bboxes = synthetic_3x3_bboxes();
|
||||
assert_eq!(bboxes.len(), 9);
|
||||
for (i, bb) in bboxes.iter().enumerate() {
|
||||
let aabb = polygon_to_aabb(bb).unwrap_or_else(|| panic!("bbox {i} invalid"));
|
||||
assert!(aabb[0] < aabb[2], "bbox {i}: x1 < x2");
|
||||
assert!(aabb[1] < aabb[3], "bbox {i}: y1 < y2");
|
||||
assert!(aabb[0] >= 0.0 && aabb[2] <= 500.0, "bbox {i}: within crop");
|
||||
assert!(aabb[1] >= 0.0 && aabb[3] <= 200.0, "bbox {i}: within crop");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_cell_bands_splits_overlapping_slanet_rows() {
|
||||
let mut cells = vec![
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: true,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [10.0, 100.0, 90.0, 120.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: true,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [90.0, 100.0, 170.0, 120.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [10.0, 116.0, 90.0, 136.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [90.0, 116.0, 170.0, 136.0],
|
||||
},
|
||||
];
|
||||
|
||||
normalize_cell_bands(&mut cells);
|
||||
|
||||
assert_eq!(cells[0].page_pt_bbox[3], cells[2].page_pt_bbox[1]);
|
||||
assert_eq!(cells[1].page_pt_bbox[3], cells[3].page_pt_bbox[1]);
|
||||
assert!(
|
||||
(cells[0].page_pt_bbox[3] - 118.0).abs() < 0.01,
|
||||
"row separator should be midpoint between row centers: {:?}",
|
||||
cells
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_cell_bands_preserves_colspan_extent() {
|
||||
let mut cells = vec![
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 2,
|
||||
is_header: true,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [8.0, 80.0, 172.0, 98.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [10.0, 96.0, 90.0, 114.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: String::new(),
|
||||
page_pt_bbox: [88.0, 96.0, 170.0, 114.0],
|
||||
},
|
||||
];
|
||||
|
||||
normalize_cell_bands(&mut cells);
|
||||
|
||||
assert!(
|
||||
cells[0].page_pt_bbox[0] <= cells[1].page_pt_bbox[0],
|
||||
"spanning cell should retain the first column's left edge"
|
||||
);
|
||||
assert!(
|
||||
cells[0].page_pt_bbox[2] >= cells[2].page_pt_bbox[2],
|
||||
"spanning cell should retain the last column's right edge"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_int_attr_basic() {
|
||||
assert_eq!(parse_int_attr(" colspan=\"4\"", "colspan"), Some(4));
|
||||
assert_eq!(parse_int_attr(" rowspan=\"2\"", "rowspan"), Some(2));
|
||||
assert_eq!(parse_int_attr("colspan='3'", "colspan"), Some(3));
|
||||
assert_eq!(parse_int_attr(" colspan=\"4\"", "rowspan"), None);
|
||||
assert_eq!(parse_int_attr(" class=\"foo\"", "colspan"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_structure_rowspan_pushes_next_row_right() {
|
||||
// <tr><td rowspan="2">A</td><td>B</td></tr><tr><td>C</td></tr>
|
||||
// Expected: A at (0,0), B at (0,1), C at (1,1) — col 0 of row 1
|
||||
// is occupied by A's rowspan.
|
||||
let tokens: Vec<String> = vec![
|
||||
"<table>",
|
||||
"<tbody>",
|
||||
"<tr>",
|
||||
"<td",
|
||||
" rowspan=\"2\"",
|
||||
">",
|
||||
"</td>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"<tr>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"</tbody>",
|
||||
"</table>",
|
||||
]
|
||||
.into_iter()
|
||||
.map(t)
|
||||
.collect();
|
||||
|
||||
let slots = parse_structure(&tokens);
|
||||
assert_eq!(slots.len(), 3);
|
||||
assert_eq!((slots[0].row, slots[0].col), (0, 0));
|
||||
assert_eq!(slots[0].rowspan, 2);
|
||||
assert_eq!((slots[1].row, slots[1].col), (0, 1));
|
||||
// C should be at (1, 1) because (1, 0) is occupied by A's rowspan.
|
||||
assert_eq!((slots[2].row, slots[2].col), (1, 1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_structure_thead_marks_headers() {
|
||||
// <thead><tr><th>H1</th><th>H2</th></tr></thead>
|
||||
// <tbody><tr><td>D1</td><td>D2</td></tr></tbody>
|
||||
let tokens: Vec<String> = vec![
|
||||
"<table>",
|
||||
"<thead>",
|
||||
"<tr>",
|
||||
"<th></th>",
|
||||
"<th></th>",
|
||||
"</tr>",
|
||||
"</thead>",
|
||||
"<tbody>",
|
||||
"<tr>",
|
||||
"<td></td>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"</tbody>",
|
||||
"</table>",
|
||||
]
|
||||
.into_iter()
|
||||
.map(t)
|
||||
.collect();
|
||||
|
||||
let slots = parse_structure(&tokens);
|
||||
assert_eq!(slots.len(), 4);
|
||||
assert!(slots[0].is_header && slots[1].is_header);
|
||||
assert!(!slots[2].is_header && !slots[3].is_header);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_structure_th_outside_thead_still_header() {
|
||||
// A row-header style: leading <th> in tbody.
|
||||
let tokens: Vec<String> = vec![
|
||||
"<table>",
|
||||
"<tbody>",
|
||||
"<tr>",
|
||||
"<th></th>",
|
||||
"<td></td>",
|
||||
"</tr>",
|
||||
"</tbody>",
|
||||
"</table>",
|
||||
]
|
||||
.into_iter()
|
||||
.map(t)
|
||||
.collect();
|
||||
|
||||
let slots = parse_structure(&tokens);
|
||||
assert_eq!(slots.len(), 2);
|
||||
assert!(slots[0].is_header);
|
||||
assert!(!slots[1].is_header);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_structure_th_with_attrs() {
|
||||
let tokens: Vec<String> = vec![
|
||||
"<table>",
|
||||
"<thead>",
|
||||
"<tr>",
|
||||
"<th",
|
||||
" colspan=\"2\"",
|
||||
">",
|
||||
"</th>",
|
||||
"</tr>",
|
||||
"</thead>",
|
||||
"</table>",
|
||||
]
|
||||
.into_iter()
|
||||
.map(t)
|
||||
.collect();
|
||||
let slots = parse_structure(&tokens);
|
||||
assert_eq!(slots.len(), 1);
|
||||
assert_eq!(slots[0].colspan, 2);
|
||||
assert!(slots[0].is_header);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cells_to_markdown_synthetic_3x3() {
|
||||
// Build the cells the parser would produce for the synthetic grid,
|
||||
// and provide some sample text so we can sanity-check output.
|
||||
let cells = vec![
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 4,
|
||||
is_header: false,
|
||||
text: "Title".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "a".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "b".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 2,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "c".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 1,
|
||||
col: 3,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "d".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
];
|
||||
|
||||
let md = cells_to_markdown(&cells);
|
||||
// Header row contains the spanning cell text in col 0 and pads to 4 cols.
|
||||
// Absorbed-by-colspan positions render as empty cells (no padding).
|
||||
assert!(md.starts_with("|Title||||\n"), "got: {md}");
|
||||
assert!(md.contains("|---|---|---|---|\n"));
|
||||
assert!(md.contains("|a|b|c|d|\n"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cells_to_markdown_escapes_pipes() {
|
||||
let cells = vec![
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "a|b".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
StructuredCell {
|
||||
row: 0,
|
||||
col: 1,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "x".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
},
|
||||
];
|
||||
let md = cells_to_markdown(&cells);
|
||||
assert!(md.contains("|a\\|b|x|"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cells_to_markdown_collapses_whitespace_and_newlines() {
|
||||
let cells = vec![StructuredCell {
|
||||
row: 0,
|
||||
col: 0,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header: false,
|
||||
text: "foo \n bar\tbaz".into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
}];
|
||||
let md = cells_to_markdown(&cells);
|
||||
assert!(md.contains("|foo bar baz|"));
|
||||
}
|
||||
|
||||
fn cell(row: usize, col: usize, is_header: bool, text: &str) -> StructuredCell {
|
||||
StructuredCell {
|
||||
row,
|
||||
col,
|
||||
rowspan: 1,
|
||||
colspan: 1,
|
||||
is_header,
|
||||
text: text.into(),
|
||||
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cells_to_markdown_separator_after_last_header_row() {
|
||||
// Two-row header (a multi-row thead), then two body rows. Separator
|
||||
// should land after row 1 (the LAST header row), not after row 0.
|
||||
let cells = vec![
|
||||
cell(0, 0, true, "H0a"),
|
||||
cell(0, 1, true, "H0b"),
|
||||
cell(1, 0, true, "H1a"),
|
||||
cell(1, 1, true, "H1b"),
|
||||
cell(2, 0, false, "d0a"),
|
||||
cell(2, 1, false, "d0b"),
|
||||
cell(3, 0, false, "d1a"),
|
||||
cell(3, 1, false, "d1b"),
|
||||
];
|
||||
let md = cells_to_markdown(&cells);
|
||||
let expected = "|H0a|H0b|\n|H1a|H1b|\n|---|---|\n|d0a|d0b|\n|d1a|d1b|\n";
|
||||
assert_eq!(md, expected, "got: {md}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cells_to_markdown_separator_when_row_0_not_header() {
|
||||
// Row 0 is not flagged as a header but row 1 is. Separator should
|
||||
// follow row 1 (the header), demonstrating that we don't blindly
|
||||
// emit after row 0.
|
||||
let cells = vec![
|
||||
cell(0, 0, false, "x0a"),
|
||||
cell(0, 1, false, "x0b"),
|
||||
cell(1, 0, true, "Hdr1"),
|
||||
cell(1, 1, true, "Hdr2"),
|
||||
cell(2, 0, false, "data1"),
|
||||
cell(2, 1, false, "data2"),
|
||||
];
|
||||
let md = cells_to_markdown(&cells);
|
||||
// Confirm the separator is NOT after row 0.
|
||||
assert!(!md.starts_with("|x0a|x0b|\n|---|"), "got: {md}");
|
||||
// Confirm it IS after row 1.
|
||||
assert!(
|
||||
md.contains("|Hdr1|Hdr2|\n|---|---|\n|data1|data2|"),
|
||||
"got: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cells_to_markdown_no_headers_falls_back_to_row_0() {
|
||||
// No header cells at all — fallback: separator after row 0 so the
|
||||
// output is still a valid markdown pipe-table.
|
||||
let cells = vec![
|
||||
cell(0, 0, false, "a"),
|
||||
cell(0, 1, false, "b"),
|
||||
cell(1, 0, false, "c"),
|
||||
cell(1, 1, false, "d"),
|
||||
];
|
||||
let md = cells_to_markdown(&cells);
|
||||
assert_eq!(md, "|a|b|\n|---|---|\n|c|d|\n");
|
||||
}
|
||||
}
|
||||
@@ -883,6 +883,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1002,6 +1004,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -1078,6 +1082,8 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
+453
-24
@@ -329,7 +329,7 @@ impl ToUnicodeCMap {
|
||||
if let (Some(start), Some(end), Some(base)) = (
|
||||
parse_hex_u16(&start_hex),
|
||||
parse_hex_u16(&end_hex),
|
||||
parse_hex_u32(&base_hex),
|
||||
hex_to_unicode_scalar(&base_hex),
|
||||
) {
|
||||
self.ranges.push((start, end, base));
|
||||
}
|
||||
@@ -520,6 +520,18 @@ impl ToUnicodeCMap {
|
||||
}
|
||||
}
|
||||
|
||||
/// Get the maximum source CID across all mappings (char_map + ranges).
|
||||
fn max_source_cid(&self) -> Option<u16> {
|
||||
let char_max = self.char_map.keys().copied().max();
|
||||
let range_max = self.ranges.iter().map(|&(_, end, _)| end).max();
|
||||
match (char_max, range_max) {
|
||||
(Some(a), Some(b)) => Some(a.max(b)),
|
||||
(a @ Some(_), None) => a,
|
||||
(None, b @ Some(_)) => b,
|
||||
(None, None) => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Remap a CMap that references pre-subsetting GIDs to sequential post-subsetting GIDs.
|
||||
/// Collects all source CIDs, sorts them, and reassigns to 1, 2, 3, ...
|
||||
pub fn remap_to_sequential(&self) -> ToUnicodeCMap {
|
||||
@@ -563,32 +575,86 @@ fn parse_hex_u16(hex: &str) -> Option<u16> {
|
||||
u16::from_str_radix(hex.trim(), 16).ok()
|
||||
}
|
||||
|
||||
/// Parse a hex string to u32
|
||||
fn parse_hex_u32(hex: &str) -> Option<u32> {
|
||||
u32::from_str_radix(hex.trim(), 16).ok()
|
||||
}
|
||||
|
||||
/// Convert a hex string to a Unicode string
|
||||
/// Handles both 2-byte (BMP) and 4-byte (supplementary) codepoints
|
||||
/// Convert a ToUnicode destination hex string to Unicode.
|
||||
///
|
||||
/// PDF ToUnicode destinations are UTF-16BE strings. Supplementary-plane
|
||||
/// characters are encoded as surrogate pairs, so treating each 4-hex chunk as
|
||||
/// a scalar drops emoji like D83CDF1F.
|
||||
fn hex_to_unicode_string(hex: &str) -> Option<String> {
|
||||
let hex = hex.trim();
|
||||
let mut result = String::new();
|
||||
|
||||
// Process 4 hex digits at a time
|
||||
let mut i = 0;
|
||||
while i + 4 <= hex.len() {
|
||||
if let Ok(cp) = u32::from_str_radix(&hex[i..i + 4], 16) {
|
||||
if let Some(c) = char::from_u32(cp) {
|
||||
result.push(c);
|
||||
}
|
||||
}
|
||||
i += 4;
|
||||
let hex: String = hex.chars().filter(|ch| !ch.is_ascii_whitespace()).collect();
|
||||
if hex.is_empty() || !hex.len().is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
|
||||
if result.is_empty() {
|
||||
None
|
||||
let bytes: Option<Vec<u8>> = (0..hex.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
|
||||
.collect();
|
||||
let bytes = bytes?;
|
||||
|
||||
if bytes.len().is_multiple_of(2) {
|
||||
let units: Vec<u16> = bytes
|
||||
.chunks_exact(2)
|
||||
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
|
||||
.collect();
|
||||
if let Ok(result) = String::from_utf16(&units) {
|
||||
if !result.is_empty() {
|
||||
return Some(normalize_tounicode_destination(result));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Be permissive for non-standard one-byte destinations.
|
||||
if bytes.len() == 1 {
|
||||
let ch = bytes[0] as char;
|
||||
if !ch.is_control() || ch == '\t' || ch == '\n' {
|
||||
return Some(ch.to_string());
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
fn normalize_tounicode_destination(text: String) -> String {
|
||||
let is_multi_char = text.chars().nth(1).is_some();
|
||||
|
||||
// Some malformed producer CMaps put a list of alternative whitespace or
|
||||
// hyphen codepoints into one destination. Keep ordinary multi-character
|
||||
// mappings intact unless that malformed signature is present.
|
||||
if is_multi_char
|
||||
&& text.chars().all(char::is_whitespace)
|
||||
&& text.chars().any(|ch| matches!(ch, '\t' | '\n' | '\r'))
|
||||
{
|
||||
return if text.contains('\t') {
|
||||
"\t".to_string()
|
||||
} else {
|
||||
" ".to_string()
|
||||
};
|
||||
}
|
||||
|
||||
if is_multi_char
|
||||
&& text.contains('\u{00ad}')
|
||||
&& text.chars().all(|ch| {
|
||||
matches!(
|
||||
ch,
|
||||
'-' | '\u{00ad}' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2212}'
|
||||
)
|
||||
})
|
||||
{
|
||||
return "-".to_string();
|
||||
}
|
||||
|
||||
text
|
||||
}
|
||||
|
||||
fn hex_to_unicode_scalar(hex: &str) -> Option<u32> {
|
||||
let text = hex_to_unicode_string(hex)?;
|
||||
let mut chars = text.chars();
|
||||
let ch = chars.next()?;
|
||||
if chars.next().is_none() {
|
||||
Some(ch as u32)
|
||||
} else {
|
||||
Some(result)
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
@@ -657,6 +723,81 @@ fn get_w_array_start_cid(cid_font_dict: &lopdf::Dictionary, doc: &Document) -> O
|
||||
}
|
||||
}
|
||||
|
||||
/// Return true if the CIDFont's W (widths) array explicitly covers the given CID.
|
||||
///
|
||||
/// The W array uses two formats (PDF 32000-1:2008, §9.7.4.3):
|
||||
/// 1. `c [w1 w2 ... wn]` — widths for CIDs c, c+1, ..., c+n-1
|
||||
/// 2. `c_first c_last w` — CIDs c_first..c_last all have width w
|
||||
fn w_array_covers_cid(cid_font_dict: &lopdf::Dictionary, doc: &Document, target: u16) -> bool {
|
||||
let Ok(w_obj) = cid_font_dict.get(b"W") else {
|
||||
return false;
|
||||
};
|
||||
let arr = match w_obj {
|
||||
Object::Array(arr) => arr,
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Array(arr)) => arr,
|
||||
_ => return false,
|
||||
},
|
||||
_ => return false,
|
||||
};
|
||||
|
||||
let resolve_int = |o: &Object| -> Option<i64> {
|
||||
match o {
|
||||
Object::Integer(n) => Some(*n),
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Integer(n)) => Some(*n),
|
||||
_ => None,
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
};
|
||||
|
||||
let resolve_arr = |o: &Object| -> Option<Vec<Object>> {
|
||||
match o {
|
||||
Object::Array(a) => Some(a.clone()),
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Array(a)) => Some(a.clone()),
|
||||
_ => None,
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
};
|
||||
|
||||
let target = target as i64;
|
||||
let mut i = 0usize;
|
||||
while i < arr.len() {
|
||||
let Some(first) = resolve_int(&arr[i]) else {
|
||||
break;
|
||||
};
|
||||
i += 1;
|
||||
if i >= arr.len() {
|
||||
break;
|
||||
}
|
||||
// Peek at arr[i] to decide format.
|
||||
if let Some(widths) = resolve_arr(&arr[i]) {
|
||||
// Format 1: c [w1 ... wn]
|
||||
let last = first + widths.len() as i64 - 1;
|
||||
if target >= first && target <= last {
|
||||
return true;
|
||||
}
|
||||
i += 1;
|
||||
} else if let Some(last) = resolve_int(&arr[i]) {
|
||||
// Format 2: c_first c_last w
|
||||
i += 1;
|
||||
if i < arr.len() {
|
||||
i += 1; // skip the width value
|
||||
}
|
||||
if target >= first && target <= last {
|
||||
return true;
|
||||
}
|
||||
} else {
|
||||
// Unknown token — abort parsing safely
|
||||
break;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Extract CIDToGIDMap as a vector of GIDs (u16) indexed by CID.
|
||||
fn get_cid_to_gid_map(cid_font_dict: &lopdf::Dictionary, doc: &Document) -> Option<Vec<u16>> {
|
||||
let obj = cid_font_dict.get(b"CIDToGIDMap").ok()?;
|
||||
@@ -752,6 +893,20 @@ fn try_remap_subset_cmap(
|
||||
_ => return (cmap, None),
|
||||
};
|
||||
|
||||
// If the W array actually covers the CMap's max source CID, the CMap is
|
||||
// aligned with the font — no sequential renumbering happened. A sparse W
|
||||
// array starting at CID 0 (for .notdef) with additional high-CID entries
|
||||
// matching the CMap is the normal subset layout, not a mismatch.
|
||||
if let Some(max_cid) = cmap.max_source_cid() {
|
||||
if w_array_covers_cid(cid_font_dict, doc, max_cid) {
|
||||
debug!(
|
||||
"Subset remap skipped for obj={}: W array covers CMap max CID {}",
|
||||
obj_num, max_cid
|
||||
);
|
||||
return (cmap, None);
|
||||
}
|
||||
}
|
||||
|
||||
debug!(
|
||||
"Subset GID mismatch detected for obj={}: W starts at CID {}, CMap min CID {}. Remapping to sequential.",
|
||||
obj_num, w_start, min_cid
|
||||
@@ -1650,7 +1805,7 @@ fn merge_cmaps(mut base: ToUnicodeCMap, overlay: ToUnicodeCMap) -> ToUnicodeCMap
|
||||
///
|
||||
/// Returns true if the median CID is >= 0x41 (letter 'A'), indicating
|
||||
/// the PDF generator likely used Unicode codepoints as CIDs.
|
||||
fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) -> bool {
|
||||
pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) -> bool {
|
||||
let w_arr = match cid_font_dict.get(b"W").ok() {
|
||||
Some(Object::Array(arr)) => arr,
|
||||
_ => return false,
|
||||
@@ -2506,6 +2661,97 @@ endbfrange
|
||||
assert_eq!(cmap.lookup(0x0005), Some("C".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_surrogate_pair_emoji() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<16> <D83CDF1F>
|
||||
<9D> <D83CDFAD>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.code_byte_length, 1);
|
||||
assert_eq!(cmap.lookup(0x16), Some("🌟".to_string()));
|
||||
assert_eq!(cmap.lookup(0x9D), Some("🎭".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfrange_surrogate_pair_base() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<C8> <C9> <D83CDFD8>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.code_byte_length, 1);
|
||||
assert_eq!(cmap.lookup(0xC8), Some("🏘".to_string()));
|
||||
assert_eq!(cmap.lookup(0xC9), Some("🏙".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfrange_preserves_single_hyphen_like_base() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<21> <22> <2013>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("–".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("—".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_spaced_destination_hex_without_control_noise() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
3 beginbfchar
|
||||
<21> < 0009 000d 0020 00a0 >
|
||||
<22> < 002d 00ad 2010 >
|
||||
<23> <00a0>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("\t".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("-".to_string()));
|
||||
assert_eq!(cmap.lookup(0x23), Some("\u{00a0}".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_preserves_valid_multi_character_destinations() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
4 beginbfchar
|
||||
<21> <002d002d>
|
||||
<22> <20132013>
|
||||
<23> <002000a0>
|
||||
<24> <00660069>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("--".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("––".to_string()));
|
||||
assert_eq!(cmap.lookup(0x23), Some(" \u{00a0}".to_string()));
|
||||
assert_eq!(cmap.lookup(0x24), Some("fi".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remap_to_sequential() {
|
||||
// Simulate a broken CMap where GIDs are from pre-subsetting:
|
||||
@@ -2717,4 +2963,187 @@ endbfchar
|
||||
assert_eq!(remapped.unwrap().char_map.len(), 50);
|
||||
assert_eq!(fallback.unwrap().char_map.len(), 10);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_max_source_cid() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<0003> <0020>
|
||||
<0031> <004E>
|
||||
endbfchar
|
||||
1 beginbfrange
|
||||
<0208> <0227> <0430>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
assert_eq!(cmap.min_source_cid(), Some(0x0003));
|
||||
assert_eq!(cmap.max_source_cid(), Some(0x0227));
|
||||
}
|
||||
|
||||
/// Helper: build a minimal CIDFont dict with a W array and check coverage.
|
||||
fn cid_font_dict_with_w(w_items: Vec<lopdf::Object>) -> lopdf::Dictionary {
|
||||
let mut d = lopdf::Dictionary::new();
|
||||
d.set("W", lopdf::Object::Array(w_items));
|
||||
d
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_w_array_covers_cid_format1() {
|
||||
// Format 1: `c [w1 w2 ... wn]` — widths for CIDs c..c+n-1.
|
||||
// Mimics the 16.pdf Tahoma W array: 0[1000] 3[313] 5[401] 11[383 383] 16[363 303 382]
|
||||
let doc = Document::new();
|
||||
let d = cid_font_dict_with_w(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(1000)]),
|
||||
lopdf::Object::Integer(3),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(313)]),
|
||||
lopdf::Object::Integer(5),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(401)]),
|
||||
lopdf::Object::Integer(11),
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(383),
|
||||
lopdf::Object::Integer(383),
|
||||
]),
|
||||
lopdf::Object::Integer(16),
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(363),
|
||||
lopdf::Object::Integer(303),
|
||||
lopdf::Object::Integer(382),
|
||||
]),
|
||||
lopdf::Object::Integer(570),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(667); 26]),
|
||||
]);
|
||||
|
||||
assert!(w_array_covers_cid(&d, &doc, 0));
|
||||
assert!(w_array_covers_cid(&d, &doc, 3));
|
||||
assert!(w_array_covers_cid(&d, &doc, 5));
|
||||
assert!(w_array_covers_cid(&d, &doc, 11));
|
||||
assert!(w_array_covers_cid(&d, &doc, 12));
|
||||
assert!(w_array_covers_cid(&d, &doc, 16));
|
||||
assert!(w_array_covers_cid(&d, &doc, 18));
|
||||
assert!(w_array_covers_cid(&d, &doc, 570));
|
||||
assert!(w_array_covers_cid(&d, &doc, 595));
|
||||
// Gaps are NOT covered
|
||||
assert!(!w_array_covers_cid(&d, &doc, 1));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 4));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 19));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 596));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_w_array_covers_cid_format2() {
|
||||
// Format 2: `c_first c_last w` — CIDs c_first..c_last all have width w.
|
||||
let doc = Document::new();
|
||||
let d = cid_font_dict_with_w(vec![
|
||||
lopdf::Object::Integer(100),
|
||||
lopdf::Object::Integer(120),
|
||||
lopdf::Object::Integer(500),
|
||||
]);
|
||||
|
||||
assert!(w_array_covers_cid(&d, &doc, 100));
|
||||
assert!(w_array_covers_cid(&d, &doc, 110));
|
||||
assert!(w_array_covers_cid(&d, &doc, 120));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 99));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 121));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_w_array_covers_cid_missing_w() {
|
||||
let doc = Document::new();
|
||||
let d = lopdf::Dictionary::new();
|
||||
assert!(!w_array_covers_cid(&d, &doc, 3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_try_remap_skipped_when_w_covers_cmap() {
|
||||
// Simulates 16.pdf: CMap's max source CID (0x0279 = 633) is explicitly
|
||||
// in the W array, so no subset-renumbering happened — remap must NOT fire.
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<0003> <0020>
|
||||
<0031> <004E>
|
||||
endbfchar
|
||||
2 beginbfrange
|
||||
<023A> <0253> <0410>
|
||||
<0255> <0279> <042B>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
let mut doc = Document::new();
|
||||
// Build a CIDFont dict with Identity CIDToGIDMap and a W array that
|
||||
// covers CID 633 via `597 [widths...]`.
|
||||
let mut cid_font = lopdf::Dictionary::new();
|
||||
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
|
||||
cid_font.set(
|
||||
"W",
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(750)]),
|
||||
lopdf::Object::Integer(597),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 37]), // 597..633
|
||||
]),
|
||||
);
|
||||
let cid_font_id = doc.add_object(cid_font);
|
||||
|
||||
// Build the Type0 font dict with Identity-H + DescendantFonts ref.
|
||||
let mut font_dict = lopdf::Dictionary::new();
|
||||
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
|
||||
font_dict.set(
|
||||
"DescendantFonts",
|
||||
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
|
||||
);
|
||||
|
||||
let (primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 123);
|
||||
assert!(
|
||||
remapped.is_none(),
|
||||
"Remap must be skipped when W covers CMap max CID (this is 16.pdf)"
|
||||
);
|
||||
assert_eq!(primary.lookup(0x0003), Some(" ".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_try_remap_fires_for_true_subset_mismatch() {
|
||||
// True mismatch: CMap has high CIDs (512-544) but W only lists low sequential CIDs.
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<0200> <0220> <0410>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
let mut doc = Document::new();
|
||||
let mut cid_font = lopdf::Dictionary::new();
|
||||
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
|
||||
cid_font.set(
|
||||
"W",
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 34]), // 0..33
|
||||
]),
|
||||
);
|
||||
let cid_font_id = doc.add_object(cid_font);
|
||||
|
||||
let mut font_dict = lopdf::Dictionary::new();
|
||||
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
|
||||
font_dict.set(
|
||||
"DescendantFonts",
|
||||
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
|
||||
);
|
||||
|
||||
let (_primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 456);
|
||||
assert!(
|
||||
remapped.is_some(),
|
||||
"Remap must fire when CMap's CIDs are outside W array coverage"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+36
-7
@@ -116,6 +116,14 @@ pub struct TextItem {
|
||||
pub is_bold: bool,
|
||||
/// Whether the font is italic
|
||||
pub is_italic: bool,
|
||||
/// Whether the text is underlined (drawn rule/thin rect under the
|
||||
/// baseline — PDFs have no underline font flag, so this is detected
|
||||
/// geometrically after extraction; see `extractor::underline`).
|
||||
pub is_underline: bool,
|
||||
/// Whether the text is struck out (drawn rule/thin rect crossing the
|
||||
/// glyphs at mid x-height). Same geometric detection as underline,
|
||||
/// different vertical window; see `extractor::underline`.
|
||||
pub is_strikeout: bool,
|
||||
/// Type of item (text, image, link)
|
||||
pub item_type: ItemType,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator.
|
||||
@@ -137,12 +145,17 @@ pub struct TextLine {
|
||||
|
||||
impl TextLine {
|
||||
pub fn text(&self) -> String {
|
||||
self.text_with_formatting(false, false)
|
||||
self.text_with_formatting(false, false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic markdown formatting
|
||||
pub fn text_with_formatting(&self, format_bold: bool, format_italic: bool) -> String {
|
||||
if !format_bold && !format_italic {
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_underline: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -151,6 +164,7 @@ impl TextLine {
|
||||
let mut result = String::new();
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -176,9 +190,13 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes
|
||||
let item_bold = format_bold && item.is_bold;
|
||||
let item_italic = format_italic && item.is_italic;
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -189,6 +207,10 @@ impl TextLine {
|
||||
result.push_str("**");
|
||||
current_bold = false;
|
||||
}
|
||||
if current_underline && !item_underline {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -196,6 +218,10 @@ impl TextLine {
|
||||
}
|
||||
|
||||
// Open new styles
|
||||
if item_underline && !current_underline {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -215,6 +241,9 @@ impl TextLine {
|
||||
if current_bold {
|
||||
result.push_str("**");
|
||||
}
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+2315
-4
File diff suppressed because it is too large
Load Diff
@@ -188,7 +188,7 @@
|
||||
|156|23/7|General renovation works to Block B at Belonie Secondary School|MOE|Belvedere Builders|SR869,505.75|
|
||||
|157|30/7|Procurement of Engine Block and Crankshaft for Engine A11|PUC|Ras Tek Pvt Ltd|Euro798,650.00|
|
||||
|158|30/7|procurement of Wartsila Engine spares|PUC|Wartsila Eastern Africa ltd|Euro158,424.00|
|
||||
|159|30/7|Proposed walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|
||||
|159|30/7|Proposed walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|
||||
|160|30/7|Procurement of transfer pump control panel|PUC|CA Engineering Consultancy Pte Ltd|SGD14,600.00|
|
||||
|161|30/7|Consultancy service for North to South Victoria Bye- Pass road and utilities organisation|MLUH|Sonnel Seychelles LTD|SR1,332,000.00|
|
||||
|162 AUG|30/7|Procurement of the supply of sodium cardonate|PUC|HPL Chemical LTD|USD42,600.00|
|
||||
@@ -237,7 +237,7 @@
|
||||
|201|24/9|Procurement of vehicle x 2|SLTA|Abhaye Valabhji Pty Ltd|SR1000.000.00|
|
||||
||OCT|||||
|
||||
|202|1/10|Proposed new traffic lane to 5th June Avenue|SLTA|Divy Constrution|SR2,864,589.00|
|
||||
|203|1/10||Proposed Walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|
||||
|203|1/10||Proposed Walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot(Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|
||||
|204|1/10|Proposed Reconstrcution of Burnt House-Au Cap|MLUH|Furui Construction|SR946,130.00|
|
||||
|205|1/10|Variation on the project associated with the procurement of seven 100m3/day containerised plant|PUC|Tornado Group (UAE)|USD172,500.00|
|
||||
|206|1/10|Works on the breaker system at Bel Omber desalination plant|PUC|United Concrete Products (Sey)Ltd|SR1,998,993.11|
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
본 가격표는 국내 거주 중인 외국인을 위한 한국어 가격표의 비공식 번역본입니다. ※ The post-tax benefit sales price is provided for your reference only, reflecting the current tax benefits and eco-friendly vehicle individual consumption tax reductions. 본 가격표와 한국어 가격표의 내용이 상이한 경우 한국어 가격표의 내용이 우선하므로, 반드시 한국어 가격표의 내용을 확인하십시오. The final sales price may vary depending on the addition of optional items and whether the eco-friendly vehicle criteria are met, so please be sure to check the quotation. This price list is an unofficial translation of the Korean price list for the convenience of foreign residents in South Korea. ※ Please check the Korean price list for information on colors, details, and fuel consumption for each model. If the price list differs from the Korean price list, please check the contents of the Korean price list first. ※ All optional item prices are listed based on pre-tax reduction amounts. The actual sales price, which reflects the total individual consumption tax reduction including optional items, may differ depending on applicable tax benefits. ※ The items (specifications, colors, etc.) and prices listed in this pricing table are subject to change without prior notice depending on the holding of new car launch events, improvements made in automobile performance, introduction of related laws and regulations, and changes in company circumstances. The all-new NEXO Release Date: June 10, 2025 / (Unit: KRW)
|
||||
|
||||
|Classification Exclusive Exclusive|Selling price before tax benefit Supply value(surtax) 80,509,000 73,190,000(7,319,000)|Selling price after tax benefit 76,435,000|Standard equipment • Powertrain/Performance: Fuel cell system(150kW drive motor, lithium-ion battery, and reducer), Regenerative braking system, Column-Type Shift By Wire(vibration warning), Drive mode select • Safety: 9 airbag system(1st-row advanced/center side airbags, 1st/2nd-row side airbags, and rollover-resistant curtain airbags), Multi-Collision Brake System, Active hood system(for pedestrian protection), Safety unlock function, Artificial engine sound(for pedestrian protection), Child seat fastening system (2 in 2nd-row), Fire extinguisher for vehicles, Pedal Misapplication Safety Assist • Smart Safety Technology: Forward Collision-avoidance Assist(vehicles/ pedestrians/two-wheeled vehicles/junction turning/front oncoming), Smart Cruise Control with Stop & Go, Lane Keeping Assist, Lane Following Assist 2, Blind-spot Collision Warning(driving), Blind-spot Collision-avoidance Assist(forward exit), Rear Cross-traffic Collision-avoidance Assist, Safety Exit Assist, Driver Attention Warning, High Beam Assist, Advanced Rear Occupant Alert, Intelligent Speed Limit Assist, Hands-On Detection, Highway Driving Assist, Navigation-based Smart Cruise Control(safety speed zone/curve control), Vibration warning steering wheel • Exterior: Full LED headlamps(projection type), LED turn signal lamps (front and rear), LED Daytime Running Lights, LED positioning lights, LED rear combination lamps, LED third brake lights, 18-inch alloy wheels & tires, Solar glass(windshield), Double-glazed soundproof glass(windshield, and 1st/ 2nd-row doors), Outside mirror(heating, power-folding, power adjustment,|Options (before tax benefit) ▶ Hi-pass(e hi-pass) [200,000]|
|
||||
|Classification|Selling price before tax benefit Supply value(surtax)|Selling price after tax benefit|Standard equipment|Options (before tax benefit)|
|
||||
|---|---|---|---|---|
|
||||
|Special|with 3.5% individual consumption tax applied 79,287,000 83,500,000 75,909,091(7,590,909) with 3.5% individual consumption tax applied 82,232,000|with 3.5% individual consumption tax applied 76,435,000 79,275,000 with 3.5% individual consumption tax applied 79,275,000|and LED turn signal lamps), Auto flush door handles, Black door garnish • Interior: Panoramic curved display, 12.3-inch color LCD cluster, Leather- upholstered steering wheel(with heating, two-tone color, and Interactive Pixel Lights), LED interior lamp (map lamp, personal lamp, sun visor lamp, and luggage lamp), Metallic door scuff plate • Seat: Synthetic leather seats, 1st-row manual seats, Heated 1st-row seats, 2nd-row 60/40-split folding seats(reclining) • Convenience: Proximity key with push-button start, Smart key remote start, Electronic Parking Brake(with automatic vehicle hold), Paddle shift(regenerative control), Dual-zone full automatic air conditioning(with high-performance antibacterial combination filter, auto defog system, fine dust sensor, air cleaning mode, and after-blow function), 2nd-row seat air vent, Auto light control system, USB Type-C Ports(1×27W switchable charging/data port in 1st-row, and 2×100W charging ports in both 1st and 2nd-row), ECM room mirror(frameless), Rain sensor, Power windows with pinch protection(1st/2nd-row), Power outlet (1 in 1st-row), Parking Distance Warning-Forward/Reverse, Rear View Monitor, Wireless phone charger(single), Walk-away lock, Route planner, Hyundai AI Assistant • Infotainment: 12.3-inch navigation(Bluelink, phone projection, Bluetooth hands-free, and In-car Payment), Audio system(6 speakers), Over-The-Air navigation updates ▶ Standard equipment of Exclusive plus • Smart Safety Technology: Forward Collision-avoidance Assist(intersection crossing/changing lanes in oncoming traffic/approaching from either side/ evasive steering assist), Highway Driving Assist 2, Navigation-based Smart Cruise Control(access road) • Exterior: Roof rack • Interior: Metallic pedal, Driving mode-dependent ambient mood lighting(crash pad, 1st/2nd-row door trim) • Seat: Synthetic leather seats(patch applied), Power-adjustable driver's|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [950,000] Parking Assist ▶ [1,150,000] Audio by BANG & OLUFSEN|
|
||||
|Prestige|87,893,000 79,902,727(7,990,273) with 3.5% individual consumption tax applied 86,559,000|83,445,000 with 3.5% individual consumption tax applied 83,445,000|seat(8-way, lumbar support, and Integrated Memory System(driver's seat and outside mirror connected)), Power-adjustable front passenger’s seat(8-way), Ventilated 1st-row seats, Heated 2nd-row seats • Convenience: Hi-pass(e hi-pass), In-car fingerprint authentication system(personalization, startup, payment, and etc.), Smart power tailgate ▶ Standard equipment of Exclusive Special plus • Smart Safety Technology: Remote Smart Parking Assist 2, Parking Collison- avoidance Assist(front/side/rear) • Exterior: Intelligent Front-Lighting System(IFS), Dynamic welcome/escort lighting(1 type), Sequential turn signals(front and rear), Ambient lighting auto flush door handles, Two-tone door garnish, Glossy black rear diffuser • Interior: Recycled PET suede interior materials(headlining/sunvisor), Fabric upholstered crash pad • Seat: BIO-processed natural leather seats(metal patch applied, embossed design punching), Passenger's seat walk-in device, 1st-row relaxation comfort seats(leg rest included), Ventilated 2nd-row seats • Convenience: Parking Distance Warning-Side, Head-Up Display, Digital key 2, Wireless phone charger(dual), Surround View Monitor, Blind-spot View Monitor, LED reverse light guide • Infotainment: Audio by BANG & OLUFSEN sound system(14 speakers, including external amp), Active road noise control, Active Sound Design|sound system ▶ [250,000] 19-inch alloy wheels & tires ▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [900,000] Vision roof ▶ [1,380,000] Digital side mirror ▶ [750,000] Camera package ▶ [250,000] 19-inch alloy wheels & tires|
|
||||
|Exclusive|80,509,000 73,190,000(7,319,000) with 3.5% individual consumption tax applied 79,287,000|76,435,000 with 3.5% individual consumption tax applied 76,435,000|• Powertrain/Performance: Fuel cell system(150kW drive motor, lithium-ion battery, and reducer), Regenerative braking system, Column-Type Shift By Wire(vibration warning), Drive mode select • Safety: 9 airbag system(1st-row advanced/center side airbags, 1st/2nd-row side airbags, and rollover-resistant curtain airbags), Multi-Collision Brake System, Active hood system(for pedestrian protection), Safety unlock function, Artificial engine sound(for pedestrian protection), Child seat fastening system (2 in 2nd-row), Fire extinguisher for vehicles, Pedal Misapplication Safety Assist • Smart Safety Technology: Forward Collision-avoidance Assist(vehicles/ pedestrians/two-wheeled vehicles/junction turning/front oncoming), Smart Cruise Control with Stop & Go, Lane Keeping Assist, Lane Following Assist 2, Blind-spot Collision Warning(driving), Blind-spot Collision-avoidance Assist(forward exit), Rear Cross-traffic Collision-avoidance Assist, Safety Exit Assist, Driver Attention Warning, High Beam Assist, Advanced Rear Occupant Alert, Intelligent Speed Limit Assist, Hands-On Detection, Highway Driving Assist, Navigation-based Smart Cruise Control(safety speed zone/curve control), Vibration warning steering wheel • Exterior: Full LED headlamps(projection type), LED turn signal lamps (front and rear), LED Daytime Running Lights, LED positioning lights, LED rear combination lamps, LED third brake lights, 18-inch alloy wheels & tires, Solar glass(windshield), Double-glazed soundproof glass(windshield, and 1st/ 2nd-row doors), Outside mirror(heating, power-folding, power adjustment, and LED turn signal lamps), Auto flush door handles, Black door garnish • Interior: Panoramic curved display, 12.3-inch color LCD cluster, Leather- upholstered steering wheel(with heating, two-tone color, and Interactive Pixel Lights), LED interior lamp (map lamp, personal lamp, sun visor lamp, and luggage lamp), Metallic door scuff plate • Seat: Synthetic leather seats, 1st-row manual seats, Heated 1st-row seats, 2nd-row 60/40-split folding seats(reclining) • Convenience: Proximity key with push-button start, Smart key remote start, Electronic Parking Brake(with automatic vehicle hold), Paddle shift(regenerative control), Dual-zone full automatic air conditioning(with high-performance antibacterial combination filter, auto defog system, fine dust sensor, air cleaning mode, and after-blow function), 2nd-row seat air vent, Auto light control system, USB Type-C Ports(1×27W switchable charging/data port in 1st-row, and 2×100W charging ports in both 1st and 2nd-row), ECM room mirror(frameless), Rain sensor, Power windows with pinch protection(1st/2nd-row), Power outlet (1 in 1st-row), Parking Distance Warning-Forward/Reverse, Rear View Monitor, Wireless phone charger(single), Walk-away lock, Route planner, Hyundai AI Assistant • Infotainment: 12.3-inch navigation(Bluelink, phone projection, Bluetooth hands-free, and In-car Payment), Audio system(6 speakers), Over-The-Air navigation updates|▶ Hi-pass(e hi-pass) [200,000]|
|
||||
|Exclusive Special|83,500,000 75,909,091(7,590,909) with 3.5% individual consumption tax applied 82,232,000|79,275,000 with 3.5% individual consumption tax applied 79,275,000|▶ Standard equipment of Exclusive plus • Smart Safety Technology: Forward Collision-avoidance Assist(intersection crossing/changing lanes in oncoming traffic/approaching from either side/ evasive steering assist), Highway Driving Assist 2, Navigation-based Smart Cruise Control(access road) • Exterior: Roof rack • Interior: Metallic pedal, Driving mode-dependent ambient mood lighting(crash pad, 1st/2nd-row door trim) • Seat: Synthetic leather seats(patch applied), Power-adjustable driver's seat(8-way, lumbar support, and Integrated Memory System(driver's seat and outside mirror connected)), Power-adjustable front passenger’s seat(8-way), Ventilated 1st-row seats, Heated 2nd-row seats • Convenience: Hi-pass(e hi-pass), In-car fingerprint authentication system(personalization, startup, payment, and etc.), Smart power tailgate|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [950,000] Parking Assist ▶ [1,150,000] Audio by BANG & OLUFSEN sound system ▶ [250,000] 19-inch alloy wheels & tires|
|
||||
|Prestige|87,893,000 79,902,727(7,990,273) with 3.5% individual consumption tax applied 86,559,000|83,445,000 with 3.5% individual consumption tax applied 83,445,000|▶ Standard equipment of Exclusive Special plus • Smart Safety Technology: Remote Smart Parking Assist 2, Parking Collison- avoidance Assist(front/side/rear) • Exterior: Intelligent Front-Lighting System(IFS), Dynamic welcome/escort lighting(1 type), Sequential turn signals(front and rear), Ambient lighting auto flush door handles, Two-tone door garnish, Glossy black rear diffuser • Interior: Recycled PET suede interior materials(headlining/sunvisor), Fabric upholstered crash pad • Seat: BIO-processed natural leather seats(metal patch applied, embossed design punching), Passenger's seat walk-in device, 1st-row relaxation comfort seats(leg rest included), Ventilated 2nd-row seats • Convenience: Parking Distance Warning-Side, Head-Up Display, Digital key 2, Wireless phone charger(dual), Surround View Monitor, Blind-spot View Monitor, LED reverse light guide • Infotainment: Audio by BANG & OLUFSEN sound system(14 speakers, including external amp), Active road noise control, Active Sound Design|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [900,000] Vision roof ▶ [1,380,000] Digital side mirror ▶ [750,000] Camera package ▶ [250,000] 19-inch alloy wheels & tires|
|
||||
|
||||
**Classification Details** **Indoor/outdoor V2L** Indoor V2L, Outdoor V2L(connectorless type) **Parking Assist** Surround View Monitor, Blind-spot View Monitor, Parking Distance Warning-Side, Parking Collison-avoidance Assist-Rear **Audio by BANG & OLUFSEN** Audio by BANG & OLUFSEN sound system(14 speakers, including external amp.), Active road noise control, Active Sound Design **sound system** **Camera package** Digital center mirror(with camera sensor cleaning system), Driver monitoring system THE ALL-NEW NEXO /// ECO-FRIENDLY CAR
|
||||
|
||||
|
||||
@@ -8,7 +8,9 @@ Department of the Treasury **Internal Revenue Service**
|
||||
|
||||
# and Report to Employer
|
||||
|
||||
**This publication contains:** **Form 4070A, Employee’s Daily Record of** Tips **Form 4070, Employee’s Report of Tips to** Employer
|
||||
### This publication contains:
|
||||
|
||||
**Form 4070A,** Employee’s Daily Record of Tips **Form 4070,** Employee’s Report of Tips to Employer
|
||||
|
||||
For the period
|
||||
|
||||
@@ -20,7 +22,7 @@ Name and address of employee
|
||||
|
||||
**Publication 1244 (Rev. 7-96)** Cat. No. 44472W
|
||||
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use Form 4070A, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—If you** receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use Form 4070, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
|
||||
*(continued on inside of back cover)*
|
||||
|
||||
@@ -28,14 +30,14 @@ Form **4070A** Employee’s Daily Record of Tips (Rev. July 1996) **This is a vo
|
||||
|
||||
Establishment name (if different)
|
||||
|
||||
Date Date **a. Tips received**
|
||||
Date Date **a.** Tips received
|
||||
|
||||
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
|
||||
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
|
||||
tips of directly from customers received other employees paid tips rec’d. entry and other employees 1 2 3 4 5 **Subtotals** **For Paperwork Reduction Act Notice, see Instructions on the back of Form 4070. Page 1**
|
||||
|
||||
Date Date **a. Tips received**
|
||||
Date Date **a.** Tips received
|
||||
|
||||
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
|
||||
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
|
||||
tips of directly from customers received other employees paid tips rec’d. entry and other employees
|
||||
|
||||
7 8 9 10 11 12 13 14 15 **Subtotals**
|
||||
@@ -48,9 +50,9 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
|
||||
27 28 29 30 31 **Subtotals** **from pages** **1, 2, and 3** **Totals**
|
||||
|
||||
**1.** Report total cash tips (col. a) on Form 4070, line 1.
|
||||
**2.** Report total credit card tips (col. b) on Form 4070, line 2.
|
||||
**3.** Report total tips paid out (col. c) on Form 4070, line 3. **Page 4**
|
||||
**1.** Report total cash tips (col. **a**) on Form 4070, line **1.**
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
|
||||
@@ -64,17 +66,17 @@ Employer’s name and address (include establishment name, if different) **1** C
|
||||
|
||||
**3** Tips paid out
|
||||
|
||||
Month or shorter period in which tips were received **4** Net tips (lines 1 + 2 - 3) from, 19, to, 19 Signature Date
|
||||
Month or shorter period in which tips were received **4** Net tips (lines **1 + 2 - 3**) from, 19, to, 19 Signature Date
|
||||
|
||||
**Paperwork Reduction Act Notice.—We ask for the** information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: Recordkeeping—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; Learning **about the law—each form, 2 min.; Preparing Form 4070,** 13 min.; Form 4070A, 55 min.; and Copying and **providing Form 4070, 10 min.; Form 4070A, 14 min.** If you have comments concerning the accuracy of these time estimates or suggestions for making these
|
||||
**Paperwork Reduction Act Notice.—**We ask for the information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: **Recordkeeping**—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; **Learning** **about the law**—each form, 2 min.; **Preparing** Form 4070, 13 min.; Form 4070A, 55 min.; and **Copying and** **providing** Form 4070, 10 min.; Form 4070A, 14 min. If you have comments concerning the accuracy of these time estimates or suggestions for making these
|
||||
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—Use this form to report tips you receive to** your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See Pub. 531, Reporting Tip Income, for more information. You can get additional copies of Pub. 1244, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—**Use this form to report tips you receive to your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See **Pub. 531**, Reporting Tip Income, for more information. You can get additional copies of **Pub. 1244**, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
|
||||
**Instructions (continued)**
|
||||
**Instructions** *(continued)*
|
||||
|
||||
**Unreported Tips.—If you received tips of $20 or** more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you must use Form 1040 and Form 4137, Social Security and Medicare Tax on Unreported Tip Income, to report them. You may not use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act cannot use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—Get Pub. 531, Reporting** Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—If you do not keep a daily** record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
**Unreported Tips.—**If you received tips of $20 or more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you **must** use Form 1040 and **Form 4137,** Social Security and Medicare Tax on Unreported Tip Income, to report them. You may **not** use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act **cannot** use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—**Get **Pub. 531,** Reporting Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—**If you do not keep a daily record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
|
||||
**Instructions (continued)**
|
||||
**Instructions** *(continued)*
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
@@ -6,9 +6,7 @@
|
||||
|
||||
8 4 Z E L L / L U R I E R E A L E S T A T E C E N T E R
|
||||
|
||||
**Table I: Cap rate correlations**
|
||||
|
||||
**Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
**Table I:** Cap rate correlations **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
|
||||
* Based on 25 years of data for the 10-yrT & S&P DivYld; and 14 years for BBB.
|
||||
**Figure 1:** NCREIF cap rates vs. 10-yearTreasury
|
||||
@@ -22,7 +20,9 @@ R E V I E W 8 5
|
||||
|
||||
**Figure 2:** Capratespreadsover10-yearTreasury
|
||||
|
||||
**Basis Points -200** -400
|
||||
**Basis Points** -200
|
||||
|
||||
-400
|
||||
|
||||
-600
|
||||
|
||||
@@ -34,9 +34,7 @@ R E V I E W 8 5
|
||||
|
||||
1982 1986 1990 1994 1998 2002 2006
|
||||
|
||||
**Table II: Correlationsofspreadsbypropertytype**
|
||||
|
||||
**Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
**Table II:** Correlationsofspreadsbypropertytype **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
|
||||
||Multifamily|Industrial|CBD Office|
|
||||
|---|---|---|---|
|
||||
|
||||
+65
-52
@@ -1,8 +1,8 @@
|
||||
(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows:
|
||||
§1.1563-3T Rules for determining stock ownership (temporary).
|
||||
<u>§1.1563-3T Rules for determining stock ownership (temporary)</u>.
|
||||
|
||||
(a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)
|
||||
through (d)(2)(iii). (iv) Statement. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
|
||||
through (d)(2)(iii). (iv) <u>Statement</u>. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
|
||||
|
||||
(A) A description of each of the controlled groups in which the corporation
|
||||
could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and
|
||||
@@ -10,11 +10,13 @@ could be included. The description must include the name and employer identifica
|
||||
(B) The following representation: [INSERT NAME AND EMPLOYER
|
||||
IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].
|
||||
|
||||
(v) Election-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of
|
||||
(v) <u>Election</u>-- (A) <u>Election filed</u>. An election filed under paragraph (d)(2)(iv) of
|
||||
this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in
|
||||
|
||||
|termination of membership in the controlled group in which such corporation has been included. (B) Election not filed.|In the event no election is filed in accordance with the|
|
||||
|termination of membership in the controlled group in which such corporation has||
|
||||
|---|---|
|
||||
|been included.||
|
||||
|(B) Election not filed.|In the event no election is filed in accordance with the|
|
||||
|provisions of paragraph (d)(2)(iv) of this section, then the Internal Revenue Service||
|
||||
|will determine the group in which such corporation is to be included. Such||
|
||||
|determination will be binding for all subsequent years unless the corporation files a||
|
||||
@@ -28,42 +30,47 @@ Federal income tax return (including any amended return filed on or before the d
|
||||
|
||||
2006.
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: §1.6012-2 Corporations required to make returns of income.
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: <u>§1.6012-2 Corporations required to make returns of income</u>.
|
||||
* * * * *
|
||||
(c) [Reserved]. For further guidance, see §1.6012-2T(c).
|
||||
* * * * *
|
||||
(k) [Reserved]. For further guidance, see §1.6012-2T(k)(1).
|
||||
|
||||
Par. 52. Section 1.6012-2T is added to read as follows: §1.6012-2T Corporations required to make returns of income (temporary).
|
||||
Par. 52. Section 1.6012-2T is added to read as follows: <u>§1.6012-2T Corporations required to make returns of income (temporary)</u>.
|
||||
|
||||
(a) through (b) [Reserved]. For further guidance, see §1.6012-2(a) through
|
||||
(b).
|
||||
(c) Insurance companies-- (1) Domestic life insurance companies-- (i) In
|
||||
general. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
|
||||
<u>general</u>. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
|
||||
|
||||
(A) A copy of its annual statement which shows the reserves used by the
|
||||
company in computing the taxable income reported on its return; and
|
||||
|
||||
(B) A copy of Schedule A (real estate) and of Schedule D (bonds and stocks),
|
||||
or any successor thereto, of such annual statement. (ii) Mutual savings banks. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
|
||||
or any successor thereto, of such annual statement. (ii) <u>Mutual savings banks</u>. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
|
||||
|
||||
(2) Domestic nonlife insurance companies. Every domestic insurance
|
||||
(2) <u>Domestic nonlife insurance companies</u>. Every domestic insurance
|
||||
company other than a life insurance company shall make a return on Form 1120PC. This includes organizations described in section 501(m)(1) that provide commercial- type insurance and organizations described in section 833. Except as provided in paragraph (c)(4) of this section, such company shall file with its return a copy of its
|
||||
|
||||
annual statement (or a pro forma annual statement), including the underwriting and investment exhibit for the year covered by such return.
|
||||
|
||||
||(3) Foreign insurance companies. The provisions of paragraphs (c)(1) and|
|
||||
|---|---|
|
||||
||(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company. (4) Exception for insurance companies filing their Federal income tax returns electronically. If an insurance company described in paragraph (c)(1), (c)(2), or (c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e). (5) Definition. For purposes of this section, the term annual statement means the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of|
|
||||
(3) <u>Foreign insurance companies</u>. The provisions of paragraphs (c)(1) and
|
||||
(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company.
|
||||
(4) <u>Exception for insurance companies filing their Federal income tax returns</u>
|
||||
<u>electronically</u>. If an insurance company described in paragraph (c)(1), (c)(2), or
|
||||
|
||||
(c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e).
|
||||
(5) <u>Definition</u>. For purposes of this section, the term <u>annual statement</u> means
|
||||
the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of
|
||||
|
||||
Columbia. The term annual statement also includes a pro forma annual statement if the insurance company is not required to file the NAIC annual statement.
|
||||
|
||||
(d) through (j) [Reserved]. For further guidance, see §1.6012-2(d) through (j).
|
||||
(k) Effective date-- (1) Applicability date. This section applies to any original
|
||||
(k) <u>Effective date</u>-- (1) <u>Applicability date</u>. This section applies to any original
|
||||
Federal income tax return (including any amended return filed on or before the due date (including extensions) of such original return) timely filed on or after May 30,
|
||||
|
||||
2006.
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
(2) <u>Expiration date</u>. The applicability of this section will expire on May 26,
|
||||
2009.
|
||||
|
||||
|||Par. 53. For each entry in the “Location” column of the following table,|
|
||||
@@ -101,45 +108,32 @@ section and paragraph
|
||||
(c)(4), and (c)(5) of this section, and paragraph
|
||||
(c)(2) of §1.382-8T
|
||||
|
||||
||§1.382-8(g), Example|
|
||||
|---|---|
|
||||
||The first sentence of §1.382-8(g), Example §1.382-8(g), Example §1.382-8(g), Example|
|
||||
|
||||
(2)(c)
|
||||
(2)(e)
|
||||
(3)(b)
|
||||
(3)(c)(1)(B)
|
||||
|
||||
||The second sentence of|
|
||||
|---|---|
|
||||
||§1.382-8(g), Example The second sentence of §1.382-8(g), Example The first sentence of §1.1502-32(b)(4)(v)(A) The first sentence of §1.1502-32(b)(4)(v)(B)|
|
||||
|
||||
(4)(c)
|
||||
(5)(c)
|
||||
|
||||
|The fifth sentence of|paragraph (c) of this|paragraphs (c)(1), (c)(3),|
|
||||
|---|---|---|
|
||||
|§1.382-8(f)|section|(c)(4), and (c)(5) of this section, and paragraph (c)(2) of §1.382-8T|
|
||||
|§1.382-8(g), Example|paragraph (c) of this|paragraphs (c)(1), (c)(3), section, and paragraph (c)(2) of §1.382-8T|
|
||||
|The second sentence of|paragraph (c) of this|paragraphs (c)(1), (c)(3),|
|
||||
|§1.382-8(g), Example|section|(c)(4), and (c)(5) of this|
|
||||
|§1.382-8(g), Example|section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraphs (c)(1) and (2) of this section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraph (b)(4)(iv) of this section paragraph (b)(4)(iv) of this section|(c)(4), and (c)(5) of this (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(1) of this section and paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (b)(4)(iv) of §1.1502-32T paragraph (b)(4)(iv) of §1.1502-32T|
|
||||
|
||||
(1)(b)(2) section (c)(4), and (c)(5) of this
|
||||
(1)(c) section, and paragraph
|
||||
|
||||
(c)(2) of §1.382-8T
|
||||
|
||||
|§1.382-8(g), Example|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|---|---|---|
|
||||
|The first sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|§1.382-8(g), Example|section|§1.382-8T|
|
||||
|§1.382-8(g), Example|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|§1.382-8(g), Example|paragraphs (c)(1) and (2)|paragraph (c)(1) of this|
|
||||
|
||||
(2)(c) section §1.382-8T
|
||||
(2)(e)
|
||||
(3)(b) section §1.382-8T
|
||||
(3)(c)(1)(B) of this section section and paragraph
|
||||
|
||||
(c)(2) of §1.382-8T
|
||||
|
||||
|The second sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|---|---|---|
|
||||
|§1.382-8(g), Example|section|§1.382-8T|
|
||||
|The second sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|§1.382-8(g), Example|section|§1.382-8T|
|
||||
|The first sentence of|paragraph (b)(4)(iv) of|paragraph (b)(4)(iv) of|
|
||||
|§1.1502-32(b)(4)(v)(A)|this section|§1.1502-32T|
|
||||
|The first sentence of|paragraph (b)(4)(iv) of|paragraph (b)(4)(iv) of|
|
||||
|§1.1502-32(b)(4)(v)(B)|this section|§1.1502-32T|
|
||||
|
||||
(4)(c)
|
||||
(5)(c)
|
||||
|
||||
|§1.1502-35(c)(4)(ii)(B)|§1.1502-76(b)(2)(ii)(D)|§1.1502-76T(b)(2)(ii)(D)|
|
||||
|---|---|---|
|
||||
|§1.1502-76(b)(2)(ii)(A)(2)|paragraph (b)(2)(ii)(D) of this section|paragraph (b)(2)(ii)(D) of §1.1502-76T|
|
||||
@@ -168,14 +162,33 @@ section and paragraph
|
||||
|§1.6043-2(a)|or 1.1081-11|3T(a), or §1.1081-11T|
|
||||
|The first sentence of §301.6011-5T(a) (twice)|§1.6012-2|paragraphs (a), (b) and (d) through (j) of §1.6012- 2, and paragraph (c) of §1.6012-2T|
|
||||
|
||||
|||PART 602--OMB CONTROL NUMBERS UNDER THE PAPERWORK||
|
||||
|---|---|---|---|
|
||||
||REDUCTION ACT Authority: 26 U.S.C. 7805. 1. The following entries to the table are removed: §602.101 OMB Control numbers.|Par. 54. The authority citation for part 602 continues to read as follows: Par. 55. In §602.101, paragraph (b) is amended to read as follows:||
|
||||
|* * * * *|(b) * * * CFR part or section where identified or described||Current OMB control No.|
|
||||
|* * * * *|1.332-6………………………………………………………………….|1.382-11……………………………………………………………….. 1545-2019 1.351-3…………………………………………………………………. 1545-2019 1.355-5…………………………………………………………………. 1545-2019 1.368-3…………………………………………………………………. 1545-2019 1.1081-11………………………………………………………………. 1545-2019|1545-2019|
|
||||
|* * * * *|§602.101 OMB Control numbers.|______________________________________________________________ 2. The following entries are added in numerical order to the table:||
|
||||
|* * * * *|(b) * * * CFR part or section where identified or described||Current OMB control No.|
|
||||
|* * * * *|1.302-2T………………………………………………………………… 1545 1.302-4T………………………………………………………………… 1545||-2019 -2019|
|
||||
PART 602--OMB CONTROL NUMBERS UNDER THE PAPERWORK REDUCTION ACT Par. 54. The authority citation for part 602 continues to read as follows: Authority: 26 U.S.C. 7805. Par. 55. In §602.101, paragraph (b) is amended to read as follows:
|
||||
|
||||
1. The following entries to the table are removed:
|
||||
<u>§602.101 OMB Control numbers</u>.
|
||||
|
||||
* * * * *
|
||||
(b) * * *
|
||||
CFR part or section where Current OMB identified or described control No.
|
||||
|
||||
* * * * *
|
||||
1.332-6…………………………………………………………………. 1545-2019
|
||||
1.382-11……………………………………………………………….. 1545-2019
|
||||
1.351-3…………………………………………………………………. 1545-2019
|
||||
1.355-5…………………………………………………………………. 1545-2019
|
||||
1.368-3…………………………………………………………………. 1545-2019
|
||||
1.1081-11………………………………………………………………. 1545-2019
|
||||
* * * * * **______________________________________________________________**
|
||||
2. The following entries are added in numerical order to the table:
|
||||
<u>§602.101 OMB Control numbers</u>.
|
||||
|
||||
* * * * *
|
||||
(b) * * *
|
||||
CFR part or section where Current OMB identified or described control No.
|
||||
|
||||
* * * * *
|
||||
1.302-2T………………………………………………………………… 1545-2019
|
||||
1.302-4T………………………………………………………………… 1545-2019
|
||||
|
||||
|1.331-1T………………………………………………………………… 1545|-2019|
|
||||
|---|---|
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
**Technical Information**
|
||||
##### Technical Information
|
||||
|
||||
## l T-12 SI
|
||||
|
||||
DuPont Fluorochemicals
|
||||
##### DuPont Fluorochemicals
|
||||
|
||||
#### Thermodynamic Properties
|
||||
|
||||
@@ -20,25 +20,22 @@ Tables of the thermodynamic **Units** properties of R-12 have been developed and
|
||||
|
||||
S.A., Lemmon, E.W., and Peskin, Vf = Fluid (liquid) specific volume
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998). kilograms per cubic meter
|
||||
H = Enthalpy (kJ/kg)
|
||||
##### H = Enthalpy (kJ/kg)
|
||||
|
||||
S = Entropy (kJ/kg.K)
|
||||
##### S = Entropy (kJ/kg.K)
|
||||
|
||||
**Physical Properties**
|
||||
##### Physical Properties
|
||||
|
||||
Chemical Formula CCl2F2
|
||||
|Chemical Formula|CCl₂F₂|
|
||||
|---|---|
|
||||
|Molecular mass|120.91|
|
||||
|Boiling Point At one atmosphere|-29.75°C|
|
||||
|Critical Temperature|111.97°C|
|
||||
|Critical Pressure|4136 kPa|
|
||||
|Critical Density|565.0 kg/m|
|
||||
|Critical Volume|0.0018 m|
|
||||
|
||||
Molecular mass 120.91
|
||||
|
||||
Boiling Point-29.75°C At one atmosphere
|
||||
|
||||
Critical Temperature 111.97°C
|
||||
|
||||
Critical Pressure 4136 kPa
|
||||
|
||||
3 Critical Density 565.0 kg/m
|
||||
|
||||
Critical Volume 0.0018 m /kg
|
||||
/kg
|
||||
|
||||
l
|
||||
|
||||
@@ -48,7 +45,7 @@ l
|
||||
|
||||
|Temp|Pressure||Volume|||Density||Enthalpy|||Entropy|Temp|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|°C|[kPa]|[m3 Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m3] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|°C|[kPa]|[m³ Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m³] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|
||||
|-100|1.2|0.0006|10.0000|1679.0|0.100|113.3|192.8|306.1|0.6077|1.7210|-100|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
@@ -262,6 +262,85 @@ class TestExtractTextInRegions:
|
||||
assert results[0].page == 0
|
||||
assert results[1].page == 1
|
||||
|
||||
def test_malformed_region_raises_value_error(self):
|
||||
with pytest.raises(ValueError, match="Invalid region"):
|
||||
pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 600.0]])],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_pages_markdown / extract_pages_markdown_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractPagesMarkdown:
|
||||
def test_default_returns_all_pages(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert len(result.pages) == 3
|
||||
assert [p.page for p in result.pages] == [0, 1, 2]
|
||||
assert all(isinstance(p.markdown, str) for p in result.pages)
|
||||
|
||||
def test_bytes_default_returns_all_pages(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.extract_pages_markdown_bytes(data)
|
||||
assert len(result.pages) == 3
|
||||
|
||||
def test_selected_pages_preserve_order(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[2, 0]
|
||||
)
|
||||
assert [p.page for p in result.pages] == [2, 0]
|
||||
|
||||
def test_bytes_selected_pages_preserve_order(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.extract_pages_markdown_bytes(data, pages=[1])
|
||||
assert len(result.pages) == 1
|
||||
assert result.pages[0].page == 1
|
||||
|
||||
def test_page_fields(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[0]
|
||||
)
|
||||
page = result.pages[0]
|
||||
assert isinstance(page.page, int)
|
||||
assert isinstance(page.markdown, str)
|
||||
assert isinstance(page.needs_ocr, bool)
|
||||
assert not page.needs_ocr # text-based fixture
|
||||
assert len(page.markdown) > 0
|
||||
|
||||
def test_result_fields(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert isinstance(result.pages, list)
|
||||
assert isinstance(result.pages_with_tables, list)
|
||||
assert isinstance(result.pages_with_columns, list)
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
assert isinstance(result.is_complex, bool)
|
||||
|
||||
def test_out_of_range_page_marks_needs_ocr(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[9999]
|
||||
)
|
||||
assert len(result.pages) == 1
|
||||
assert result.pages[0].needs_ocr
|
||||
assert result.pages[0].markdown == ""
|
||||
|
||||
def test_repr(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[0]
|
||||
)
|
||||
assert "PagesExtractionResult" in repr(result)
|
||||
assert "PageMarkdown" in repr(result.pages[0])
|
||||
|
||||
def test_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_pages_markdown_bytes(b"not a pdf")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error handling
|
||||
|
||||
Reference in New Issue
Block a user